vllm-cpu-avx512vnni 0.13.0__cp313-cp313-manylinux_2_28_x86_64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of vllm-cpu-avx512vnni might be problematic. Click here for more details.

Files changed (1641) hide show
  1. vllm/_C.abi3.so +0 -0
  2. vllm/__init__.py +225 -0
  3. vllm/_aiter_ops.py +1260 -0
  4. vllm/_bc_linter.py +54 -0
  5. vllm/_custom_ops.py +3080 -0
  6. vllm/_ipex_ops.py +457 -0
  7. vllm/_version.py +34 -0
  8. vllm/assets/__init__.py +0 -0
  9. vllm/assets/audio.py +43 -0
  10. vllm/assets/base.py +40 -0
  11. vllm/assets/image.py +59 -0
  12. vllm/assets/video.py +149 -0
  13. vllm/attention/__init__.py +0 -0
  14. vllm/attention/backends/__init__.py +0 -0
  15. vllm/attention/backends/abstract.py +443 -0
  16. vllm/attention/backends/registry.py +254 -0
  17. vllm/attention/backends/utils.py +33 -0
  18. vllm/attention/layer.py +969 -0
  19. vllm/attention/layers/__init__.py +0 -0
  20. vllm/attention/layers/chunked_local_attention.py +120 -0
  21. vllm/attention/layers/cross_attention.py +178 -0
  22. vllm/attention/layers/encoder_only_attention.py +103 -0
  23. vllm/attention/layers/mm_encoder_attention.py +284 -0
  24. vllm/attention/ops/__init__.py +0 -0
  25. vllm/attention/ops/chunked_prefill_paged_decode.py +401 -0
  26. vllm/attention/ops/common.py +469 -0
  27. vllm/attention/ops/flashmla.py +251 -0
  28. vllm/attention/ops/merge_attn_states.py +47 -0
  29. vllm/attention/ops/paged_attn.py +51 -0
  30. vllm/attention/ops/pallas_kv_cache_update.py +130 -0
  31. vllm/attention/ops/prefix_prefill.py +814 -0
  32. vllm/attention/ops/rocm_aiter_mla_sparse.py +210 -0
  33. vllm/attention/ops/triton_decode_attention.py +712 -0
  34. vllm/attention/ops/triton_merge_attn_states.py +116 -0
  35. vllm/attention/ops/triton_reshape_and_cache_flash.py +184 -0
  36. vllm/attention/ops/triton_unified_attention.py +1047 -0
  37. vllm/attention/ops/vit_attn_wrappers.py +139 -0
  38. vllm/attention/selector.py +145 -0
  39. vllm/attention/utils/__init__.py +0 -0
  40. vllm/attention/utils/fa_utils.py +118 -0
  41. vllm/attention/utils/kv_sharing_utils.py +33 -0
  42. vllm/attention/utils/kv_transfer_utils.py +60 -0
  43. vllm/beam_search.py +88 -0
  44. vllm/benchmarks/__init__.py +0 -0
  45. vllm/benchmarks/datasets.py +3228 -0
  46. vllm/benchmarks/latency.py +170 -0
  47. vllm/benchmarks/lib/__init__.py +3 -0
  48. vllm/benchmarks/lib/endpoint_request_func.py +777 -0
  49. vllm/benchmarks/lib/ready_checker.py +72 -0
  50. vllm/benchmarks/lib/utils.py +79 -0
  51. vllm/benchmarks/serve.py +1538 -0
  52. vllm/benchmarks/startup.py +326 -0
  53. vllm/benchmarks/sweep/__init__.py +0 -0
  54. vllm/benchmarks/sweep/cli.py +41 -0
  55. vllm/benchmarks/sweep/param_sweep.py +158 -0
  56. vllm/benchmarks/sweep/plot.py +675 -0
  57. vllm/benchmarks/sweep/plot_pareto.py +393 -0
  58. vllm/benchmarks/sweep/serve.py +450 -0
  59. vllm/benchmarks/sweep/serve_sla.py +492 -0
  60. vllm/benchmarks/sweep/server.py +114 -0
  61. vllm/benchmarks/sweep/sla_sweep.py +132 -0
  62. vllm/benchmarks/sweep/utils.py +4 -0
  63. vllm/benchmarks/throughput.py +808 -0
  64. vllm/collect_env.py +857 -0
  65. vllm/compilation/__init__.py +0 -0
  66. vllm/compilation/activation_quant_fusion.py +209 -0
  67. vllm/compilation/backends.py +839 -0
  68. vllm/compilation/base_static_graph.py +57 -0
  69. vllm/compilation/caching.py +180 -0
  70. vllm/compilation/collective_fusion.py +1215 -0
  71. vllm/compilation/compiler_interface.py +639 -0
  72. vllm/compilation/counter.py +48 -0
  73. vllm/compilation/cuda_graph.py +302 -0
  74. vllm/compilation/decorators.py +626 -0
  75. vllm/compilation/fix_functionalization.py +266 -0
  76. vllm/compilation/fusion.py +550 -0
  77. vllm/compilation/fusion_attn.py +359 -0
  78. vllm/compilation/fx_utils.py +91 -0
  79. vllm/compilation/inductor_pass.py +138 -0
  80. vllm/compilation/matcher_utils.py +361 -0
  81. vllm/compilation/monitor.py +62 -0
  82. vllm/compilation/noop_elimination.py +130 -0
  83. vllm/compilation/partition_rules.py +72 -0
  84. vllm/compilation/pass_manager.py +155 -0
  85. vllm/compilation/piecewise_backend.py +178 -0
  86. vllm/compilation/post_cleanup.py +21 -0
  87. vllm/compilation/qk_norm_rope_fusion.py +238 -0
  88. vllm/compilation/rocm_aiter_fusion.py +242 -0
  89. vllm/compilation/sequence_parallelism.py +364 -0
  90. vllm/compilation/torch25_custom_graph_pass.py +44 -0
  91. vllm/compilation/vllm_inductor_pass.py +173 -0
  92. vllm/compilation/wrapper.py +319 -0
  93. vllm/config/__init__.py +108 -0
  94. vllm/config/attention.py +114 -0
  95. vllm/config/cache.py +232 -0
  96. vllm/config/compilation.py +1140 -0
  97. vllm/config/device.py +75 -0
  98. vllm/config/ec_transfer.py +110 -0
  99. vllm/config/kv_events.py +56 -0
  100. vllm/config/kv_transfer.py +119 -0
  101. vllm/config/load.py +124 -0
  102. vllm/config/lora.py +96 -0
  103. vllm/config/model.py +2190 -0
  104. vllm/config/multimodal.py +247 -0
  105. vllm/config/observability.py +140 -0
  106. vllm/config/parallel.py +660 -0
  107. vllm/config/pooler.py +126 -0
  108. vllm/config/profiler.py +199 -0
  109. vllm/config/scheduler.py +299 -0
  110. vllm/config/speculative.py +644 -0
  111. vllm/config/speech_to_text.py +38 -0
  112. vllm/config/structured_outputs.py +78 -0
  113. vllm/config/utils.py +370 -0
  114. vllm/config/vllm.py +1434 -0
  115. vllm/connections.py +189 -0
  116. vllm/device_allocator/__init__.py +0 -0
  117. vllm/device_allocator/cumem.py +327 -0
  118. vllm/distributed/__init__.py +6 -0
  119. vllm/distributed/communication_op.py +43 -0
  120. vllm/distributed/device_communicators/__init__.py +0 -0
  121. vllm/distributed/device_communicators/all2all.py +490 -0
  122. vllm/distributed/device_communicators/all_reduce_utils.py +344 -0
  123. vllm/distributed/device_communicators/base_device_communicator.py +297 -0
  124. vllm/distributed/device_communicators/cpu_communicator.py +209 -0
  125. vllm/distributed/device_communicators/cuda_communicator.py +340 -0
  126. vllm/distributed/device_communicators/cuda_wrapper.py +216 -0
  127. vllm/distributed/device_communicators/custom_all_reduce.py +326 -0
  128. vllm/distributed/device_communicators/mnnvl_compat.py +27 -0
  129. vllm/distributed/device_communicators/pynccl.py +386 -0
  130. vllm/distributed/device_communicators/pynccl_allocator.py +191 -0
  131. vllm/distributed/device_communicators/pynccl_wrapper.py +564 -0
  132. vllm/distributed/device_communicators/quick_all_reduce.py +290 -0
  133. vllm/distributed/device_communicators/ray_communicator.py +259 -0
  134. vllm/distributed/device_communicators/shm_broadcast.py +778 -0
  135. vllm/distributed/device_communicators/shm_object_storage.py +697 -0
  136. vllm/distributed/device_communicators/symm_mem.py +156 -0
  137. vllm/distributed/device_communicators/tpu_communicator.py +99 -0
  138. vllm/distributed/device_communicators/xpu_communicator.py +95 -0
  139. vllm/distributed/ec_transfer/__init__.py +14 -0
  140. vllm/distributed/ec_transfer/ec_connector/__init__.py +0 -0
  141. vllm/distributed/ec_transfer/ec_connector/base.py +247 -0
  142. vllm/distributed/ec_transfer/ec_connector/example_connector.py +201 -0
  143. vllm/distributed/ec_transfer/ec_connector/factory.py +85 -0
  144. vllm/distributed/ec_transfer/ec_transfer_state.py +42 -0
  145. vllm/distributed/eplb/__init__.py +3 -0
  146. vllm/distributed/eplb/async_worker.py +115 -0
  147. vllm/distributed/eplb/eplb_state.py +1164 -0
  148. vllm/distributed/eplb/policy/__init__.py +19 -0
  149. vllm/distributed/eplb/policy/abstract.py +40 -0
  150. vllm/distributed/eplb/policy/default.py +267 -0
  151. vllm/distributed/eplb/rebalance_execute.py +529 -0
  152. vllm/distributed/kv_events.py +499 -0
  153. vllm/distributed/kv_transfer/README.md +29 -0
  154. vllm/distributed/kv_transfer/__init__.py +20 -0
  155. vllm/distributed/kv_transfer/disagg_prefill_workflow.jpg +0 -0
  156. vllm/distributed/kv_transfer/kv_connector/__init__.py +0 -0
  157. vllm/distributed/kv_transfer/kv_connector/base.py +10 -0
  158. vllm/distributed/kv_transfer/kv_connector/factory.py +197 -0
  159. vllm/distributed/kv_transfer/kv_connector/utils.py +322 -0
  160. vllm/distributed/kv_transfer/kv_connector/v1/__init__.py +19 -0
  161. vllm/distributed/kv_transfer/kv_connector/v1/base.py +597 -0
  162. vllm/distributed/kv_transfer/kv_connector/v1/decode_bench_connector.py +419 -0
  163. vllm/distributed/kv_transfer/kv_connector/v1/example_connector.py +450 -0
  164. vllm/distributed/kv_transfer/kv_connector/v1/lmcache_connector.py +327 -0
  165. vllm/distributed/kv_transfer/kv_connector/v1/lmcache_integration/__init__.py +18 -0
  166. vllm/distributed/kv_transfer/kv_connector/v1/lmcache_integration/multi_process_adapter.py +378 -0
  167. vllm/distributed/kv_transfer/kv_connector/v1/lmcache_integration/utils.py +221 -0
  168. vllm/distributed/kv_transfer/kv_connector/v1/lmcache_integration/vllm_v1_adapter.py +1418 -0
  169. vllm/distributed/kv_transfer/kv_connector/v1/lmcache_mp_connector.py +895 -0
  170. vllm/distributed/kv_transfer/kv_connector/v1/metrics.py +186 -0
  171. vllm/distributed/kv_transfer/kv_connector/v1/mooncake_connector.py +914 -0
  172. vllm/distributed/kv_transfer/kv_connector/v1/multi_connector.py +464 -0
  173. vllm/distributed/kv_transfer/kv_connector/v1/nixl_connector.py +2526 -0
  174. vllm/distributed/kv_transfer/kv_connector/v1/offloading_connector.py +538 -0
  175. vllm/distributed/kv_transfer/kv_connector/v1/p2p/__init__.py +0 -0
  176. vllm/distributed/kv_transfer/kv_connector/v1/p2p/p2p_nccl_connector.py +531 -0
  177. vllm/distributed/kv_transfer/kv_connector/v1/p2p/p2p_nccl_engine.py +632 -0
  178. vllm/distributed/kv_transfer/kv_connector/v1/p2p/tensor_memory_pool.py +273 -0
  179. vllm/distributed/kv_transfer/kv_transfer_state.py +78 -0
  180. vllm/distributed/parallel_state.py +1795 -0
  181. vllm/distributed/tpu_distributed_utils.py +188 -0
  182. vllm/distributed/utils.py +545 -0
  183. vllm/engine/__init__.py +0 -0
  184. vllm/engine/arg_utils.py +2068 -0
  185. vllm/engine/async_llm_engine.py +6 -0
  186. vllm/engine/llm_engine.py +6 -0
  187. vllm/engine/protocol.py +190 -0
  188. vllm/entrypoints/__init__.py +0 -0
  189. vllm/entrypoints/anthropic/__init__.py +0 -0
  190. vllm/entrypoints/anthropic/protocol.py +162 -0
  191. vllm/entrypoints/anthropic/serving_messages.py +468 -0
  192. vllm/entrypoints/api_server.py +185 -0
  193. vllm/entrypoints/chat_utils.py +1903 -0
  194. vllm/entrypoints/cli/__init__.py +15 -0
  195. vllm/entrypoints/cli/benchmark/__init__.py +0 -0
  196. vllm/entrypoints/cli/benchmark/base.py +25 -0
  197. vllm/entrypoints/cli/benchmark/latency.py +21 -0
  198. vllm/entrypoints/cli/benchmark/main.py +56 -0
  199. vllm/entrypoints/cli/benchmark/serve.py +21 -0
  200. vllm/entrypoints/cli/benchmark/startup.py +21 -0
  201. vllm/entrypoints/cli/benchmark/sweep.py +21 -0
  202. vllm/entrypoints/cli/benchmark/throughput.py +21 -0
  203. vllm/entrypoints/cli/collect_env.py +38 -0
  204. vllm/entrypoints/cli/main.py +79 -0
  205. vllm/entrypoints/cli/openai.py +260 -0
  206. vllm/entrypoints/cli/run_batch.py +68 -0
  207. vllm/entrypoints/cli/serve.py +249 -0
  208. vllm/entrypoints/cli/types.py +29 -0
  209. vllm/entrypoints/constants.py +12 -0
  210. vllm/entrypoints/context.py +835 -0
  211. vllm/entrypoints/launcher.py +175 -0
  212. vllm/entrypoints/llm.py +1790 -0
  213. vllm/entrypoints/logger.py +84 -0
  214. vllm/entrypoints/openai/__init__.py +0 -0
  215. vllm/entrypoints/openai/api_server.py +1469 -0
  216. vllm/entrypoints/openai/cli_args.py +302 -0
  217. vllm/entrypoints/openai/orca_metrics.py +120 -0
  218. vllm/entrypoints/openai/parser/__init__.py +0 -0
  219. vllm/entrypoints/openai/parser/harmony_utils.py +825 -0
  220. vllm/entrypoints/openai/parser/responses_parser.py +135 -0
  221. vllm/entrypoints/openai/protocol.py +2496 -0
  222. vllm/entrypoints/openai/run_batch.py +631 -0
  223. vllm/entrypoints/openai/serving_chat.py +1822 -0
  224. vllm/entrypoints/openai/serving_completion.py +729 -0
  225. vllm/entrypoints/openai/serving_engine.py +1542 -0
  226. vllm/entrypoints/openai/serving_models.py +304 -0
  227. vllm/entrypoints/openai/serving_responses.py +2080 -0
  228. vllm/entrypoints/openai/serving_transcription.py +168 -0
  229. vllm/entrypoints/openai/speech_to_text.py +559 -0
  230. vllm/entrypoints/openai/tool_parsers/__init__.py +33 -0
  231. vllm/entrypoints/openai/utils.py +49 -0
  232. vllm/entrypoints/pooling/__init__.py +16 -0
  233. vllm/entrypoints/pooling/classify/__init__.py +0 -0
  234. vllm/entrypoints/pooling/classify/api_router.py +50 -0
  235. vllm/entrypoints/pooling/classify/protocol.py +181 -0
  236. vllm/entrypoints/pooling/classify/serving.py +233 -0
  237. vllm/entrypoints/pooling/embed/__init__.py +0 -0
  238. vllm/entrypoints/pooling/embed/api_router.py +67 -0
  239. vllm/entrypoints/pooling/embed/protocol.py +208 -0
  240. vllm/entrypoints/pooling/embed/serving.py +684 -0
  241. vllm/entrypoints/pooling/pooling/__init__.py +0 -0
  242. vllm/entrypoints/pooling/pooling/api_router.py +63 -0
  243. vllm/entrypoints/pooling/pooling/protocol.py +148 -0
  244. vllm/entrypoints/pooling/pooling/serving.py +354 -0
  245. vllm/entrypoints/pooling/score/__init__.py +0 -0
  246. vllm/entrypoints/pooling/score/api_router.py +149 -0
  247. vllm/entrypoints/pooling/score/protocol.py +146 -0
  248. vllm/entrypoints/pooling/score/serving.py +508 -0
  249. vllm/entrypoints/renderer.py +410 -0
  250. vllm/entrypoints/responses_utils.py +249 -0
  251. vllm/entrypoints/sagemaker/__init__.py +4 -0
  252. vllm/entrypoints/sagemaker/routes.py +118 -0
  253. vllm/entrypoints/score_utils.py +237 -0
  254. vllm/entrypoints/serve/__init__.py +60 -0
  255. vllm/entrypoints/serve/disagg/__init__.py +0 -0
  256. vllm/entrypoints/serve/disagg/api_router.py +110 -0
  257. vllm/entrypoints/serve/disagg/protocol.py +90 -0
  258. vllm/entrypoints/serve/disagg/serving.py +285 -0
  259. vllm/entrypoints/serve/elastic_ep/__init__.py +0 -0
  260. vllm/entrypoints/serve/elastic_ep/api_router.py +96 -0
  261. vllm/entrypoints/serve/elastic_ep/middleware.py +49 -0
  262. vllm/entrypoints/serve/instrumentator/__init__.py +0 -0
  263. vllm/entrypoints/serve/instrumentator/health.py +33 -0
  264. vllm/entrypoints/serve/instrumentator/metrics.py +45 -0
  265. vllm/entrypoints/serve/lora/__init__.py +0 -0
  266. vllm/entrypoints/serve/lora/api_router.py +70 -0
  267. vllm/entrypoints/serve/profile/__init__.py +0 -0
  268. vllm/entrypoints/serve/profile/api_router.py +46 -0
  269. vllm/entrypoints/serve/rlhf/__init__.py +0 -0
  270. vllm/entrypoints/serve/rlhf/api_router.py +102 -0
  271. vllm/entrypoints/serve/sleep/__init__.py +0 -0
  272. vllm/entrypoints/serve/sleep/api_router.py +60 -0
  273. vllm/entrypoints/serve/tokenize/__init__.py +0 -0
  274. vllm/entrypoints/serve/tokenize/api_router.py +118 -0
  275. vllm/entrypoints/serve/tokenize/serving.py +204 -0
  276. vllm/entrypoints/ssl.py +78 -0
  277. vllm/entrypoints/tool.py +187 -0
  278. vllm/entrypoints/tool_server.py +234 -0
  279. vllm/entrypoints/utils.py +319 -0
  280. vllm/env_override.py +378 -0
  281. vllm/envs.py +1744 -0
  282. vllm/forward_context.py +358 -0
  283. vllm/inputs/__init__.py +44 -0
  284. vllm/inputs/data.py +359 -0
  285. vllm/inputs/parse.py +146 -0
  286. vllm/inputs/preprocess.py +717 -0
  287. vllm/logger.py +303 -0
  288. vllm/logging_utils/__init__.py +13 -0
  289. vllm/logging_utils/dump_input.py +83 -0
  290. vllm/logging_utils/formatter.py +127 -0
  291. vllm/logging_utils/lazy.py +20 -0
  292. vllm/logging_utils/log_time.py +34 -0
  293. vllm/logits_process.py +121 -0
  294. vllm/logprobs.py +206 -0
  295. vllm/lora/__init__.py +0 -0
  296. vllm/lora/layers/__init__.py +42 -0
  297. vllm/lora/layers/base.py +66 -0
  298. vllm/lora/layers/base_linear.py +165 -0
  299. vllm/lora/layers/column_parallel_linear.py +577 -0
  300. vllm/lora/layers/fused_moe.py +747 -0
  301. vllm/lora/layers/logits_processor.py +203 -0
  302. vllm/lora/layers/replicated_linear.py +70 -0
  303. vllm/lora/layers/row_parallel_linear.py +176 -0
  304. vllm/lora/layers/utils.py +74 -0
  305. vllm/lora/layers/vocal_parallel_embedding.py +140 -0
  306. vllm/lora/lora_model.py +246 -0
  307. vllm/lora/lora_weights.py +227 -0
  308. vllm/lora/model_manager.py +690 -0
  309. vllm/lora/ops/__init__.py +0 -0
  310. vllm/lora/ops/ipex_ops/__init__.py +6 -0
  311. vllm/lora/ops/ipex_ops/lora_ops.py +57 -0
  312. vllm/lora/ops/torch_ops/__init__.py +20 -0
  313. vllm/lora/ops/torch_ops/lora_ops.py +128 -0
  314. vllm/lora/ops/triton_ops/README_TUNING.md +60 -0
  315. vllm/lora/ops/triton_ops/__init__.py +21 -0
  316. vllm/lora/ops/triton_ops/fused_moe_lora_op.py +665 -0
  317. vllm/lora/ops/triton_ops/kernel_utils.py +340 -0
  318. vllm/lora/ops/triton_ops/lora_expand_op.py +310 -0
  319. vllm/lora/ops/triton_ops/lora_kernel_metadata.py +154 -0
  320. vllm/lora/ops/triton_ops/lora_shrink_op.py +287 -0
  321. vllm/lora/ops/triton_ops/utils.py +295 -0
  322. vllm/lora/ops/xla_ops/__init__.py +6 -0
  323. vllm/lora/ops/xla_ops/lora_ops.py +141 -0
  324. vllm/lora/peft_helper.py +128 -0
  325. vllm/lora/punica_wrapper/__init__.py +10 -0
  326. vllm/lora/punica_wrapper/punica_base.py +493 -0
  327. vllm/lora/punica_wrapper/punica_cpu.py +351 -0
  328. vllm/lora/punica_wrapper/punica_gpu.py +412 -0
  329. vllm/lora/punica_wrapper/punica_selector.py +21 -0
  330. vllm/lora/punica_wrapper/punica_tpu.py +358 -0
  331. vllm/lora/punica_wrapper/punica_xpu.py +276 -0
  332. vllm/lora/punica_wrapper/utils.py +150 -0
  333. vllm/lora/request.py +100 -0
  334. vllm/lora/resolver.py +88 -0
  335. vllm/lora/utils.py +315 -0
  336. vllm/lora/worker_manager.py +268 -0
  337. vllm/model_executor/__init__.py +11 -0
  338. vllm/model_executor/custom_op.py +199 -0
  339. vllm/model_executor/layers/__init__.py +0 -0
  340. vllm/model_executor/layers/activation.py +595 -0
  341. vllm/model_executor/layers/attention_layer_base.py +32 -0
  342. vllm/model_executor/layers/batch_invariant.py +1067 -0
  343. vllm/model_executor/layers/conv.py +256 -0
  344. vllm/model_executor/layers/fla/__init__.py +8 -0
  345. vllm/model_executor/layers/fla/ops/__init__.py +17 -0
  346. vllm/model_executor/layers/fla/ops/chunk.py +240 -0
  347. vllm/model_executor/layers/fla/ops/chunk_delta_h.py +344 -0
  348. vllm/model_executor/layers/fla/ops/chunk_o.py +183 -0
  349. vllm/model_executor/layers/fla/ops/chunk_scaled_dot_kkt.py +154 -0
  350. vllm/model_executor/layers/fla/ops/cumsum.py +280 -0
  351. vllm/model_executor/layers/fla/ops/fused_recurrent.py +390 -0
  352. vllm/model_executor/layers/fla/ops/index.py +41 -0
  353. vllm/model_executor/layers/fla/ops/kda.py +1351 -0
  354. vllm/model_executor/layers/fla/ops/l2norm.py +146 -0
  355. vllm/model_executor/layers/fla/ops/layernorm_guard.py +396 -0
  356. vllm/model_executor/layers/fla/ops/op.py +60 -0
  357. vllm/model_executor/layers/fla/ops/solve_tril.py +556 -0
  358. vllm/model_executor/layers/fla/ops/utils.py +194 -0
  359. vllm/model_executor/layers/fla/ops/wy_fast.py +158 -0
  360. vllm/model_executor/layers/fused_moe/__init__.py +114 -0
  361. vllm/model_executor/layers/fused_moe/all2all_utils.py +171 -0
  362. vllm/model_executor/layers/fused_moe/batched_deep_gemm_moe.py +409 -0
  363. vllm/model_executor/layers/fused_moe/config.py +1043 -0
  364. vllm/model_executor/layers/fused_moe/configs/E=1,N=14336,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +146 -0
  365. vllm/model_executor/layers/fused_moe/configs/E=1,N=14336,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  366. vllm/model_executor/layers/fused_moe/configs/E=1,N=1792,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +218 -0
  367. vllm/model_executor/layers/fused_moe/configs/E=1,N=1792,device_name=NVIDIA_A100-SXM4-80GB.json +218 -0
  368. vllm/model_executor/layers/fused_moe/configs/E=1,N=1792,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  369. vllm/model_executor/layers/fused_moe/configs/E=1,N=3072,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +218 -0
  370. vllm/model_executor/layers/fused_moe/configs/E=1,N=3072,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  371. vllm/model_executor/layers/fused_moe/configs/E=1,N=3072,device_name=NVIDIA_H100_80GB_HBM3.json +218 -0
  372. vllm/model_executor/layers/fused_moe/configs/E=1,N=3072,device_name=NVIDIA_H200,dtype=int8_w8a16.json +146 -0
  373. vllm/model_executor/layers/fused_moe/configs/E=1,N=3584,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +218 -0
  374. vllm/model_executor/layers/fused_moe/configs/E=1,N=3584,device_name=NVIDIA_A100-SXM4-80GB.json +218 -0
  375. vllm/model_executor/layers/fused_moe/configs/E=1,N=3584,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  376. vllm/model_executor/layers/fused_moe/configs/E=1,N=7168,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +218 -0
  377. vllm/model_executor/layers/fused_moe/configs/E=1,N=7168,device_name=NVIDIA_A100-SXM4-80GB.json +218 -0
  378. vllm/model_executor/layers/fused_moe/configs/E=1,N=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  379. vllm/model_executor/layers/fused_moe/configs/E=128,N=1024,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  380. vllm/model_executor/layers/fused_moe/configs/E=128,N=1024,device_name=AMD_Instinct_MI300X.json +200 -0
  381. vllm/model_executor/layers/fused_moe/configs/E=128,N=1024,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  382. vllm/model_executor/layers/fused_moe/configs/E=128,N=1024,device_name=NVIDIA_H100,dtype=fp8_w8a8.json +123 -0
  383. vllm/model_executor/layers/fused_moe/configs/E=128,N=1024,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  384. vllm/model_executor/layers/fused_moe/configs/E=128,N=1024,device_name=NVIDIA_H200.json +146 -0
  385. vllm/model_executor/layers/fused_moe/configs/E=128,N=1856,device_name=NVIDIA_H100_80GB_HBM3.json +147 -0
  386. vllm/model_executor/layers/fused_moe/configs/E=128,N=1856,device_name=NVIDIA_L40S.json +147 -0
  387. vllm/model_executor/layers/fused_moe/configs/E=128,N=192,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  388. vllm/model_executor/layers/fused_moe/configs/E=128,N=192,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  389. vllm/model_executor/layers/fused_moe/configs/E=128,N=192,device_name=NVIDIA_H20-3e.json +146 -0
  390. vllm/model_executor/layers/fused_moe/configs/E=128,N=192,device_name=NVIDIA_H20.json +146 -0
  391. vllm/model_executor/layers/fused_moe/configs/E=128,N=192,device_name=NVIDIA_H200.json +146 -0
  392. vllm/model_executor/layers/fused_moe/configs/E=128,N=352,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +122 -0
  393. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  394. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  395. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=NVIDIA_GB200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  396. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  397. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=NVIDIA_H20-3e,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  398. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=NVIDIA_H20-3e.json +146 -0
  399. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=NVIDIA_H20.json +146 -0
  400. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  401. vllm/model_executor/layers/fused_moe/configs/E=128,N=384,device_name=NVIDIA_H200.json +146 -0
  402. vllm/model_executor/layers/fused_moe/configs/E=128,N=512,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  403. vllm/model_executor/layers/fused_moe/configs/E=128,N=512,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  404. vllm/model_executor/layers/fused_moe/configs/E=128,N=704,device_name=NVIDIA_B200,dtype=fp8_w8a8.json +146 -0
  405. vllm/model_executor/layers/fused_moe/configs/E=128,N=704,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +114 -0
  406. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  407. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=AMD_Instinct_MI308X.json +213 -0
  408. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  409. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=NVIDIA_GB200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  410. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  411. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=NVIDIA_H20-3e,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  412. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=NVIDIA_H20.json +146 -0
  413. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  414. vllm/model_executor/layers/fused_moe/configs/E=128,N=768,device_name=NVIDIA_H200.json +146 -0
  415. vllm/model_executor/layers/fused_moe/configs/E=128,N=8960,device_name=NVIDIA_H100_80GB_HBM3,dtype=bf16.json +82 -0
  416. vllm/model_executor/layers/fused_moe/configs/E=128,N=8960,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +82 -0
  417. vllm/model_executor/layers/fused_moe/configs/E=128,N=928,device_name=NVIDIA_H100_80GB_HBM3.json +147 -0
  418. vllm/model_executor/layers/fused_moe/configs/E=128,N=928,device_name=NVIDIA_L40S.json +147 -0
  419. vllm/model_executor/layers/fused_moe/configs/E=128,N=96,device_name=NVIDIA_H20.json +146 -0
  420. vllm/model_executor/layers/fused_moe/configs/E=16,N=1024,device_name=AMD_Instinct_MI300X.json +200 -0
  421. vllm/model_executor/layers/fused_moe/configs/E=16,N=1024,device_name=NVIDIA_B200,dtype=fp8_w8a8.json +147 -0
  422. vllm/model_executor/layers/fused_moe/configs/E=16,N=1024,device_name=NVIDIA_B200.json +146 -0
  423. vllm/model_executor/layers/fused_moe/configs/E=16,N=1024,device_name=NVIDIA_H100.json +146 -0
  424. vllm/model_executor/layers/fused_moe/configs/E=16,N=1024,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  425. vllm/model_executor/layers/fused_moe/configs/E=16,N=1024,device_name=NVIDIA_H200.json +146 -0
  426. vllm/model_executor/layers/fused_moe/configs/E=16,N=1344,device_name=NVIDIA_A100-SXM4-40GB.json +146 -0
  427. vllm/model_executor/layers/fused_moe/configs/E=16,N=1344,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  428. vllm/model_executor/layers/fused_moe/configs/E=16,N=1344,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  429. vllm/model_executor/layers/fused_moe/configs/E=16,N=14336,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +146 -0
  430. vllm/model_executor/layers/fused_moe/configs/E=16,N=14336,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  431. vllm/model_executor/layers/fused_moe/configs/E=16,N=14336,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  432. vllm/model_executor/layers/fused_moe/configs/E=16,N=1792,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +218 -0
  433. vllm/model_executor/layers/fused_moe/configs/E=16,N=1792,device_name=NVIDIA_A100-SXM4-80GB.json +218 -0
  434. vllm/model_executor/layers/fused_moe/configs/E=16,N=1792,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  435. vllm/model_executor/layers/fused_moe/configs/E=16,N=1792,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  436. vllm/model_executor/layers/fused_moe/configs/E=16,N=2048,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  437. vllm/model_executor/layers/fused_moe/configs/E=16,N=2048,device_name=NVIDIA_H200.json +146 -0
  438. vllm/model_executor/layers/fused_moe/configs/E=16,N=2688,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  439. vllm/model_executor/layers/fused_moe/configs/E=16,N=2688,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  440. vllm/model_executor/layers/fused_moe/configs/E=16,N=3072,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +146 -0
  441. vllm/model_executor/layers/fused_moe/configs/E=16,N=3072,device_name=NVIDIA_H100_80GB_HBM3,dtype=float8.json +146 -0
  442. vllm/model_executor/layers/fused_moe/configs/E=16,N=3072,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  443. vllm/model_executor/layers/fused_moe/configs/E=16,N=3072,device_name=NVIDIA_H200,dtype=int8_w8a16.json +146 -0
  444. vllm/model_executor/layers/fused_moe/configs/E=16,N=3200,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +130 -0
  445. vllm/model_executor/layers/fused_moe/configs/E=16,N=3584,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +146 -0
  446. vllm/model_executor/layers/fused_moe/configs/E=16,N=3584,device_name=NVIDIA_A100-SXM4-80GB.json +218 -0
  447. vllm/model_executor/layers/fused_moe/configs/E=16,N=3584,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  448. vllm/model_executor/layers/fused_moe/configs/E=16,N=6400,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +130 -0
  449. vllm/model_executor/layers/fused_moe/configs/E=16,N=7168,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a16.json +146 -0
  450. vllm/model_executor/layers/fused_moe/configs/E=16,N=7168,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  451. vllm/model_executor/layers/fused_moe/configs/E=16,N=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=float8.json +146 -0
  452. vllm/model_executor/layers/fused_moe/configs/E=16,N=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=int8_w8a16.json +146 -0
  453. vllm/model_executor/layers/fused_moe/configs/E=16,N=800,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +130 -0
  454. vllm/model_executor/layers/fused_moe/configs/E=160,N=192,device_name=AMD_Instinct_MI300X.json +201 -0
  455. vllm/model_executor/layers/fused_moe/configs/E=160,N=192,device_name=AMD_Instinct_MI350_OAM,dtype=fp8_w8a8.json +164 -0
  456. vllm/model_executor/layers/fused_moe/configs/E=160,N=192,device_name=NVIDIA_A800-SXM4-80GB.json +146 -0
  457. vllm/model_executor/layers/fused_moe/configs/E=160,N=192,device_name=NVIDIA_H20-3e.json +146 -0
  458. vllm/model_executor/layers/fused_moe/configs/E=160,N=192,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +147 -0
  459. vllm/model_executor/layers/fused_moe/configs/E=160,N=320,device_name=NVIDIA_H20-3e.json +146 -0
  460. vllm/model_executor/layers/fused_moe/configs/E=160,N=384,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  461. vllm/model_executor/layers/fused_moe/configs/E=160,N=384,device_name=AMD_Instinct_MI350_OAM,dtype=fp8_w8a8.json +164 -0
  462. vllm/model_executor/layers/fused_moe/configs/E=160,N=384,device_name=AMD_Instinct_MI355_OAM,dtype=fp8_w8a8.json +164 -0
  463. vllm/model_executor/layers/fused_moe/configs/E=160,N=640,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  464. vllm/model_executor/layers/fused_moe/configs/E=160,N=640,device_name=NVIDIA_GB200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  465. vllm/model_executor/layers/fused_moe/configs/E=160,N=640,device_name=NVIDIA_H100,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  466. vllm/model_executor/layers/fused_moe/configs/E=20,N=1536,device_name=NVIDIA_RTX_PRO_6000_Blackwell_Server_Edition,dtype=fp8_w8a8.json +147 -0
  467. vllm/model_executor/layers/fused_moe/configs/E=20,N=2560,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  468. vllm/model_executor/layers/fused_moe/configs/E=20,N=2560,device_name=NVIDIA_GB200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  469. vllm/model_executor/layers/fused_moe/configs/E=20,N=2560,device_name=NVIDIA_H100,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  470. vllm/model_executor/layers/fused_moe/configs/E=20,N=2560,device_name=NVIDIA_H20-3e,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  471. vllm/model_executor/layers/fused_moe/configs/E=256,N=1024,device_name=AMD_Instinct_MI325X,block_shape=[128,128].json +200 -0
  472. vllm/model_executor/layers/fused_moe/configs/E=256,N=1024,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +200 -0
  473. vllm/model_executor/layers/fused_moe/configs/E=256,N=128,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  474. vllm/model_executor/layers/fused_moe/configs/E=256,N=128,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8.json +146 -0
  475. vllm/model_executor/layers/fused_moe/configs/E=256,N=128,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  476. vllm/model_executor/layers/fused_moe/configs/E=256,N=128,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8.json +146 -0
  477. vllm/model_executor/layers/fused_moe/configs/E=256,N=128,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  478. vllm/model_executor/layers/fused_moe/configs/E=256,N=128,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  479. vllm/model_executor/layers/fused_moe/configs/E=256,N=128,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  480. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  481. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +200 -0
  482. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +200 -0
  483. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  484. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  485. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  486. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=NVIDIA_H20-3e,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  487. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  488. vllm/model_executor/layers/fused_moe/configs/E=256,N=256,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  489. vllm/model_executor/layers/fused_moe/configs/E=256,N=384,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +147 -0
  490. vllm/model_executor/layers/fused_moe/configs/E=256,N=512,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +200 -0
  491. vllm/model_executor/layers/fused_moe/configs/E=256,N=512,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  492. vllm/model_executor/layers/fused_moe/configs/E=256,N=64,device_name=NVIDIA_A800-SXM4-80GB.json +146 -0
  493. vllm/model_executor/layers/fused_moe/configs/E=32,N=1408,device_name=NVIDIA_B200.json +147 -0
  494. vllm/model_executor/layers/fused_moe/configs/E=32,N=2048,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +147 -0
  495. vllm/model_executor/layers/fused_moe/configs/E=32,N=2048,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +147 -0
  496. vllm/model_executor/layers/fused_moe/configs/E=384,N=128,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  497. vllm/model_executor/layers/fused_moe/configs/E=384,N=128,device_name=NVIDIA_GB200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  498. vllm/model_executor/layers/fused_moe/configs/E=384,N=128,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  499. vllm/model_executor/layers/fused_moe/configs/E=384,N=256,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  500. vllm/model_executor/layers/fused_moe/configs/E=384,N=256,device_name=NVIDIA_GB200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  501. vllm/model_executor/layers/fused_moe/configs/E=40,N=1536,device_name=NVIDIA_B200,dtype=fp8_w8a8.json +147 -0
  502. vllm/model_executor/layers/fused_moe/configs/E=40,N=2560,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  503. vllm/model_executor/layers/fused_moe/configs/E=40,N=2560,device_name=NVIDIA_GB200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  504. vllm/model_executor/layers/fused_moe/configs/E=40,N=2560,device_name=NVIDIA_H100,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  505. vllm/model_executor/layers/fused_moe/configs/E=512,N=128,device_name=NVIDIA_A100-SXM4-80GB.json +147 -0
  506. vllm/model_executor/layers/fused_moe/configs/E=512,N=128,device_name=NVIDIA_B200.json +146 -0
  507. vllm/model_executor/layers/fused_moe/configs/E=512,N=128,device_name=NVIDIA_GB200,dtype=fp8_w8a8.json +147 -0
  508. vllm/model_executor/layers/fused_moe/configs/E=512,N=128,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  509. vllm/model_executor/layers/fused_moe/configs/E=512,N=128,device_name=NVIDIA_H20-3e.json +146 -0
  510. vllm/model_executor/layers/fused_moe/configs/E=512,N=128,device_name=NVIDIA_H200.json +146 -0
  511. vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=NVIDIA_B200.json +146 -0
  512. vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=NVIDIA_GB200,dtype=fp8_w8a8.json +146 -0
  513. vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +147 -0
  514. vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  515. vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=NVIDIA_H20-3e.json +146 -0
  516. vllm/model_executor/layers/fused_moe/configs/E=512,N=256,device_name=NVIDIA_H200.json +146 -0
  517. vllm/model_executor/layers/fused_moe/configs/E=512,N=512,device_name=NVIDIA_B200.json +146 -0
  518. vllm/model_executor/layers/fused_moe/configs/E=512,N=512,device_name=NVIDIA_GB200,dtype=fp8_w8a8.json +146 -0
  519. vllm/model_executor/layers/fused_moe/configs/E=512,N=512,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  520. vllm/model_executor/layers/fused_moe/configs/E=512,N=512,device_name=NVIDIA_H20-3e.json +146 -0
  521. vllm/model_executor/layers/fused_moe/configs/E=512,N=512,device_name=NVIDIA_H200.json +146 -0
  522. vllm/model_executor/layers/fused_moe/configs/E=512,N=64,device_name=NVIDIA_A100-SXM4-80GB.json +147 -0
  523. vllm/model_executor/layers/fused_moe/configs/E=512,N=64,device_name=NVIDIA_B200.json +146 -0
  524. vllm/model_executor/layers/fused_moe/configs/E=512,N=64,device_name=NVIDIA_H20-3e.json +146 -0
  525. vllm/model_executor/layers/fused_moe/configs/E=512,N=64,device_name=NVIDIA_H200.json +146 -0
  526. vllm/model_executor/layers/fused_moe/configs/E=60,N=1408,device_name=AMD_Instinct_MI300X.json +200 -0
  527. vllm/model_executor/layers/fused_moe/configs/E=60,N=176,device_name=AMD_Instinct_MI300X.json +200 -0
  528. vllm/model_executor/layers/fused_moe/configs/E=60,N=352,device_name=AMD_Instinct_MI300X.json +200 -0
  529. vllm/model_executor/layers/fused_moe/configs/E=60,N=704,device_name=AMD_Instinct_MI300X.json +200 -0
  530. vllm/model_executor/layers/fused_moe/configs/E=62,N=128,device_name=AMD_Instinct_MI300X.json +200 -0
  531. vllm/model_executor/layers/fused_moe/configs/E=62,N=256,device_name=AMD_Instinct_MI300X.json +200 -0
  532. vllm/model_executor/layers/fused_moe/configs/E=62,N=256,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  533. vllm/model_executor/layers/fused_moe/configs/E=62,N=512,device_name=AMD_Instinct_MI300X.json +200 -0
  534. vllm/model_executor/layers/fused_moe/configs/E=62,N=512,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  535. vllm/model_executor/layers/fused_moe/configs/E=64,N=1280,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  536. vllm/model_executor/layers/fused_moe/configs/E=64,N=1280,device_name=NVIDIA_A800-SXM4-80GB.json +146 -0
  537. vllm/model_executor/layers/fused_moe/configs/E=64,N=1280,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  538. vllm/model_executor/layers/fused_moe/configs/E=64,N=1280,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  539. vllm/model_executor/layers/fused_moe/configs/E=64,N=1280,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  540. vllm/model_executor/layers/fused_moe/configs/E=64,N=1280,device_name=NVIDIA_H200.json +146 -0
  541. vllm/model_executor/layers/fused_moe/configs/E=64,N=1408,device_name=NVIDIA_B200.json +147 -0
  542. vllm/model_executor/layers/fused_moe/configs/E=64,N=1536,device_name=NVIDIA_H20,dtype=fp8_w8a8.json +146 -0
  543. vllm/model_executor/layers/fused_moe/configs/E=64,N=2560,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  544. vllm/model_executor/layers/fused_moe/configs/E=64,N=2560,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  545. vllm/model_executor/layers/fused_moe/configs/E=64,N=2560,device_name=NVIDIA_H200.json +146 -0
  546. vllm/model_executor/layers/fused_moe/configs/E=64,N=3072,device_name=NVIDIA_H20,dtype=fp8_w8a8.json +146 -0
  547. vllm/model_executor/layers/fused_moe/configs/E=64,N=3072,device_name=NVIDIA_H20.json +146 -0
  548. vllm/model_executor/layers/fused_moe/configs/E=64,N=320,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  549. vllm/model_executor/layers/fused_moe/configs/E=64,N=320,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  550. vllm/model_executor/layers/fused_moe/configs/E=64,N=320,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  551. vllm/model_executor/layers/fused_moe/configs/E=64,N=320,device_name=NVIDIA_H200.json +146 -0
  552. vllm/model_executor/layers/fused_moe/configs/E=64,N=384,device_name=NVIDIA_H20,dtype=fp8_w8a8.json +146 -0
  553. vllm/model_executor/layers/fused_moe/configs/E=64,N=384,device_name=NVIDIA_H20.json +146 -0
  554. vllm/model_executor/layers/fused_moe/configs/E=64,N=640,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  555. vllm/model_executor/layers/fused_moe/configs/E=64,N=640,device_name=NVIDIA_A800-SXM4-80GB.json +146 -0
  556. vllm/model_executor/layers/fused_moe/configs/E=64,N=640,device_name=NVIDIA_GeForce_RTX_4090,dtype=fp8_w8a8.json +146 -0
  557. vllm/model_executor/layers/fused_moe/configs/E=64,N=640,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  558. vllm/model_executor/layers/fused_moe/configs/E=64,N=640,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  559. vllm/model_executor/layers/fused_moe/configs/E=64,N=640,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  560. vllm/model_executor/layers/fused_moe/configs/E=64,N=640,device_name=NVIDIA_H200.json +146 -0
  561. vllm/model_executor/layers/fused_moe/configs/E=64,N=768,device_name=NVIDIA_H100_PCIe,dtype=fp8_w8a8,block_shape=[128,128].json +147 -0
  562. vllm/model_executor/layers/fused_moe/configs/E=64,N=768,device_name=NVIDIA_H20,dtype=fp8_w8a8.json +146 -0
  563. vllm/model_executor/layers/fused_moe/configs/E=64,N=768,device_name=NVIDIA_H20.json +146 -0
  564. vllm/model_executor/layers/fused_moe/configs/E=64,N=896,device_name=NVIDIA_H20.json +146 -0
  565. vllm/model_executor/layers/fused_moe/configs/E=64,N=8960,device_name=NVIDIA_H100_80GB_HBM3,dtype=bf16.json +82 -0
  566. vllm/model_executor/layers/fused_moe/configs/E=64,N=8960,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +82 -0
  567. vllm/model_executor/layers/fused_moe/configs/E=72,N=192,device_name=AMD_Instinct_MI300X.json +200 -0
  568. vllm/model_executor/layers/fused_moe/configs/E=72,N=384,device_name=AMD_Instinct_MI300X.json +200 -0
  569. vllm/model_executor/layers/fused_moe/configs/E=72,N=384,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  570. vllm/model_executor/layers/fused_moe/configs/E=72,N=768,device_name=AMD_Instinct_MI300X.json +200 -0
  571. vllm/model_executor/layers/fused_moe/configs/E=72,N=768,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  572. vllm/model_executor/layers/fused_moe/configs/E=8,N=14336,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  573. vllm/model_executor/layers/fused_moe/configs/E=8,N=14336,device_name=AMD_Instinct_MI300X.json +200 -0
  574. vllm/model_executor/layers/fused_moe/configs/E=8,N=14336,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  575. vllm/model_executor/layers/fused_moe/configs/E=8,N=14336,device_name=AMD_Instinct_MI325X.json +200 -0
  576. vllm/model_executor/layers/fused_moe/configs/E=8,N=14336,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +138 -0
  577. vllm/model_executor/layers/fused_moe/configs/E=8,N=14336,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  578. vllm/model_executor/layers/fused_moe/configs/E=8,N=14336,device_name=NVIDIA_H200.json +146 -0
  579. vllm/model_executor/layers/fused_moe/configs/E=8,N=16384,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  580. vllm/model_executor/layers/fused_moe/configs/E=8,N=16384,device_name=AMD_Instinct_MI300X.json +200 -0
  581. vllm/model_executor/layers/fused_moe/configs/E=8,N=16384,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  582. vllm/model_executor/layers/fused_moe/configs/E=8,N=16384,device_name=AMD_Instinct_MI325X.json +200 -0
  583. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  584. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=AMD_Instinct_MI300X.json +200 -0
  585. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  586. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=AMD_Instinct_MI325X.json +200 -0
  587. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=NVIDIA_A100-SXM4-40GB.json +146 -0
  588. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  589. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  590. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  591. vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=NVIDIA_H200.json +146 -0
  592. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  593. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=AMD_Instinct_MI300X.json +200 -0
  594. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  595. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=AMD_Instinct_MI325X.json +200 -0
  596. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  597. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  598. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  599. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +154 -0
  600. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  601. vllm/model_executor/layers/fused_moe/configs/E=8,N=2048,device_name=NVIDIA_H200.json +146 -0
  602. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  603. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=AMD_Instinct_MI300X.json +200 -0
  604. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  605. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=AMD_Instinct_MI325X.json +200 -0
  606. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=NVIDIA_A100-SXM4-40GB.json +146 -0
  607. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  608. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=NVIDIA_GeForce_RTX_4090,dtype=fp8_w8a8.json +146 -0
  609. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  610. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  611. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  612. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=NVIDIA_H200.json +146 -0
  613. vllm/model_executor/layers/fused_moe/configs/E=8,N=3584,device_name=NVIDIA_L40S.json +173 -0
  614. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  615. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=AMD_Instinct_MI300X.json +200 -0
  616. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  617. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=AMD_Instinct_MI325X.json +200 -0
  618. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  619. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  620. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  621. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  622. vllm/model_executor/layers/fused_moe/configs/E=8,N=4096,device_name=NVIDIA_H200.json +146 -0
  623. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  624. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=AMD_Instinct_MI300X.json +200 -0
  625. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  626. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=AMD_Instinct_MI325X.json +200 -0
  627. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=NVIDIA_A100-SXM4-80GB.json +146 -0
  628. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  629. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=NVIDIA_H100_80GB_HBM3.json +146 -0
  630. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  631. vllm/model_executor/layers/fused_moe/configs/E=8,N=7168,device_name=NVIDIA_H200.json +146 -0
  632. vllm/model_executor/layers/fused_moe/configs/E=8,N=8192,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8.json +164 -0
  633. vllm/model_executor/layers/fused_moe/configs/E=8,N=8192,device_name=AMD_Instinct_MI300X.json +200 -0
  634. vllm/model_executor/layers/fused_moe/configs/E=8,N=8192,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8.json +164 -0
  635. vllm/model_executor/layers/fused_moe/configs/E=8,N=8192,device_name=AMD_Instinct_MI325X.json +200 -0
  636. vllm/model_executor/layers/fused_moe/configs/E=8,N=8192,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8.json +146 -0
  637. vllm/model_executor/layers/fused_moe/configs/E=8,N=8192,device_name=NVIDIA_H200,dtype=fp8_w8a8.json +146 -0
  638. vllm/model_executor/layers/fused_moe/configs/README +12 -0
  639. vllm/model_executor/layers/fused_moe/cpu_fused_moe.py +292 -0
  640. vllm/model_executor/layers/fused_moe/cutlass_moe.py +1453 -0
  641. vllm/model_executor/layers/fused_moe/deep_gemm_moe.py +358 -0
  642. vllm/model_executor/layers/fused_moe/deep_gemm_utils.py +427 -0
  643. vllm/model_executor/layers/fused_moe/deepep_ht_prepare_finalize.py +420 -0
  644. vllm/model_executor/layers/fused_moe/deepep_ll_prepare_finalize.py +434 -0
  645. vllm/model_executor/layers/fused_moe/flashinfer_cutedsl_moe.py +376 -0
  646. vllm/model_executor/layers/fused_moe/flashinfer_cutlass_moe.py +307 -0
  647. vllm/model_executor/layers/fused_moe/flashinfer_cutlass_prepare_finalize.py +362 -0
  648. vllm/model_executor/layers/fused_moe/flashinfer_trtllm_moe.py +192 -0
  649. vllm/model_executor/layers/fused_moe/fused_batched_moe.py +1012 -0
  650. vllm/model_executor/layers/fused_moe/fused_marlin_moe.py +825 -0
  651. vllm/model_executor/layers/fused_moe/fused_moe.py +2223 -0
  652. vllm/model_executor/layers/fused_moe/fused_moe_method_base.py +103 -0
  653. vllm/model_executor/layers/fused_moe/fused_moe_modular_method.py +119 -0
  654. vllm/model_executor/layers/fused_moe/gpt_oss_triton_kernels_moe.py +524 -0
  655. vllm/model_executor/layers/fused_moe/layer.py +2133 -0
  656. vllm/model_executor/layers/fused_moe/modular_kernel.py +1302 -0
  657. vllm/model_executor/layers/fused_moe/moe_align_block_size.py +192 -0
  658. vllm/model_executor/layers/fused_moe/moe_pallas.py +83 -0
  659. vllm/model_executor/layers/fused_moe/moe_permute_unpermute.py +229 -0
  660. vllm/model_executor/layers/fused_moe/moe_torch_iterative.py +60 -0
  661. vllm/model_executor/layers/fused_moe/pplx_prepare_finalize.py +362 -0
  662. vllm/model_executor/layers/fused_moe/prepare_finalize.py +78 -0
  663. vllm/model_executor/layers/fused_moe/rocm_aiter_fused_moe.py +265 -0
  664. vllm/model_executor/layers/fused_moe/routing_simulator.py +310 -0
  665. vllm/model_executor/layers/fused_moe/shared_fused_moe.py +96 -0
  666. vllm/model_executor/layers/fused_moe/topk_weight_and_reduce.py +171 -0
  667. vllm/model_executor/layers/fused_moe/triton_deep_gemm_moe.py +163 -0
  668. vllm/model_executor/layers/fused_moe/trtllm_moe.py +143 -0
  669. vllm/model_executor/layers/fused_moe/unquantized_fused_moe_method.py +455 -0
  670. vllm/model_executor/layers/fused_moe/utils.py +332 -0
  671. vllm/model_executor/layers/kda.py +442 -0
  672. vllm/model_executor/layers/layernorm.py +442 -0
  673. vllm/model_executor/layers/lightning_attn.py +735 -0
  674. vllm/model_executor/layers/linear.py +1424 -0
  675. vllm/model_executor/layers/logits_processor.py +106 -0
  676. vllm/model_executor/layers/mamba/__init__.py +0 -0
  677. vllm/model_executor/layers/mamba/abstract.py +68 -0
  678. vllm/model_executor/layers/mamba/linear_attn.py +388 -0
  679. vllm/model_executor/layers/mamba/mamba_mixer.py +526 -0
  680. vllm/model_executor/layers/mamba/mamba_mixer2.py +930 -0
  681. vllm/model_executor/layers/mamba/mamba_utils.py +225 -0
  682. vllm/model_executor/layers/mamba/ops/__init__.py +0 -0
  683. vllm/model_executor/layers/mamba/ops/causal_conv1d.py +1240 -0
  684. vllm/model_executor/layers/mamba/ops/layernorm_gated.py +172 -0
  685. vllm/model_executor/layers/mamba/ops/mamba_ssm.py +586 -0
  686. vllm/model_executor/layers/mamba/ops/ssd_bmm.py +211 -0
  687. vllm/model_executor/layers/mamba/ops/ssd_chunk_scan.py +456 -0
  688. vllm/model_executor/layers/mamba/ops/ssd_chunk_state.py +700 -0
  689. vllm/model_executor/layers/mamba/ops/ssd_combined.py +230 -0
  690. vllm/model_executor/layers/mamba/ops/ssd_state_passing.py +157 -0
  691. vllm/model_executor/layers/mamba/short_conv.py +255 -0
  692. vllm/model_executor/layers/mla.py +176 -0
  693. vllm/model_executor/layers/pooler.py +830 -0
  694. vllm/model_executor/layers/quantization/__init__.py +179 -0
  695. vllm/model_executor/layers/quantization/auto_round.py +454 -0
  696. vllm/model_executor/layers/quantization/awq.py +277 -0
  697. vllm/model_executor/layers/quantization/awq_marlin.py +793 -0
  698. vllm/model_executor/layers/quantization/awq_triton.py +337 -0
  699. vllm/model_executor/layers/quantization/base_config.py +170 -0
  700. vllm/model_executor/layers/quantization/bitblas.py +502 -0
  701. vllm/model_executor/layers/quantization/bitsandbytes.py +626 -0
  702. vllm/model_executor/layers/quantization/compressed_tensors/__init__.py +3 -0
  703. vllm/model_executor/layers/quantization/compressed_tensors/compressed_tensors.py +986 -0
  704. vllm/model_executor/layers/quantization/compressed_tensors/compressed_tensors_moe.py +2645 -0
  705. vllm/model_executor/layers/quantization/compressed_tensors/schemes/__init__.py +35 -0
  706. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_24.py +392 -0
  707. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_scheme.py +55 -0
  708. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w4a16_24.py +176 -0
  709. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w4a16_nvfp4.py +124 -0
  710. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w4a4_nvfp4.py +218 -0
  711. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w4a8_fp8.py +176 -0
  712. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w4a8_int.py +153 -0
  713. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w8a16_fp8.py +138 -0
  714. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w8a8_fp8.py +200 -0
  715. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_w8a8_int8.py +125 -0
  716. vllm/model_executor/layers/quantization/compressed_tensors/schemes/compressed_tensors_wNa16.py +230 -0
  717. vllm/model_executor/layers/quantization/compressed_tensors/transform/__init__.py +0 -0
  718. vllm/model_executor/layers/quantization/compressed_tensors/transform/linear.py +260 -0
  719. vllm/model_executor/layers/quantization/compressed_tensors/transform/module.py +173 -0
  720. vllm/model_executor/layers/quantization/compressed_tensors/transform/schemes/__init__.py +0 -0
  721. vllm/model_executor/layers/quantization/compressed_tensors/transform/schemes/linear_qutlass_nvfp4.py +64 -0
  722. vllm/model_executor/layers/quantization/compressed_tensors/transform/utils.py +13 -0
  723. vllm/model_executor/layers/quantization/compressed_tensors/triton_scaled_mm.py +224 -0
  724. vllm/model_executor/layers/quantization/compressed_tensors/utils.py +216 -0
  725. vllm/model_executor/layers/quantization/cpu_wna16.py +625 -0
  726. vllm/model_executor/layers/quantization/deepspeedfp.py +218 -0
  727. vllm/model_executor/layers/quantization/experts_int8.py +207 -0
  728. vllm/model_executor/layers/quantization/fbgemm_fp8.py +195 -0
  729. vllm/model_executor/layers/quantization/fp8.py +1461 -0
  730. vllm/model_executor/layers/quantization/fp_quant.py +420 -0
  731. vllm/model_executor/layers/quantization/gguf.py +677 -0
  732. vllm/model_executor/layers/quantization/gptq.py +393 -0
  733. vllm/model_executor/layers/quantization/gptq_bitblas.py +482 -0
  734. vllm/model_executor/layers/quantization/gptq_marlin.py +932 -0
  735. vllm/model_executor/layers/quantization/gptq_marlin_24.py +320 -0
  736. vllm/model_executor/layers/quantization/hqq_marlin.py +372 -0
  737. vllm/model_executor/layers/quantization/inc.py +65 -0
  738. vllm/model_executor/layers/quantization/input_quant_fp8.py +202 -0
  739. vllm/model_executor/layers/quantization/ipex_quant.py +487 -0
  740. vllm/model_executor/layers/quantization/kernels/__init__.py +0 -0
  741. vllm/model_executor/layers/quantization/kernels/mixed_precision/MPLinearKernel.py +94 -0
  742. vllm/model_executor/layers/quantization/kernels/mixed_precision/__init__.py +109 -0
  743. vllm/model_executor/layers/quantization/kernels/mixed_precision/allspark.py +115 -0
  744. vllm/model_executor/layers/quantization/kernels/mixed_precision/bitblas.py +323 -0
  745. vllm/model_executor/layers/quantization/kernels/mixed_precision/conch.py +98 -0
  746. vllm/model_executor/layers/quantization/kernels/mixed_precision/cutlass.py +130 -0
  747. vllm/model_executor/layers/quantization/kernels/mixed_precision/dynamic_4bit.py +111 -0
  748. vllm/model_executor/layers/quantization/kernels/mixed_precision/exllama.py +161 -0
  749. vllm/model_executor/layers/quantization/kernels/mixed_precision/machete.py +159 -0
  750. vllm/model_executor/layers/quantization/kernels/mixed_precision/marlin.py +200 -0
  751. vllm/model_executor/layers/quantization/kernels/mixed_precision/xpu.py +97 -0
  752. vllm/model_executor/layers/quantization/kernels/scaled_mm/ScaledMMLinearKernel.py +76 -0
  753. vllm/model_executor/layers/quantization/kernels/scaled_mm/__init__.py +81 -0
  754. vllm/model_executor/layers/quantization/kernels/scaled_mm/aiter.py +128 -0
  755. vllm/model_executor/layers/quantization/kernels/scaled_mm/cpu.py +220 -0
  756. vllm/model_executor/layers/quantization/kernels/scaled_mm/cutlass.py +147 -0
  757. vllm/model_executor/layers/quantization/kernels/scaled_mm/triton.py +71 -0
  758. vllm/model_executor/layers/quantization/kernels/scaled_mm/xla.py +106 -0
  759. vllm/model_executor/layers/quantization/kv_cache.py +153 -0
  760. vllm/model_executor/layers/quantization/modelopt.py +1684 -0
  761. vllm/model_executor/layers/quantization/moe_wna16.py +516 -0
  762. vllm/model_executor/layers/quantization/mxfp4.py +1140 -0
  763. vllm/model_executor/layers/quantization/petit.py +319 -0
  764. vllm/model_executor/layers/quantization/ptpc_fp8.py +136 -0
  765. vllm/model_executor/layers/quantization/quark/__init__.py +0 -0
  766. vllm/model_executor/layers/quantization/quark/quark.py +527 -0
  767. vllm/model_executor/layers/quantization/quark/quark_moe.py +622 -0
  768. vllm/model_executor/layers/quantization/quark/schemes/__init__.py +9 -0
  769. vllm/model_executor/layers/quantization/quark/schemes/quark_ocp_mx.py +343 -0
  770. vllm/model_executor/layers/quantization/quark/schemes/quark_scheme.py +55 -0
  771. vllm/model_executor/layers/quantization/quark/schemes/quark_w8a8_fp8.py +179 -0
  772. vllm/model_executor/layers/quantization/quark/schemes/quark_w8a8_int8.py +139 -0
  773. vllm/model_executor/layers/quantization/quark/utils.py +105 -0
  774. vllm/model_executor/layers/quantization/qutlass_utils.py +185 -0
  775. vllm/model_executor/layers/quantization/rtn.py +621 -0
  776. vllm/model_executor/layers/quantization/schema.py +90 -0
  777. vllm/model_executor/layers/quantization/torchao.py +380 -0
  778. vllm/model_executor/layers/quantization/tpu_int8.py +139 -0
  779. vllm/model_executor/layers/quantization/utils/__init__.py +6 -0
  780. vllm/model_executor/layers/quantization/utils/allspark_utils.py +67 -0
  781. vllm/model_executor/layers/quantization/utils/bitblas_utils.py +229 -0
  782. vllm/model_executor/layers/quantization/utils/configs/N=10240,K=5120,device_name=NVIDIA_L40S,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  783. vllm/model_executor/layers/quantization/utils/configs/N=12288,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  784. vllm/model_executor/layers/quantization/utils/configs/N=12288,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  785. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=1536,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  786. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=1536,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  787. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=1536,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  788. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=1536,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  789. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=1536,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  790. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=1536,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  791. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=1536,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  792. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=1536,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  793. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  794. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  795. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  796. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  797. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  798. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  799. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  800. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  801. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  802. vllm/model_executor/layers/quantization/utils/configs/N=1536,K=7168,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  803. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  804. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  805. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  806. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  807. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  808. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  809. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  810. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  811. vllm/model_executor/layers/quantization/utils/configs/N=2048,K=512,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  812. vllm/model_executor/layers/quantization/utils/configs/N=2112,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  813. vllm/model_executor/layers/quantization/utils/configs/N=2112,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  814. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  815. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  816. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  817. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  818. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  819. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  820. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  821. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  822. vllm/model_executor/layers/quantization/utils/configs/N=2304,K=7168,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  823. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=1536,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  824. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=1536,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  825. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  826. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  827. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  828. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  829. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  830. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  831. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  832. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  833. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  834. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  835. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  836. vllm/model_executor/layers/quantization/utils/configs/N=24576,K=7168,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  837. vllm/model_executor/layers/quantization/utils/configs/N=256,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  838. vllm/model_executor/layers/quantization/utils/configs/N=256,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  839. vllm/model_executor/layers/quantization/utils/configs/N=256,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  840. vllm/model_executor/layers/quantization/utils/configs/N=256,K=7168,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  841. vllm/model_executor/layers/quantization/utils/configs/N=256,K=7168,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  842. vllm/model_executor/layers/quantization/utils/configs/N=256,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  843. vllm/model_executor/layers/quantization/utils/configs/N=256,K=7168,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  844. vllm/model_executor/layers/quantization/utils/configs/N=256,K=7168,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  845. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=1536,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  846. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=1536,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  847. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=1536,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  848. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=1536,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  849. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=1536,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  850. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=1536,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  851. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=1536,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  852. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  853. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  854. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  855. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=7168,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  856. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  857. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=7168,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  858. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  859. vllm/model_executor/layers/quantization/utils/configs/N=3072,K=7168,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  860. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  861. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  862. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  863. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  864. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  865. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  866. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  867. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  868. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  869. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  870. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  871. vllm/model_executor/layers/quantization/utils/configs/N=32768,K=512,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  872. vllm/model_executor/layers/quantization/utils/configs/N=36864,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  873. vllm/model_executor/layers/quantization/utils/configs/N=36864,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  874. vllm/model_executor/layers/quantization/utils/configs/N=36864,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  875. vllm/model_executor/layers/quantization/utils/configs/N=36864,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  876. vllm/model_executor/layers/quantization/utils/configs/N=36864,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  877. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=512,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  878. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=512,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  879. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=512,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  880. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=512,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  881. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=512,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  882. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=512,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  883. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=512,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  884. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=512,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  885. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  886. vllm/model_executor/layers/quantization/utils/configs/N=4096,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  887. vllm/model_executor/layers/quantization/utils/configs/N=4608,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  888. vllm/model_executor/layers/quantization/utils/configs/N=4608,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  889. vllm/model_executor/layers/quantization/utils/configs/N=4608,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  890. vllm/model_executor/layers/quantization/utils/configs/N=4608,K=7168,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  891. vllm/model_executor/layers/quantization/utils/configs/N=4608,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  892. vllm/model_executor/layers/quantization/utils/configs/N=4608,K=7168,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  893. vllm/model_executor/layers/quantization/utils/configs/N=4608,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  894. vllm/model_executor/layers/quantization/utils/configs/N=4608,K=7168,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  895. vllm/model_executor/layers/quantization/utils/configs/N=512,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  896. vllm/model_executor/layers/quantization/utils/configs/N=512,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  897. vllm/model_executor/layers/quantization/utils/configs/N=512,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  898. vllm/model_executor/layers/quantization/utils/configs/N=512,K=7168,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  899. vllm/model_executor/layers/quantization/utils/configs/N=512,K=7168,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  900. vllm/model_executor/layers/quantization/utils/configs/N=512,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  901. vllm/model_executor/layers/quantization/utils/configs/N=512,K=7168,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  902. vllm/model_executor/layers/quantization/utils/configs/N=5120,K=25600,device_name=NVIDIA_L40S,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  903. vllm/model_executor/layers/quantization/utils/configs/N=5120,K=8192,device_name=NVIDIA_L40S,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  904. vllm/model_executor/layers/quantization/utils/configs/N=51200,K=5120,device_name=NVIDIA_L40S,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  905. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  906. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  907. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  908. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  909. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  910. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  911. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  912. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  913. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  914. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  915. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +18 -0
  916. vllm/model_executor/layers/quantization/utils/configs/N=576,K=7168,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  917. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  918. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  919. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  920. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  921. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  922. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  923. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  924. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  925. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1024,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  926. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  927. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  928. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  929. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  930. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  931. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  932. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  933. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  934. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=1152,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  935. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=128,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  936. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=128,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  937. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=128,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  938. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=128,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  939. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=128,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  940. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=128,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  941. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=128,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  942. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=128,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  943. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  944. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  945. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  946. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  947. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  948. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  949. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  950. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  951. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  952. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  953. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  954. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=16384,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  955. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  956. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  957. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  958. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_A100-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  959. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  960. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  961. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  962. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  963. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  964. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  965. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  966. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=18432,device_name=NVIDIA_L20Y,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  967. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2048,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  968. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2048,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  969. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2048,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  970. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2048,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  971. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2048,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  972. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2048,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  973. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2048,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  974. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2048,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  975. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2304,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  976. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2304,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  977. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2304,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  978. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2304,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  979. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2304,device_name=NVIDIA_H100_80GB_HBM3,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  980. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2304,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  981. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2304,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  982. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=2304,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  983. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=256,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  984. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=256,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  985. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=256,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  986. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=256,device_name=NVIDIA_B200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  987. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=256,device_name=NVIDIA_H20,dtype=int8_w8a8,block_shape=[128,128].json +146 -0
  988. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=256,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128,128].json +146 -0
  989. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=256,device_name=NVIDIA_L20,dtype=fp8_w8a8,block_shape=[128,128].json +26 -0
  990. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=8192,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  991. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=8192,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  992. vllm/model_executor/layers/quantization/utils/configs/N=7168,K=8192,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  993. vllm/model_executor/layers/quantization/utils/configs/N=8192,K=1536,device_name=AMD_Instinct_MI300X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  994. vllm/model_executor/layers/quantization/utils/configs/N=8192,K=1536,device_name=AMD_Instinct_MI325X,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  995. vllm/model_executor/layers/quantization/utils/configs/N=8192,K=1536,device_name=AMD_Instinct_MI325_OAM,dtype=fp8_w8a8,block_shape=[128,128].json +164 -0
  996. vllm/model_executor/layers/quantization/utils/configs/README.md +3 -0
  997. vllm/model_executor/layers/quantization/utils/flashinfer_fp4_moe.py +412 -0
  998. vllm/model_executor/layers/quantization/utils/flashinfer_utils.py +312 -0
  999. vllm/model_executor/layers/quantization/utils/fp8_utils.py +1453 -0
  1000. vllm/model_executor/layers/quantization/utils/gptq_utils.py +158 -0
  1001. vllm/model_executor/layers/quantization/utils/int8_utils.py +474 -0
  1002. vllm/model_executor/layers/quantization/utils/layer_utils.py +41 -0
  1003. vllm/model_executor/layers/quantization/utils/machete_utils.py +56 -0
  1004. vllm/model_executor/layers/quantization/utils/marlin_utils.py +678 -0
  1005. vllm/model_executor/layers/quantization/utils/marlin_utils_fp4.py +452 -0
  1006. vllm/model_executor/layers/quantization/utils/marlin_utils_fp8.py +381 -0
  1007. vllm/model_executor/layers/quantization/utils/marlin_utils_test.py +219 -0
  1008. vllm/model_executor/layers/quantization/utils/marlin_utils_test_24.py +467 -0
  1009. vllm/model_executor/layers/quantization/utils/mxfp4_utils.py +189 -0
  1010. vllm/model_executor/layers/quantization/utils/mxfp6_utils.py +142 -0
  1011. vllm/model_executor/layers/quantization/utils/mxfp8_utils.py +24 -0
  1012. vllm/model_executor/layers/quantization/utils/nvfp4_emulation_utils.py +142 -0
  1013. vllm/model_executor/layers/quantization/utils/nvfp4_moe_support.py +67 -0
  1014. vllm/model_executor/layers/quantization/utils/ocp_mx_utils.py +51 -0
  1015. vllm/model_executor/layers/quantization/utils/petit_utils.py +124 -0
  1016. vllm/model_executor/layers/quantization/utils/quant_utils.py +741 -0
  1017. vllm/model_executor/layers/quantization/utils/w8a8_utils.py +519 -0
  1018. vllm/model_executor/layers/resampler.py +283 -0
  1019. vllm/model_executor/layers/rotary_embedding/__init__.py +289 -0
  1020. vllm/model_executor/layers/rotary_embedding/base.py +254 -0
  1021. vllm/model_executor/layers/rotary_embedding/common.py +279 -0
  1022. vllm/model_executor/layers/rotary_embedding/deepseek_scaling_rope.py +165 -0
  1023. vllm/model_executor/layers/rotary_embedding/dual_chunk_rope.py +215 -0
  1024. vllm/model_executor/layers/rotary_embedding/dynamic_ntk_alpha_rope.py +43 -0
  1025. vllm/model_executor/layers/rotary_embedding/dynamic_ntk_scaling_rope.py +68 -0
  1026. vllm/model_executor/layers/rotary_embedding/ernie45_vl_rope.py +82 -0
  1027. vllm/model_executor/layers/rotary_embedding/linear_scaling_rope.py +115 -0
  1028. vllm/model_executor/layers/rotary_embedding/llama3_rope.py +54 -0
  1029. vllm/model_executor/layers/rotary_embedding/llama4_vision_rope.py +80 -0
  1030. vllm/model_executor/layers/rotary_embedding/mrope.py +412 -0
  1031. vllm/model_executor/layers/rotary_embedding/ntk_scaling_rope.py +47 -0
  1032. vllm/model_executor/layers/rotary_embedding/phi3_long_rope_scaled_rope.py +159 -0
  1033. vllm/model_executor/layers/rotary_embedding/xdrope.py +160 -0
  1034. vllm/model_executor/layers/rotary_embedding/yarn_scaling_rope.py +84 -0
  1035. vllm/model_executor/layers/utils.py +251 -0
  1036. vllm/model_executor/layers/vocab_parallel_embedding.py +558 -0
  1037. vllm/model_executor/model_loader/__init__.py +150 -0
  1038. vllm/model_executor/model_loader/base_loader.py +57 -0
  1039. vllm/model_executor/model_loader/bitsandbytes_loader.py +822 -0
  1040. vllm/model_executor/model_loader/default_loader.py +321 -0
  1041. vllm/model_executor/model_loader/dummy_loader.py +28 -0
  1042. vllm/model_executor/model_loader/gguf_loader.py +371 -0
  1043. vllm/model_executor/model_loader/online_quantization.py +275 -0
  1044. vllm/model_executor/model_loader/runai_streamer_loader.py +116 -0
  1045. vllm/model_executor/model_loader/sharded_state_loader.py +214 -0
  1046. vllm/model_executor/model_loader/tensorizer.py +790 -0
  1047. vllm/model_executor/model_loader/tensorizer_loader.py +151 -0
  1048. vllm/model_executor/model_loader/tpu.py +118 -0
  1049. vllm/model_executor/model_loader/utils.py +292 -0
  1050. vllm/model_executor/model_loader/weight_utils.py +1157 -0
  1051. vllm/model_executor/models/__init__.py +44 -0
  1052. vllm/model_executor/models/adapters.py +522 -0
  1053. vllm/model_executor/models/afmoe.py +696 -0
  1054. vllm/model_executor/models/aimv2.py +248 -0
  1055. vllm/model_executor/models/apertus.py +565 -0
  1056. vllm/model_executor/models/arcee.py +428 -0
  1057. vllm/model_executor/models/arctic.py +633 -0
  1058. vllm/model_executor/models/aria.py +653 -0
  1059. vllm/model_executor/models/audioflamingo3.py +639 -0
  1060. vllm/model_executor/models/aya_vision.py +448 -0
  1061. vllm/model_executor/models/bagel.py +584 -0
  1062. vllm/model_executor/models/baichuan.py +493 -0
  1063. vllm/model_executor/models/bailing_moe.py +642 -0
  1064. vllm/model_executor/models/bamba.py +511 -0
  1065. vllm/model_executor/models/bee.py +157 -0
  1066. vllm/model_executor/models/bert.py +925 -0
  1067. vllm/model_executor/models/bert_with_rope.py +732 -0
  1068. vllm/model_executor/models/blip.py +350 -0
  1069. vllm/model_executor/models/blip2.py +693 -0
  1070. vllm/model_executor/models/bloom.py +390 -0
  1071. vllm/model_executor/models/chameleon.py +1095 -0
  1072. vllm/model_executor/models/chatglm.py +502 -0
  1073. vllm/model_executor/models/clip.py +1004 -0
  1074. vllm/model_executor/models/cohere2_vision.py +470 -0
  1075. vllm/model_executor/models/commandr.py +469 -0
  1076. vllm/model_executor/models/config.py +531 -0
  1077. vllm/model_executor/models/dbrx.py +484 -0
  1078. vllm/model_executor/models/deepencoder.py +676 -0
  1079. vllm/model_executor/models/deepseek_eagle.py +252 -0
  1080. vllm/model_executor/models/deepseek_mtp.py +446 -0
  1081. vllm/model_executor/models/deepseek_ocr.py +591 -0
  1082. vllm/model_executor/models/deepseek_v2.py +1710 -0
  1083. vllm/model_executor/models/deepseek_vl2.py +642 -0
  1084. vllm/model_executor/models/dots1.py +565 -0
  1085. vllm/model_executor/models/dots_ocr.py +821 -0
  1086. vllm/model_executor/models/ernie45.py +53 -0
  1087. vllm/model_executor/models/ernie45_moe.py +754 -0
  1088. vllm/model_executor/models/ernie45_vl.py +1621 -0
  1089. vllm/model_executor/models/ernie45_vl_moe.py +800 -0
  1090. vllm/model_executor/models/ernie_mtp.py +279 -0
  1091. vllm/model_executor/models/exaone.py +524 -0
  1092. vllm/model_executor/models/exaone4.py +516 -0
  1093. vllm/model_executor/models/fairseq2_llama.py +154 -0
  1094. vllm/model_executor/models/falcon.py +543 -0
  1095. vllm/model_executor/models/falcon_h1.py +675 -0
  1096. vllm/model_executor/models/flex_olmo.py +155 -0
  1097. vllm/model_executor/models/fuyu.py +371 -0
  1098. vllm/model_executor/models/gemma.py +425 -0
  1099. vllm/model_executor/models/gemma2.py +435 -0
  1100. vllm/model_executor/models/gemma3.py +507 -0
  1101. vllm/model_executor/models/gemma3_mm.py +664 -0
  1102. vllm/model_executor/models/gemma3n.py +1166 -0
  1103. vllm/model_executor/models/gemma3n_mm.py +810 -0
  1104. vllm/model_executor/models/glm.py +24 -0
  1105. vllm/model_executor/models/glm4.py +295 -0
  1106. vllm/model_executor/models/glm4_1v.py +1808 -0
  1107. vllm/model_executor/models/glm4_moe.py +736 -0
  1108. vllm/model_executor/models/glm4_moe_mtp.py +359 -0
  1109. vllm/model_executor/models/glm4v.py +783 -0
  1110. vllm/model_executor/models/gpt2.py +397 -0
  1111. vllm/model_executor/models/gpt_bigcode.py +339 -0
  1112. vllm/model_executor/models/gpt_j.py +346 -0
  1113. vllm/model_executor/models/gpt_neox.py +340 -0
  1114. vllm/model_executor/models/gpt_oss.py +744 -0
  1115. vllm/model_executor/models/granite.py +475 -0
  1116. vllm/model_executor/models/granite_speech.py +912 -0
  1117. vllm/model_executor/models/granitemoe.py +560 -0
  1118. vllm/model_executor/models/granitemoehybrid.py +703 -0
  1119. vllm/model_executor/models/granitemoeshared.py +328 -0
  1120. vllm/model_executor/models/gritlm.py +243 -0
  1121. vllm/model_executor/models/grok1.py +554 -0
  1122. vllm/model_executor/models/h2ovl.py +554 -0
  1123. vllm/model_executor/models/hunyuan_v1.py +1040 -0
  1124. vllm/model_executor/models/hunyuan_vision.py +1034 -0
  1125. vllm/model_executor/models/hyperclovax_vision.py +1164 -0
  1126. vllm/model_executor/models/idefics2_vision_model.py +427 -0
  1127. vllm/model_executor/models/idefics3.py +716 -0
  1128. vllm/model_executor/models/interfaces.py +1179 -0
  1129. vllm/model_executor/models/interfaces_base.py +228 -0
  1130. vllm/model_executor/models/intern_vit.py +454 -0
  1131. vllm/model_executor/models/internlm2.py +453 -0
  1132. vllm/model_executor/models/internlm2_ve.py +139 -0
  1133. vllm/model_executor/models/interns1.py +828 -0
  1134. vllm/model_executor/models/interns1_vit.py +433 -0
  1135. vllm/model_executor/models/internvl.py +1450 -0
  1136. vllm/model_executor/models/jais.py +397 -0
  1137. vllm/model_executor/models/jais2.py +529 -0
  1138. vllm/model_executor/models/jamba.py +609 -0
  1139. vllm/model_executor/models/jina_vl.py +147 -0
  1140. vllm/model_executor/models/keye.py +1706 -0
  1141. vllm/model_executor/models/keye_vl1_5.py +726 -0
  1142. vllm/model_executor/models/kimi_linear.py +658 -0
  1143. vllm/model_executor/models/kimi_vl.py +576 -0
  1144. vllm/model_executor/models/lfm2.py +515 -0
  1145. vllm/model_executor/models/lfm2_moe.py +745 -0
  1146. vllm/model_executor/models/lightonocr.py +195 -0
  1147. vllm/model_executor/models/llama.py +700 -0
  1148. vllm/model_executor/models/llama4.py +856 -0
  1149. vllm/model_executor/models/llama4_eagle.py +225 -0
  1150. vllm/model_executor/models/llama_eagle.py +213 -0
  1151. vllm/model_executor/models/llama_eagle3.py +375 -0
  1152. vllm/model_executor/models/llava.py +840 -0
  1153. vllm/model_executor/models/llava_next.py +581 -0
  1154. vllm/model_executor/models/llava_next_video.py +465 -0
  1155. vllm/model_executor/models/llava_onevision.py +921 -0
  1156. vllm/model_executor/models/longcat_flash.py +743 -0
  1157. vllm/model_executor/models/longcat_flash_mtp.py +349 -0
  1158. vllm/model_executor/models/mamba.py +276 -0
  1159. vllm/model_executor/models/mamba2.py +288 -0
  1160. vllm/model_executor/models/medusa.py +179 -0
  1161. vllm/model_executor/models/midashenglm.py +826 -0
  1162. vllm/model_executor/models/mimo.py +188 -0
  1163. vllm/model_executor/models/mimo_mtp.py +294 -0
  1164. vllm/model_executor/models/minicpm.py +656 -0
  1165. vllm/model_executor/models/minicpm3.py +233 -0
  1166. vllm/model_executor/models/minicpm_eagle.py +385 -0
  1167. vllm/model_executor/models/minicpmo.py +768 -0
  1168. vllm/model_executor/models/minicpmv.py +1742 -0
  1169. vllm/model_executor/models/minimax_m2.py +550 -0
  1170. vllm/model_executor/models/minimax_text_01.py +1007 -0
  1171. vllm/model_executor/models/minimax_vl_01.py +394 -0
  1172. vllm/model_executor/models/mistral3.py +635 -0
  1173. vllm/model_executor/models/mistral_large_3.py +63 -0
  1174. vllm/model_executor/models/mistral_large_3_eagle.py +136 -0
  1175. vllm/model_executor/models/mixtral.py +598 -0
  1176. vllm/model_executor/models/mllama4.py +1149 -0
  1177. vllm/model_executor/models/mlp_speculator.py +235 -0
  1178. vllm/model_executor/models/modernbert.py +451 -0
  1179. vllm/model_executor/models/module_mapping.py +74 -0
  1180. vllm/model_executor/models/molmo.py +1550 -0
  1181. vllm/model_executor/models/moonvit.py +686 -0
  1182. vllm/model_executor/models/mpt.py +335 -0
  1183. vllm/model_executor/models/nano_nemotron_vl.py +1730 -0
  1184. vllm/model_executor/models/nemotron.py +499 -0
  1185. vllm/model_executor/models/nemotron_h.py +900 -0
  1186. vllm/model_executor/models/nemotron_nas.py +471 -0
  1187. vllm/model_executor/models/nemotron_vl.py +651 -0
  1188. vllm/model_executor/models/nvlm_d.py +216 -0
  1189. vllm/model_executor/models/olmo.py +412 -0
  1190. vllm/model_executor/models/olmo2.py +454 -0
  1191. vllm/model_executor/models/olmoe.py +493 -0
  1192. vllm/model_executor/models/opencua.py +262 -0
  1193. vllm/model_executor/models/openpangu.py +1049 -0
  1194. vllm/model_executor/models/openpangu_mtp.py +265 -0
  1195. vllm/model_executor/models/opt.py +426 -0
  1196. vllm/model_executor/models/orion.py +365 -0
  1197. vllm/model_executor/models/ouro.py +507 -0
  1198. vllm/model_executor/models/ovis.py +557 -0
  1199. vllm/model_executor/models/ovis2_5.py +661 -0
  1200. vllm/model_executor/models/paddleocr_vl.py +1300 -0
  1201. vllm/model_executor/models/paligemma.py +408 -0
  1202. vllm/model_executor/models/persimmon.py +373 -0
  1203. vllm/model_executor/models/phi.py +363 -0
  1204. vllm/model_executor/models/phi3.py +18 -0
  1205. vllm/model_executor/models/phi3v.py +729 -0
  1206. vllm/model_executor/models/phi4mm.py +1251 -0
  1207. vllm/model_executor/models/phi4mm_audio.py +1296 -0
  1208. vllm/model_executor/models/phi4mm_utils.py +1907 -0
  1209. vllm/model_executor/models/phimoe.py +669 -0
  1210. vllm/model_executor/models/pixtral.py +1379 -0
  1211. vllm/model_executor/models/plamo2.py +965 -0
  1212. vllm/model_executor/models/plamo3.py +440 -0
  1213. vllm/model_executor/models/qwen.py +365 -0
  1214. vllm/model_executor/models/qwen2.py +600 -0
  1215. vllm/model_executor/models/qwen2_5_omni_thinker.py +1219 -0
  1216. vllm/model_executor/models/qwen2_5_vl.py +1569 -0
  1217. vllm/model_executor/models/qwen2_audio.py +471 -0
  1218. vllm/model_executor/models/qwen2_moe.py +597 -0
  1219. vllm/model_executor/models/qwen2_rm.py +123 -0
  1220. vllm/model_executor/models/qwen2_vl.py +1568 -0
  1221. vllm/model_executor/models/qwen3.py +331 -0
  1222. vllm/model_executor/models/qwen3_moe.py +751 -0
  1223. vllm/model_executor/models/qwen3_next.py +1395 -0
  1224. vllm/model_executor/models/qwen3_next_mtp.py +296 -0
  1225. vllm/model_executor/models/qwen3_omni_moe_thinker.py +1793 -0
  1226. vllm/model_executor/models/qwen3_vl.py +2092 -0
  1227. vllm/model_executor/models/qwen3_vl_moe.py +474 -0
  1228. vllm/model_executor/models/qwen_vl.py +801 -0
  1229. vllm/model_executor/models/radio.py +555 -0
  1230. vllm/model_executor/models/registry.py +1189 -0
  1231. vllm/model_executor/models/roberta.py +259 -0
  1232. vllm/model_executor/models/rvl.py +107 -0
  1233. vllm/model_executor/models/seed_oss.py +492 -0
  1234. vllm/model_executor/models/siglip.py +1244 -0
  1235. vllm/model_executor/models/siglip2navit.py +658 -0
  1236. vllm/model_executor/models/skyworkr1v.py +951 -0
  1237. vllm/model_executor/models/smolvlm.py +38 -0
  1238. vllm/model_executor/models/solar.py +484 -0
  1239. vllm/model_executor/models/stablelm.py +354 -0
  1240. vllm/model_executor/models/starcoder2.py +365 -0
  1241. vllm/model_executor/models/step3_text.py +554 -0
  1242. vllm/model_executor/models/step3_vl.py +1147 -0
  1243. vllm/model_executor/models/swin.py +514 -0
  1244. vllm/model_executor/models/tarsier.py +617 -0
  1245. vllm/model_executor/models/telechat2.py +153 -0
  1246. vllm/model_executor/models/teleflm.py +78 -0
  1247. vllm/model_executor/models/terratorch.py +318 -0
  1248. vllm/model_executor/models/transformers/__init__.py +127 -0
  1249. vllm/model_executor/models/transformers/base.py +518 -0
  1250. vllm/model_executor/models/transformers/causal.py +65 -0
  1251. vllm/model_executor/models/transformers/legacy.py +90 -0
  1252. vllm/model_executor/models/transformers/moe.py +325 -0
  1253. vllm/model_executor/models/transformers/multimodal.py +411 -0
  1254. vllm/model_executor/models/transformers/pooling.py +119 -0
  1255. vllm/model_executor/models/transformers/utils.py +213 -0
  1256. vllm/model_executor/models/ultravox.py +766 -0
  1257. vllm/model_executor/models/utils.py +832 -0
  1258. vllm/model_executor/models/vision.py +546 -0
  1259. vllm/model_executor/models/voxtral.py +841 -0
  1260. vllm/model_executor/models/whisper.py +971 -0
  1261. vllm/model_executor/models/zamba2.py +979 -0
  1262. vllm/model_executor/parameter.py +642 -0
  1263. vllm/model_executor/utils.py +119 -0
  1264. vllm/model_executor/warmup/__init__.py +0 -0
  1265. vllm/model_executor/warmup/deep_gemm_warmup.py +314 -0
  1266. vllm/model_executor/warmup/kernel_warmup.py +98 -0
  1267. vllm/multimodal/__init__.py +40 -0
  1268. vllm/multimodal/audio.py +147 -0
  1269. vllm/multimodal/base.py +56 -0
  1270. vllm/multimodal/cache.py +823 -0
  1271. vllm/multimodal/evs.py +294 -0
  1272. vllm/multimodal/hasher.py +120 -0
  1273. vllm/multimodal/image.py +142 -0
  1274. vllm/multimodal/inputs.py +1089 -0
  1275. vllm/multimodal/parse.py +565 -0
  1276. vllm/multimodal/processing.py +2240 -0
  1277. vllm/multimodal/profiling.py +351 -0
  1278. vllm/multimodal/registry.py +357 -0
  1279. vllm/multimodal/utils.py +513 -0
  1280. vllm/multimodal/video.py +340 -0
  1281. vllm/outputs.py +345 -0
  1282. vllm/platforms/__init__.py +277 -0
  1283. vllm/platforms/cpu.py +421 -0
  1284. vllm/platforms/cuda.py +618 -0
  1285. vllm/platforms/interface.py +695 -0
  1286. vllm/platforms/rocm.py +564 -0
  1287. vllm/platforms/tpu.py +295 -0
  1288. vllm/platforms/xpu.py +277 -0
  1289. vllm/plugins/__init__.py +81 -0
  1290. vllm/plugins/io_processors/__init__.py +68 -0
  1291. vllm/plugins/io_processors/interface.py +77 -0
  1292. vllm/plugins/lora_resolvers/__init__.py +0 -0
  1293. vllm/plugins/lora_resolvers/filesystem_resolver.py +52 -0
  1294. vllm/pooling_params.py +230 -0
  1295. vllm/profiler/__init__.py +0 -0
  1296. vllm/profiler/layerwise_profile.py +392 -0
  1297. vllm/profiler/utils.py +151 -0
  1298. vllm/profiler/wrapper.py +241 -0
  1299. vllm/py.typed +2 -0
  1300. vllm/ray/__init__.py +0 -0
  1301. vllm/ray/lazy_utils.py +30 -0
  1302. vllm/ray/ray_env.py +79 -0
  1303. vllm/reasoning/__init__.py +96 -0
  1304. vllm/reasoning/abs_reasoning_parsers.py +318 -0
  1305. vllm/reasoning/basic_parsers.py +175 -0
  1306. vllm/reasoning/deepseek_r1_reasoning_parser.py +67 -0
  1307. vllm/reasoning/deepseek_v3_reasoning_parser.py +67 -0
  1308. vllm/reasoning/ernie45_reasoning_parser.py +165 -0
  1309. vllm/reasoning/glm4_moe_reasoning_parser.py +171 -0
  1310. vllm/reasoning/gptoss_reasoning_parser.py +173 -0
  1311. vllm/reasoning/granite_reasoning_parser.py +363 -0
  1312. vllm/reasoning/holo2_reasoning_parser.py +88 -0
  1313. vllm/reasoning/hunyuan_a13b_reasoning_parser.py +237 -0
  1314. vllm/reasoning/identity_reasoning_parser.py +63 -0
  1315. vllm/reasoning/minimax_m2_reasoning_parser.py +110 -0
  1316. vllm/reasoning/mistral_reasoning_parser.py +154 -0
  1317. vllm/reasoning/olmo3_reasoning_parser.py +302 -0
  1318. vllm/reasoning/qwen3_reasoning_parser.py +67 -0
  1319. vllm/reasoning/seedoss_reasoning_parser.py +27 -0
  1320. vllm/reasoning/step3_reasoning_parser.py +107 -0
  1321. vllm/sampling_params.py +597 -0
  1322. vllm/scalar_type.py +355 -0
  1323. vllm/scripts.py +17 -0
  1324. vllm/sequence.py +98 -0
  1325. vllm/tasks.py +13 -0
  1326. vllm/third_party/__init__.py +0 -0
  1327. vllm/third_party/pynvml.py +6140 -0
  1328. vllm/tokenizers/__init__.py +20 -0
  1329. vllm/tokenizers/deepseek_v32.py +175 -0
  1330. vllm/tokenizers/deepseek_v32_encoding.py +459 -0
  1331. vllm/tokenizers/detokenizer_utils.py +198 -0
  1332. vllm/tokenizers/hf.py +119 -0
  1333. vllm/tokenizers/mistral.py +567 -0
  1334. vllm/tokenizers/protocol.py +114 -0
  1335. vllm/tokenizers/registry.py +233 -0
  1336. vllm/tool_parsers/__init__.py +150 -0
  1337. vllm/tool_parsers/abstract_tool_parser.py +273 -0
  1338. vllm/tool_parsers/deepseekv31_tool_parser.py +388 -0
  1339. vllm/tool_parsers/deepseekv32_tool_parser.py +591 -0
  1340. vllm/tool_parsers/deepseekv3_tool_parser.py +390 -0
  1341. vllm/tool_parsers/ernie45_tool_parser.py +210 -0
  1342. vllm/tool_parsers/gigachat3_tool_parser.py +190 -0
  1343. vllm/tool_parsers/glm4_moe_tool_parser.py +200 -0
  1344. vllm/tool_parsers/granite_20b_fc_tool_parser.py +273 -0
  1345. vllm/tool_parsers/granite_tool_parser.py +253 -0
  1346. vllm/tool_parsers/hermes_tool_parser.py +495 -0
  1347. vllm/tool_parsers/hunyuan_a13b_tool_parser.py +420 -0
  1348. vllm/tool_parsers/internlm2_tool_parser.py +227 -0
  1349. vllm/tool_parsers/jamba_tool_parser.py +323 -0
  1350. vllm/tool_parsers/kimi_k2_tool_parser.py +590 -0
  1351. vllm/tool_parsers/llama4_pythonic_tool_parser.py +341 -0
  1352. vllm/tool_parsers/llama_tool_parser.py +324 -0
  1353. vllm/tool_parsers/longcat_tool_parser.py +37 -0
  1354. vllm/tool_parsers/minimax_m2_tool_parser.py +643 -0
  1355. vllm/tool_parsers/minimax_tool_parser.py +849 -0
  1356. vllm/tool_parsers/mistral_tool_parser.py +585 -0
  1357. vllm/tool_parsers/olmo3_tool_parser.py +366 -0
  1358. vllm/tool_parsers/openai_tool_parser.py +102 -0
  1359. vllm/tool_parsers/phi4mini_tool_parser.py +120 -0
  1360. vllm/tool_parsers/pythonic_tool_parser.py +332 -0
  1361. vllm/tool_parsers/qwen3coder_tool_parser.py +781 -0
  1362. vllm/tool_parsers/qwen3xml_tool_parser.py +1316 -0
  1363. vllm/tool_parsers/seed_oss_tool_parser.py +744 -0
  1364. vllm/tool_parsers/step3_tool_parser.py +303 -0
  1365. vllm/tool_parsers/utils.py +229 -0
  1366. vllm/tool_parsers/xlam_tool_parser.py +556 -0
  1367. vllm/tracing.py +135 -0
  1368. vllm/transformers_utils/__init__.py +26 -0
  1369. vllm/transformers_utils/chat_templates/__init__.py +5 -0
  1370. vllm/transformers_utils/chat_templates/registry.py +73 -0
  1371. vllm/transformers_utils/chat_templates/template_basic.jinja +3 -0
  1372. vllm/transformers_utils/chat_templates/template_blip2.jinja +11 -0
  1373. vllm/transformers_utils/chat_templates/template_chatml.jinja +10 -0
  1374. vllm/transformers_utils/chat_templates/template_deepseek_ocr.jinja +14 -0
  1375. vllm/transformers_utils/chat_templates/template_deepseek_vl2.jinja +23 -0
  1376. vllm/transformers_utils/chat_templates/template_fuyu.jinja +3 -0
  1377. vllm/transformers_utils/chat_templates/template_minicpmv45.jinja +93 -0
  1378. vllm/transformers_utils/config.py +1144 -0
  1379. vllm/transformers_utils/config_parser_base.py +20 -0
  1380. vllm/transformers_utils/configs/__init__.py +102 -0
  1381. vllm/transformers_utils/configs/afmoe.py +87 -0
  1382. vllm/transformers_utils/configs/arctic.py +216 -0
  1383. vllm/transformers_utils/configs/bagel.py +53 -0
  1384. vllm/transformers_utils/configs/chatglm.py +75 -0
  1385. vllm/transformers_utils/configs/deepseek_vl2.py +126 -0
  1386. vllm/transformers_utils/configs/dotsocr.py +71 -0
  1387. vllm/transformers_utils/configs/eagle.py +90 -0
  1388. vllm/transformers_utils/configs/falcon.py +89 -0
  1389. vllm/transformers_utils/configs/flex_olmo.py +82 -0
  1390. vllm/transformers_utils/configs/hunyuan_vl.py +322 -0
  1391. vllm/transformers_utils/configs/jais.py +243 -0
  1392. vllm/transformers_utils/configs/kimi_linear.py +148 -0
  1393. vllm/transformers_utils/configs/kimi_vl.py +38 -0
  1394. vllm/transformers_utils/configs/lfm2_moe.py +163 -0
  1395. vllm/transformers_utils/configs/medusa.py +65 -0
  1396. vllm/transformers_utils/configs/midashenglm.py +103 -0
  1397. vllm/transformers_utils/configs/mistral.py +235 -0
  1398. vllm/transformers_utils/configs/mlp_speculator.py +69 -0
  1399. vllm/transformers_utils/configs/moonvit.py +33 -0
  1400. vllm/transformers_utils/configs/nemotron.py +220 -0
  1401. vllm/transformers_utils/configs/nemotron_h.py +284 -0
  1402. vllm/transformers_utils/configs/olmo3.py +83 -0
  1403. vllm/transformers_utils/configs/ovis.py +182 -0
  1404. vllm/transformers_utils/configs/qwen3_next.py +277 -0
  1405. vllm/transformers_utils/configs/radio.py +89 -0
  1406. vllm/transformers_utils/configs/speculators/__init__.py +2 -0
  1407. vllm/transformers_utils/configs/speculators/algos.py +38 -0
  1408. vllm/transformers_utils/configs/speculators/base.py +114 -0
  1409. vllm/transformers_utils/configs/step3_vl.py +178 -0
  1410. vllm/transformers_utils/configs/tarsier2.py +24 -0
  1411. vllm/transformers_utils/configs/ultravox.py +120 -0
  1412. vllm/transformers_utils/dynamic_module.py +59 -0
  1413. vllm/transformers_utils/gguf_utils.py +280 -0
  1414. vllm/transformers_utils/processor.py +424 -0
  1415. vllm/transformers_utils/processors/__init__.py +25 -0
  1416. vllm/transformers_utils/processors/bagel.py +73 -0
  1417. vllm/transformers_utils/processors/deepseek_ocr.py +438 -0
  1418. vllm/transformers_utils/processors/deepseek_vl2.py +406 -0
  1419. vllm/transformers_utils/processors/hunyuan_vl.py +233 -0
  1420. vllm/transformers_utils/processors/hunyuan_vl_image.py +477 -0
  1421. vllm/transformers_utils/processors/ovis.py +453 -0
  1422. vllm/transformers_utils/processors/ovis2_5.py +468 -0
  1423. vllm/transformers_utils/repo_utils.py +287 -0
  1424. vllm/transformers_utils/runai_utils.py +102 -0
  1425. vllm/transformers_utils/s3_utils.py +95 -0
  1426. vllm/transformers_utils/tokenizer.py +127 -0
  1427. vllm/transformers_utils/tokenizer_base.py +33 -0
  1428. vllm/transformers_utils/utils.py +112 -0
  1429. vllm/triton_utils/__init__.py +20 -0
  1430. vllm/triton_utils/importing.py +103 -0
  1431. vllm/usage/__init__.py +0 -0
  1432. vllm/usage/usage_lib.py +294 -0
  1433. vllm/utils/__init__.py +66 -0
  1434. vllm/utils/argparse_utils.py +492 -0
  1435. vllm/utils/async_utils.py +310 -0
  1436. vllm/utils/cache.py +214 -0
  1437. vllm/utils/collection_utils.py +112 -0
  1438. vllm/utils/counter.py +45 -0
  1439. vllm/utils/deep_gemm.py +400 -0
  1440. vllm/utils/flashinfer.py +528 -0
  1441. vllm/utils/func_utils.py +236 -0
  1442. vllm/utils/gc_utils.py +151 -0
  1443. vllm/utils/hashing.py +117 -0
  1444. vllm/utils/import_utils.py +449 -0
  1445. vllm/utils/jsontree.py +158 -0
  1446. vllm/utils/math_utils.py +32 -0
  1447. vllm/utils/mem_constants.py +13 -0
  1448. vllm/utils/mem_utils.py +232 -0
  1449. vllm/utils/nccl.py +64 -0
  1450. vllm/utils/network_utils.py +331 -0
  1451. vllm/utils/nvtx_pytorch_hooks.py +286 -0
  1452. vllm/utils/platform_utils.py +59 -0
  1453. vllm/utils/profiling.py +56 -0
  1454. vllm/utils/registry.py +51 -0
  1455. vllm/utils/serial_utils.py +214 -0
  1456. vllm/utils/system_utils.py +269 -0
  1457. vllm/utils/tensor_schema.py +255 -0
  1458. vllm/utils/torch_utils.py +648 -0
  1459. vllm/v1/__init__.py +0 -0
  1460. vllm/v1/attention/__init__.py +0 -0
  1461. vllm/v1/attention/backends/__init__.py +0 -0
  1462. vllm/v1/attention/backends/cpu_attn.py +497 -0
  1463. vllm/v1/attention/backends/flash_attn.py +1051 -0
  1464. vllm/v1/attention/backends/flashinfer.py +1575 -0
  1465. vllm/v1/attention/backends/flex_attention.py +1028 -0
  1466. vllm/v1/attention/backends/gdn_attn.py +375 -0
  1467. vllm/v1/attention/backends/linear_attn.py +77 -0
  1468. vllm/v1/attention/backends/mamba1_attn.py +159 -0
  1469. vllm/v1/attention/backends/mamba2_attn.py +348 -0
  1470. vllm/v1/attention/backends/mamba_attn.py +117 -0
  1471. vllm/v1/attention/backends/mla/__init__.py +0 -0
  1472. vllm/v1/attention/backends/mla/aiter_triton_mla.py +74 -0
  1473. vllm/v1/attention/backends/mla/common.py +2114 -0
  1474. vllm/v1/attention/backends/mla/cutlass_mla.py +278 -0
  1475. vllm/v1/attention/backends/mla/flashattn_mla.py +342 -0
  1476. vllm/v1/attention/backends/mla/flashinfer_mla.py +174 -0
  1477. vllm/v1/attention/backends/mla/flashmla.py +317 -0
  1478. vllm/v1/attention/backends/mla/flashmla_sparse.py +1020 -0
  1479. vllm/v1/attention/backends/mla/indexer.py +345 -0
  1480. vllm/v1/attention/backends/mla/rocm_aiter_mla.py +275 -0
  1481. vllm/v1/attention/backends/mla/rocm_aiter_mla_sparse.py +325 -0
  1482. vllm/v1/attention/backends/mla/triton_mla.py +171 -0
  1483. vllm/v1/attention/backends/pallas.py +436 -0
  1484. vllm/v1/attention/backends/rocm_aiter_fa.py +1000 -0
  1485. vllm/v1/attention/backends/rocm_aiter_unified_attn.py +206 -0
  1486. vllm/v1/attention/backends/rocm_attn.py +359 -0
  1487. vllm/v1/attention/backends/short_conv_attn.py +104 -0
  1488. vllm/v1/attention/backends/tree_attn.py +428 -0
  1489. vllm/v1/attention/backends/triton_attn.py +497 -0
  1490. vllm/v1/attention/backends/utils.py +1212 -0
  1491. vllm/v1/core/__init__.py +0 -0
  1492. vllm/v1/core/block_pool.py +485 -0
  1493. vllm/v1/core/encoder_cache_manager.py +402 -0
  1494. vllm/v1/core/kv_cache_coordinator.py +570 -0
  1495. vllm/v1/core/kv_cache_manager.py +419 -0
  1496. vllm/v1/core/kv_cache_metrics.py +96 -0
  1497. vllm/v1/core/kv_cache_utils.py +1476 -0
  1498. vllm/v1/core/sched/__init__.py +0 -0
  1499. vllm/v1/core/sched/async_scheduler.py +68 -0
  1500. vllm/v1/core/sched/interface.py +189 -0
  1501. vllm/v1/core/sched/output.py +230 -0
  1502. vllm/v1/core/sched/request_queue.py +217 -0
  1503. vllm/v1/core/sched/scheduler.py +1826 -0
  1504. vllm/v1/core/sched/utils.py +64 -0
  1505. vllm/v1/core/single_type_kv_cache_manager.py +801 -0
  1506. vllm/v1/cudagraph_dispatcher.py +183 -0
  1507. vllm/v1/engine/__init__.py +217 -0
  1508. vllm/v1/engine/async_llm.py +866 -0
  1509. vllm/v1/engine/coordinator.py +377 -0
  1510. vllm/v1/engine/core.py +1455 -0
  1511. vllm/v1/engine/core_client.py +1416 -0
  1512. vllm/v1/engine/detokenizer.py +351 -0
  1513. vllm/v1/engine/exceptions.py +18 -0
  1514. vllm/v1/engine/input_processor.py +643 -0
  1515. vllm/v1/engine/llm_engine.py +414 -0
  1516. vllm/v1/engine/logprobs.py +189 -0
  1517. vllm/v1/engine/output_processor.py +659 -0
  1518. vllm/v1/engine/parallel_sampling.py +145 -0
  1519. vllm/v1/engine/processor.py +20 -0
  1520. vllm/v1/engine/utils.py +1068 -0
  1521. vllm/v1/executor/__init__.py +6 -0
  1522. vllm/v1/executor/abstract.py +352 -0
  1523. vllm/v1/executor/multiproc_executor.py +890 -0
  1524. vllm/v1/executor/ray_distributed_executor.py +8 -0
  1525. vllm/v1/executor/ray_executor.py +626 -0
  1526. vllm/v1/executor/ray_utils.py +465 -0
  1527. vllm/v1/executor/uniproc_executor.py +186 -0
  1528. vllm/v1/kv_cache_interface.py +404 -0
  1529. vllm/v1/kv_offload/__init__.py +0 -0
  1530. vllm/v1/kv_offload/abstract.py +161 -0
  1531. vllm/v1/kv_offload/arc_manager.py +237 -0
  1532. vllm/v1/kv_offload/backend.py +97 -0
  1533. vllm/v1/kv_offload/backends/__init__.py +0 -0
  1534. vllm/v1/kv_offload/backends/cpu.py +62 -0
  1535. vllm/v1/kv_offload/cpu.py +86 -0
  1536. vllm/v1/kv_offload/factory.py +56 -0
  1537. vllm/v1/kv_offload/lru_manager.py +139 -0
  1538. vllm/v1/kv_offload/mediums.py +39 -0
  1539. vllm/v1/kv_offload/spec.py +66 -0
  1540. vllm/v1/kv_offload/worker/__init__.py +0 -0
  1541. vllm/v1/kv_offload/worker/cpu_gpu.py +280 -0
  1542. vllm/v1/kv_offload/worker/worker.py +144 -0
  1543. vllm/v1/metrics/__init__.py +0 -0
  1544. vllm/v1/metrics/loggers.py +1305 -0
  1545. vllm/v1/metrics/prometheus.py +82 -0
  1546. vllm/v1/metrics/ray_wrappers.py +194 -0
  1547. vllm/v1/metrics/reader.py +257 -0
  1548. vllm/v1/metrics/stats.py +437 -0
  1549. vllm/v1/outputs.py +245 -0
  1550. vllm/v1/pool/__init__.py +0 -0
  1551. vllm/v1/pool/metadata.py +126 -0
  1552. vllm/v1/request.py +282 -0
  1553. vllm/v1/sample/__init__.py +0 -0
  1554. vllm/v1/sample/logits_processor/__init__.py +352 -0
  1555. vllm/v1/sample/logits_processor/builtin.py +278 -0
  1556. vllm/v1/sample/logits_processor/interface.py +106 -0
  1557. vllm/v1/sample/logits_processor/state.py +165 -0
  1558. vllm/v1/sample/metadata.py +44 -0
  1559. vllm/v1/sample/ops/__init__.py +0 -0
  1560. vllm/v1/sample/ops/bad_words.py +52 -0
  1561. vllm/v1/sample/ops/logprobs.py +25 -0
  1562. vllm/v1/sample/ops/penalties.py +57 -0
  1563. vllm/v1/sample/ops/topk_topp_sampler.py +384 -0
  1564. vllm/v1/sample/rejection_sampler.py +805 -0
  1565. vllm/v1/sample/sampler.py +319 -0
  1566. vllm/v1/sample/tpu/__init__.py +0 -0
  1567. vllm/v1/sample/tpu/metadata.py +120 -0
  1568. vllm/v1/sample/tpu/sampler.py +215 -0
  1569. vllm/v1/serial_utils.py +514 -0
  1570. vllm/v1/spec_decode/__init__.py +0 -0
  1571. vllm/v1/spec_decode/eagle.py +1331 -0
  1572. vllm/v1/spec_decode/medusa.py +73 -0
  1573. vllm/v1/spec_decode/metadata.py +66 -0
  1574. vllm/v1/spec_decode/metrics.py +225 -0
  1575. vllm/v1/spec_decode/ngram_proposer.py +291 -0
  1576. vllm/v1/spec_decode/suffix_decoding.py +101 -0
  1577. vllm/v1/spec_decode/utils.py +121 -0
  1578. vllm/v1/structured_output/__init__.py +353 -0
  1579. vllm/v1/structured_output/backend_guidance.py +265 -0
  1580. vllm/v1/structured_output/backend_lm_format_enforcer.py +177 -0
  1581. vllm/v1/structured_output/backend_outlines.py +324 -0
  1582. vllm/v1/structured_output/backend_types.py +136 -0
  1583. vllm/v1/structured_output/backend_xgrammar.py +378 -0
  1584. vllm/v1/structured_output/request.py +94 -0
  1585. vllm/v1/structured_output/utils.py +469 -0
  1586. vllm/v1/utils.py +414 -0
  1587. vllm/v1/worker/__init__.py +0 -0
  1588. vllm/v1/worker/block_table.py +343 -0
  1589. vllm/v1/worker/cp_utils.py +42 -0
  1590. vllm/v1/worker/cpu_model_runner.py +122 -0
  1591. vllm/v1/worker/cpu_worker.py +192 -0
  1592. vllm/v1/worker/dp_utils.py +240 -0
  1593. vllm/v1/worker/ec_connector_model_runner_mixin.py +87 -0
  1594. vllm/v1/worker/gpu/README.md +4 -0
  1595. vllm/v1/worker/gpu/__init__.py +0 -0
  1596. vllm/v1/worker/gpu/async_utils.py +98 -0
  1597. vllm/v1/worker/gpu/attn_utils.py +189 -0
  1598. vllm/v1/worker/gpu/block_table.py +314 -0
  1599. vllm/v1/worker/gpu/cudagraph_utils.py +259 -0
  1600. vllm/v1/worker/gpu/dp_utils.py +31 -0
  1601. vllm/v1/worker/gpu/input_batch.py +479 -0
  1602. vllm/v1/worker/gpu/metrics/__init__.py +0 -0
  1603. vllm/v1/worker/gpu/metrics/logits.py +42 -0
  1604. vllm/v1/worker/gpu/model_runner.py +1006 -0
  1605. vllm/v1/worker/gpu/sample/__init__.py +0 -0
  1606. vllm/v1/worker/gpu/sample/gumbel.py +101 -0
  1607. vllm/v1/worker/gpu/sample/logprob.py +167 -0
  1608. vllm/v1/worker/gpu/sample/metadata.py +192 -0
  1609. vllm/v1/worker/gpu/sample/min_p.py +51 -0
  1610. vllm/v1/worker/gpu/sample/output.py +14 -0
  1611. vllm/v1/worker/gpu/sample/penalties.py +155 -0
  1612. vllm/v1/worker/gpu/sample/sampler.py +87 -0
  1613. vllm/v1/worker/gpu/spec_decode/__init__.py +18 -0
  1614. vllm/v1/worker/gpu/spec_decode/eagle.py +565 -0
  1615. vllm/v1/worker/gpu/spec_decode/eagle_cudagraph.py +115 -0
  1616. vllm/v1/worker/gpu/spec_decode/rejection_sample.py +71 -0
  1617. vllm/v1/worker/gpu/states.py +316 -0
  1618. vllm/v1/worker/gpu/structured_outputs.py +76 -0
  1619. vllm/v1/worker/gpu_input_batch.py +990 -0
  1620. vllm/v1/worker/gpu_model_runner.py +5470 -0
  1621. vllm/v1/worker/gpu_ubatch_wrapper.py +472 -0
  1622. vllm/v1/worker/gpu_worker.py +955 -0
  1623. vllm/v1/worker/kv_connector_model_runner_mixin.py +302 -0
  1624. vllm/v1/worker/lora_model_runner_mixin.py +212 -0
  1625. vllm/v1/worker/tpu_input_batch.py +583 -0
  1626. vllm/v1/worker/tpu_model_runner.py +2191 -0
  1627. vllm/v1/worker/tpu_worker.py +352 -0
  1628. vllm/v1/worker/ubatch_utils.py +109 -0
  1629. vllm/v1/worker/ubatching.py +231 -0
  1630. vllm/v1/worker/utils.py +375 -0
  1631. vllm/v1/worker/worker_base.py +377 -0
  1632. vllm/v1/worker/workspace.py +253 -0
  1633. vllm/v1/worker/xpu_model_runner.py +48 -0
  1634. vllm/v1/worker/xpu_worker.py +174 -0
  1635. vllm/version.py +39 -0
  1636. vllm/vllm_flash_attn/.gitkeep +0 -0
  1637. vllm_cpu_avx512vnni-0.13.0.dist-info/METADATA +339 -0
  1638. vllm_cpu_avx512vnni-0.13.0.dist-info/RECORD +1641 -0
  1639. vllm_cpu_avx512vnni-0.13.0.dist-info/WHEEL +5 -0
  1640. vllm_cpu_avx512vnni-0.13.0.dist-info/entry_points.txt +5 -0
  1641. vllm_cpu_avx512vnni-0.13.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1903 @@
1
+ # SPDX-License-Identifier: Apache-2.0
2
+ # SPDX-FileCopyrightText: Copyright contributors to the vLLM project
3
+
4
+ import asyncio
5
+ import inspect
6
+ import json
7
+ from abc import ABC, abstractmethod
8
+ from collections import Counter, defaultdict, deque
9
+ from collections.abc import Awaitable, Callable, Iterable
10
+ from functools import cached_property, lru_cache, partial
11
+ from pathlib import Path
12
+ from typing import TYPE_CHECKING, Any, Generic, Literal, TypeAlias, TypeVar, cast
13
+
14
+ import jinja2
15
+ import jinja2.ext
16
+ import jinja2.meta
17
+ import jinja2.nodes
18
+ import jinja2.parser
19
+ import jinja2.sandbox
20
+ import transformers.utils.chat_template_utils as hf_chat_utils
21
+ from openai.types.chat import (
22
+ ChatCompletionAssistantMessageParam,
23
+ ChatCompletionContentPartImageParam,
24
+ ChatCompletionContentPartInputAudioParam,
25
+ ChatCompletionContentPartRefusalParam,
26
+ ChatCompletionContentPartTextParam,
27
+ ChatCompletionFunctionToolParam,
28
+ ChatCompletionMessageToolCallParam,
29
+ ChatCompletionToolMessageParam,
30
+ )
31
+ from openai.types.chat import (
32
+ ChatCompletionContentPartParam as OpenAIChatCompletionContentPartParam,
33
+ )
34
+ from openai.types.chat import (
35
+ ChatCompletionMessageParam as OpenAIChatCompletionMessageParam,
36
+ )
37
+ from openai.types.chat.chat_completion_content_part_input_audio_param import InputAudio
38
+ from openai.types.responses import ResponseInputImageParam
39
+ from openai_harmony import Message as OpenAIHarmonyMessage
40
+ from PIL import Image
41
+ from pydantic import BaseModel, ConfigDict, TypeAdapter
42
+ from transformers import PreTrainedTokenizer, PreTrainedTokenizerFast, ProcessorMixin
43
+
44
+ # pydantic needs the TypedDict from typing_extensions
45
+ from typing_extensions import Required, TypedDict
46
+
47
+ from vllm import envs
48
+ from vllm.config import ModelConfig
49
+ from vllm.logger import init_logger
50
+ from vllm.model_executor.models import SupportsMultiModal
51
+ from vllm.multimodal import MULTIMODAL_REGISTRY, MultiModalDataDict, MultiModalUUIDDict
52
+ from vllm.multimodal.utils import MEDIA_CONNECTOR_REGISTRY, MediaConnector
53
+ from vllm.tokenizers import TokenizerLike
54
+ from vllm.transformers_utils.chat_templates import get_chat_template_fallback_path
55
+ from vllm.transformers_utils.processor import cached_get_processor
56
+ from vllm.utils import random_uuid
57
+ from vllm.utils.collection_utils import is_list_of
58
+ from vllm.utils.func_utils import supports_kw
59
+ from vllm.utils.import_utils import LazyLoader
60
+
61
+ if TYPE_CHECKING:
62
+ import torch
63
+
64
+ from vllm.tokenizers.mistral import MistralTokenizer
65
+ else:
66
+ torch = LazyLoader("torch", globals(), "torch")
67
+
68
+ logger = init_logger(__name__)
69
+
70
+ MODALITY_PLACEHOLDERS_MAP = {
71
+ "image": "<##IMAGE##>",
72
+ "audio": "<##AUDIO##>",
73
+ "video": "<##VIDEO##>",
74
+ }
75
+
76
+
77
+ class AudioURL(TypedDict, total=False):
78
+ url: Required[str]
79
+ """
80
+ Either a URL of the audio or a data URL with base64 encoded audio data.
81
+ """
82
+
83
+
84
+ class ChatCompletionContentPartAudioParam(TypedDict, total=False):
85
+ audio_url: Required[AudioURL]
86
+
87
+ type: Required[Literal["audio_url"]]
88
+ """The type of the content part."""
89
+
90
+
91
+ class ChatCompletionContentPartImageEmbedsParam(TypedDict, total=False):
92
+ image_embeds: str | dict[str, str] | None
93
+ """
94
+ The image embeddings. It can be either:
95
+ - A single base64 string.
96
+ - A dictionary where each value is a base64 string.
97
+ """
98
+ type: Required[Literal["image_embeds"]]
99
+ """The type of the content part."""
100
+ uuid: str | None
101
+ """
102
+ User-provided UUID of a media. User must guarantee that it is properly
103
+ generated and unique for different medias.
104
+ """
105
+
106
+
107
+ class ChatCompletionContentPartAudioEmbedsParam(TypedDict, total=False):
108
+ audio_embeds: str | dict[str, str] | None
109
+ """
110
+ The audio embeddings. It can be either:
111
+ - A single base64 string representing a serialized torch tensor.
112
+ - A dictionary where each value is a base64 string.
113
+ """
114
+ type: Required[Literal["audio_embeds"]]
115
+ """The type of the content part."""
116
+ uuid: str | None
117
+ """
118
+ User-provided UUID of a media. User must guarantee that it is properly
119
+ generated and unique for different medias.
120
+ """
121
+
122
+
123
+ class VideoURL(TypedDict, total=False):
124
+ url: Required[str]
125
+ """
126
+ Either a URL of the video or a data URL with base64 encoded video data.
127
+ """
128
+
129
+
130
+ class ChatCompletionContentPartVideoParam(TypedDict, total=False):
131
+ video_url: Required[VideoURL]
132
+
133
+ type: Required[Literal["video_url"]]
134
+ """The type of the content part."""
135
+
136
+
137
+ class PILImage(BaseModel):
138
+ """
139
+ A PIL.Image.Image object.
140
+ """
141
+
142
+ image_pil: Image.Image
143
+ model_config = ConfigDict(arbitrary_types_allowed=True)
144
+
145
+
146
+ class CustomChatCompletionContentPILImageParam(TypedDict, total=False):
147
+ """A simpler version of the param that only accepts a PIL image.
148
+
149
+ Example:
150
+ {
151
+ "image_pil": ImageAsset('cherry_blossom').pil_image
152
+ }
153
+ """
154
+
155
+ image_pil: PILImage | None
156
+ uuid: str | None
157
+ """
158
+ User-provided UUID of a media. User must guarantee that it is properly
159
+ generated and unique for different medias.
160
+ """
161
+
162
+
163
+ class CustomChatCompletionContentSimpleImageParam(TypedDict, total=False):
164
+ """A simpler version of the param that only accepts a plain image_url.
165
+ This is supported by OpenAI API, although it is not documented.
166
+
167
+ Example:
168
+ {
169
+ "image_url": "https://example.com/image.jpg"
170
+ }
171
+ """
172
+
173
+ image_url: str | None
174
+ uuid: str | None
175
+ """
176
+ User-provided UUID of a media. User must guarantee that it is properly
177
+ generated and unique for different medias.
178
+ """
179
+
180
+
181
+ class CustomChatCompletionContentSimpleAudioParam(TypedDict, total=False):
182
+ """A simpler version of the param that only accepts a plain audio_url.
183
+
184
+ Example:
185
+ {
186
+ "audio_url": "https://example.com/audio.mp3"
187
+ }
188
+ """
189
+
190
+ audio_url: str | None
191
+
192
+
193
+ class CustomChatCompletionContentSimpleVideoParam(TypedDict, total=False):
194
+ """A simpler version of the param that only accepts a plain audio_url.
195
+
196
+ Example:
197
+ {
198
+ "video_url": "https://example.com/video.mp4"
199
+ }
200
+ """
201
+
202
+ video_url: str | None
203
+ uuid: str | None
204
+ """
205
+ User-provided UUID of a media. User must guarantee that it is properly
206
+ generated and unique for different medias.
207
+ """
208
+
209
+
210
+ class CustomThinkCompletionContentParam(TypedDict, total=False):
211
+ """A Think Completion Content Param that accepts a plain text and a boolean.
212
+
213
+ Example:
214
+ {
215
+ "thinking": "I am thinking about the answer",
216
+ "closed": True,
217
+ "type": "thinking"
218
+ }
219
+ """
220
+
221
+ thinking: Required[str]
222
+ """The thinking content."""
223
+
224
+ closed: bool
225
+ """Whether the thinking is closed."""
226
+
227
+ type: Required[Literal["thinking"]]
228
+ """The thinking type."""
229
+
230
+
231
+ ChatCompletionContentPartParam: TypeAlias = (
232
+ OpenAIChatCompletionContentPartParam
233
+ | ChatCompletionContentPartAudioParam
234
+ | ChatCompletionContentPartInputAudioParam
235
+ | ChatCompletionContentPartVideoParam
236
+ | ChatCompletionContentPartRefusalParam
237
+ | CustomChatCompletionContentPILImageParam
238
+ | CustomChatCompletionContentSimpleImageParam
239
+ | ChatCompletionContentPartImageEmbedsParam
240
+ | ChatCompletionContentPartAudioEmbedsParam
241
+ | CustomChatCompletionContentSimpleAudioParam
242
+ | CustomChatCompletionContentSimpleVideoParam
243
+ | str
244
+ | CustomThinkCompletionContentParam
245
+ )
246
+
247
+
248
+ class CustomChatCompletionMessageParam(TypedDict, total=False):
249
+ """Enables custom roles in the Chat Completion API."""
250
+
251
+ role: Required[str]
252
+ """The role of the message's author."""
253
+
254
+ content: str | list[ChatCompletionContentPartParam]
255
+ """The contents of the message."""
256
+
257
+ name: str
258
+ """An optional name for the participant.
259
+
260
+ Provides the model information to differentiate between participants of the
261
+ same role.
262
+ """
263
+
264
+ tool_call_id: str | None
265
+ """Tool call that this message is responding to."""
266
+
267
+ tool_calls: Iterable[ChatCompletionMessageToolCallParam] | None
268
+ """The tool calls generated by the model, such as function calls."""
269
+
270
+ reasoning: str | None
271
+ """The reasoning content for interleaved thinking."""
272
+
273
+ tools: list[ChatCompletionFunctionToolParam] | None
274
+ """The tools for developer role."""
275
+
276
+
277
+ ChatCompletionMessageParam: TypeAlias = (
278
+ OpenAIChatCompletionMessageParam
279
+ | CustomChatCompletionMessageParam
280
+ | OpenAIHarmonyMessage
281
+ )
282
+
283
+
284
+ # TODO: Make fields ReadOnly once mypy supports it
285
+ class ConversationMessage(TypedDict, total=False):
286
+ role: Required[str]
287
+ """The role of the message's author."""
288
+
289
+ content: str | None | list[dict[str, str]]
290
+ """The contents of the message"""
291
+
292
+ tool_call_id: str | None
293
+ """Tool call that this message is responding to."""
294
+
295
+ name: str | None
296
+ """The name of the function to call"""
297
+
298
+ tool_calls: Iterable[ChatCompletionMessageToolCallParam] | None
299
+ """The tool calls generated by the model, such as function calls."""
300
+
301
+ reasoning: str | None
302
+ """The reasoning content for interleaved thinking."""
303
+
304
+ reasoning_content: str | None
305
+ """Deprecated: The reasoning content for interleaved thinking."""
306
+
307
+ tools: list[ChatCompletionFunctionToolParam] | None
308
+ """The tools for developer role."""
309
+
310
+
311
+ # Passed in by user
312
+ ChatTemplateContentFormatOption = Literal["auto", "string", "openai"]
313
+
314
+ # Used internally
315
+ _ChatTemplateContentFormat = Literal["string", "openai"]
316
+
317
+
318
+ def _is_var_access(node: jinja2.nodes.Node, varname: str) -> bool:
319
+ if isinstance(node, jinja2.nodes.Name):
320
+ return node.ctx == "load" and node.name == varname
321
+
322
+ return False
323
+
324
+
325
+ def _is_attr_access(node: jinja2.nodes.Node, varname: str, key: str) -> bool:
326
+ if isinstance(node, jinja2.nodes.Getitem):
327
+ return (
328
+ _is_var_access(node.node, varname)
329
+ and isinstance(node.arg, jinja2.nodes.Const)
330
+ and node.arg.value == key
331
+ )
332
+
333
+ if isinstance(node, jinja2.nodes.Getattr):
334
+ return _is_var_access(node.node, varname) and node.attr == key
335
+
336
+ return False
337
+
338
+
339
+ def _is_var_or_elems_access(
340
+ node: jinja2.nodes.Node,
341
+ varname: str,
342
+ key: str | None = None,
343
+ ) -> bool:
344
+ if isinstance(node, jinja2.nodes.Filter):
345
+ return node.node is not None and _is_var_or_elems_access(
346
+ node.node, varname, key
347
+ )
348
+ if isinstance(node, jinja2.nodes.Test):
349
+ return _is_var_or_elems_access(node.node, varname, key)
350
+
351
+ if isinstance(node, jinja2.nodes.Getitem) and isinstance(
352
+ node.arg, jinja2.nodes.Slice
353
+ ):
354
+ return _is_var_or_elems_access(node.node, varname, key)
355
+
356
+ return _is_attr_access(node, varname, key) if key else _is_var_access(node, varname)
357
+
358
+
359
+ def _iter_nodes_assign_var_or_elems(root: jinja2.nodes.Node, varname: str):
360
+ # Global variable that is implicitly defined at the root
361
+ yield root, varname
362
+
363
+ # Iterative BFS
364
+ related_varnames = deque([varname])
365
+ while related_varnames:
366
+ related_varname = related_varnames.popleft()
367
+
368
+ for assign_ast in root.find_all(jinja2.nodes.Assign):
369
+ lhs = assign_ast.target
370
+ rhs = assign_ast.node
371
+
372
+ if _is_var_or_elems_access(rhs, related_varname):
373
+ assert isinstance(lhs, jinja2.nodes.Name)
374
+ yield assign_ast, lhs.name
375
+
376
+ # Avoid infinite looping for self-assignment
377
+ if lhs.name != related_varname:
378
+ related_varnames.append(lhs.name)
379
+
380
+
381
+ # NOTE: The proper way to handle this is to build a CFG so that we can handle
382
+ # the scope in which each variable is defined, but that is too complicated
383
+ def _iter_nodes_assign_messages_item(root: jinja2.nodes.Node):
384
+ messages_varnames = [
385
+ varname for _, varname in _iter_nodes_assign_var_or_elems(root, "messages")
386
+ ]
387
+
388
+ # Search for {%- for message in messages -%} loops
389
+ for loop_ast in root.find_all(jinja2.nodes.For):
390
+ loop_iter = loop_ast.iter
391
+ loop_target = loop_ast.target
392
+
393
+ for varname in messages_varnames:
394
+ if _is_var_or_elems_access(loop_iter, varname):
395
+ assert isinstance(loop_target, jinja2.nodes.Name)
396
+ yield loop_ast, loop_target.name
397
+ break
398
+
399
+
400
+ def _iter_nodes_assign_content_item(root: jinja2.nodes.Node):
401
+ message_varnames = [
402
+ varname for _, varname in _iter_nodes_assign_messages_item(root)
403
+ ]
404
+
405
+ # Search for {%- for content in message['content'] -%} loops
406
+ for loop_ast in root.find_all(jinja2.nodes.For):
407
+ loop_iter = loop_ast.iter
408
+ loop_target = loop_ast.target
409
+
410
+ for varname in message_varnames:
411
+ if _is_var_or_elems_access(loop_iter, varname, "content"):
412
+ assert isinstance(loop_target, jinja2.nodes.Name)
413
+ yield loop_ast, loop_target.name
414
+ break
415
+
416
+
417
+ def _try_extract_ast(chat_template: str) -> jinja2.nodes.Template | None:
418
+ try:
419
+ jinja_compiled = hf_chat_utils._compile_jinja_template(chat_template)
420
+ return jinja_compiled.environment.parse(chat_template)
421
+ except Exception:
422
+ logger.exception("Error when compiling Jinja template")
423
+ return None
424
+
425
+
426
+ @lru_cache(maxsize=32)
427
+ def _detect_content_format(
428
+ chat_template: str,
429
+ *,
430
+ default: _ChatTemplateContentFormat,
431
+ ) -> _ChatTemplateContentFormat:
432
+ jinja_ast = _try_extract_ast(chat_template)
433
+ if jinja_ast is None:
434
+ return default
435
+
436
+ try:
437
+ next(_iter_nodes_assign_content_item(jinja_ast))
438
+ except StopIteration:
439
+ return "string"
440
+ except Exception:
441
+ logger.exception("Error when parsing AST of Jinja template")
442
+ return default
443
+ else:
444
+ return "openai"
445
+
446
+
447
+ def resolve_mistral_chat_template(
448
+ chat_template: str | None,
449
+ **kwargs: Any,
450
+ ) -> str | None:
451
+ if chat_template is not None or kwargs.get("chat_template_kwargs") is not None:
452
+ raise ValueError(
453
+ "'chat_template' or 'chat_template_kwargs' cannot be overridden "
454
+ "for mistral tokenizer."
455
+ )
456
+
457
+ return None
458
+
459
+
460
+ _PROCESSOR_CHAT_TEMPLATES = dict[tuple[str, bool], str | None]()
461
+ """
462
+ Used in `_try_get_processor_chat_template` to avoid calling
463
+ `cached_get_processor` again if the processor fails to be loaded.
464
+
465
+ This is needed because `lru_cache` does not cache when an exception happens.
466
+ """
467
+
468
+
469
+ def _try_get_processor_chat_template(
470
+ tokenizer: PreTrainedTokenizer | PreTrainedTokenizerFast,
471
+ model_config: ModelConfig,
472
+ ) -> str | None:
473
+ cache_key = (tokenizer.name_or_path, model_config.trust_remote_code)
474
+ if cache_key in _PROCESSOR_CHAT_TEMPLATES:
475
+ return _PROCESSOR_CHAT_TEMPLATES[cache_key]
476
+
477
+ try:
478
+ processor = cached_get_processor(
479
+ tokenizer.name_or_path,
480
+ processor_cls=(
481
+ PreTrainedTokenizer,
482
+ PreTrainedTokenizerFast,
483
+ ProcessorMixin,
484
+ ),
485
+ trust_remote_code=model_config.trust_remote_code,
486
+ )
487
+ if (
488
+ isinstance(processor, ProcessorMixin)
489
+ and hasattr(processor, "chat_template")
490
+ and (chat_template := processor.chat_template) is not None
491
+ ):
492
+ _PROCESSOR_CHAT_TEMPLATES[cache_key] = chat_template
493
+ return chat_template
494
+ except Exception:
495
+ logger.debug(
496
+ "Failed to load AutoProcessor chat template for %s",
497
+ tokenizer.name_or_path,
498
+ exc_info=True,
499
+ )
500
+
501
+ _PROCESSOR_CHAT_TEMPLATES[cache_key] = None
502
+ return None
503
+
504
+
505
+ def resolve_hf_chat_template(
506
+ tokenizer: PreTrainedTokenizer | PreTrainedTokenizerFast,
507
+ chat_template: str | None,
508
+ tools: list[dict[str, Any]] | None,
509
+ *,
510
+ model_config: ModelConfig,
511
+ ) -> str | None:
512
+ # 1st priority: The given chat template
513
+ if chat_template is not None:
514
+ return chat_template
515
+
516
+ # 2nd priority: AutoProcessor chat template, unless tool calling is enabled
517
+ if tools is None:
518
+ chat_template = _try_get_processor_chat_template(tokenizer, model_config)
519
+ if chat_template is not None:
520
+ return chat_template
521
+
522
+ # 3rd priority: AutoTokenizer chat template
523
+ try:
524
+ return tokenizer.get_chat_template(chat_template, tools=tools)
525
+ except Exception:
526
+ logger.debug(
527
+ "Failed to load AutoTokenizer chat template for %s",
528
+ tokenizer.name_or_path,
529
+ exc_info=True,
530
+ )
531
+
532
+ # 4th priority: Predefined fallbacks
533
+ path = get_chat_template_fallback_path(
534
+ model_type=model_config.hf_config.model_type,
535
+ tokenizer_name_or_path=model_config.tokenizer,
536
+ )
537
+ if path is not None:
538
+ logger.info_once(
539
+ "Loading chat template fallback for %s as there isn't one "
540
+ "defined on HF Hub.",
541
+ tokenizer.name_or_path,
542
+ )
543
+ chat_template = load_chat_template(path)
544
+ else:
545
+ logger.debug_once(
546
+ "There is no chat template fallback for %s", tokenizer.name_or_path
547
+ )
548
+
549
+ return chat_template
550
+
551
+
552
+ def _resolve_chat_template_content_format(
553
+ chat_template: str | None,
554
+ tools: list[dict[str, Any]] | None,
555
+ tokenizer: TokenizerLike | None,
556
+ *,
557
+ model_config: ModelConfig,
558
+ ) -> _ChatTemplateContentFormat:
559
+ if isinstance(tokenizer, (PreTrainedTokenizer, PreTrainedTokenizerFast)):
560
+ hf_chat_template = resolve_hf_chat_template(
561
+ tokenizer,
562
+ chat_template=chat_template,
563
+ tools=tools,
564
+ model_config=model_config,
565
+ )
566
+ else:
567
+ hf_chat_template = None
568
+
569
+ jinja_text = (
570
+ hf_chat_template
571
+ if isinstance(hf_chat_template, str)
572
+ else load_chat_template(chat_template, is_literal=True)
573
+ )
574
+
575
+ detected_format = (
576
+ "string"
577
+ if jinja_text is None
578
+ else _detect_content_format(jinja_text, default="string")
579
+ )
580
+
581
+ return detected_format
582
+
583
+
584
+ @lru_cache
585
+ def _log_chat_template_content_format(
586
+ chat_template: str | None,
587
+ given_format: ChatTemplateContentFormatOption,
588
+ detected_format: ChatTemplateContentFormatOption,
589
+ ):
590
+ logger.info(
591
+ "Detected the chat template content format to be '%s'. "
592
+ "You can set `--chat-template-content-format` to override this.",
593
+ detected_format,
594
+ )
595
+
596
+ if given_format != "auto" and given_format != detected_format:
597
+ logger.warning(
598
+ "You specified `--chat-template-content-format %s` "
599
+ "which is different from the detected format '%s'. "
600
+ "If our automatic detection is incorrect, please consider "
601
+ "opening a GitHub issue so that we can improve it: "
602
+ "https://github.com/vllm-project/vllm/issues/new/choose",
603
+ given_format,
604
+ detected_format,
605
+ )
606
+
607
+
608
+ def resolve_chat_template_content_format(
609
+ chat_template: str | None,
610
+ tools: list[dict[str, Any]] | None,
611
+ given_format: ChatTemplateContentFormatOption,
612
+ tokenizer: TokenizerLike | None,
613
+ *,
614
+ model_config: ModelConfig,
615
+ ) -> _ChatTemplateContentFormat:
616
+ if given_format != "auto":
617
+ return given_format
618
+
619
+ detected_format = _resolve_chat_template_content_format(
620
+ chat_template,
621
+ tools,
622
+ tokenizer,
623
+ model_config=model_config,
624
+ )
625
+
626
+ _log_chat_template_content_format(
627
+ chat_template,
628
+ given_format=given_format,
629
+ detected_format=detected_format,
630
+ )
631
+
632
+ return detected_format
633
+
634
+
635
+ ModalityStr = Literal["image", "audio", "video", "image_embeds", "audio_embeds"]
636
+ _T = TypeVar("_T")
637
+
638
+
639
+ def _extract_embeds(tensors: list[torch.Tensor]):
640
+ if len(tensors) == 0:
641
+ return tensors
642
+
643
+ if len(tensors) == 1:
644
+ tensors[0]._is_single_item = True # type: ignore
645
+ return tensors[0] # To keep backwards compatibility for single item input
646
+
647
+ first_shape = tensors[0].shape
648
+ if all(t.shape == first_shape for t in tensors):
649
+ return torch.stack(tensors)
650
+
651
+ return tensors
652
+
653
+
654
+ def _get_embeds_data(items_by_modality: dict[str, list[Any]], modality: str):
655
+ embeds_key = f"{modality}_embeds"
656
+ embeds = items_by_modality[embeds_key]
657
+
658
+ if len(embeds) == 0:
659
+ return embeds
660
+ if is_list_of(embeds, torch.Tensor):
661
+ return _extract_embeds(embeds)
662
+ if is_list_of(embeds, dict):
663
+ if not embeds:
664
+ return {}
665
+
666
+ first_keys = set(embeds[0].keys())
667
+ if any(set(item.keys()) != first_keys for item in embeds[1:]):
668
+ raise ValueError(
669
+ "All dictionaries in the list of embeddings must have the same keys."
670
+ )
671
+
672
+ return {k: _extract_embeds([item[k] for item in embeds]) for k in first_keys}
673
+
674
+ return embeds
675
+
676
+
677
+ class BaseMultiModalItemTracker(ABC, Generic[_T]):
678
+ """
679
+ Tracks multi-modal items in a given request and ensures that the number
680
+ of multi-modal items in a given request does not exceed the configured
681
+ maximum per prompt.
682
+ """
683
+
684
+ def __init__(self, model_config: ModelConfig):
685
+ super().__init__()
686
+
687
+ self._model_config = model_config
688
+
689
+ self._items_by_modality = defaultdict[str, list[_T | None]](list)
690
+ self._uuids_by_modality = defaultdict[str, list[str | None]](list)
691
+
692
+ @property
693
+ def model_config(self) -> ModelConfig:
694
+ return self._model_config
695
+
696
+ @cached_property
697
+ def model_cls(self) -> type[SupportsMultiModal]:
698
+ from vllm.model_executor.model_loader import get_model_cls
699
+
700
+ model_cls = get_model_cls(self.model_config)
701
+ return cast(type[SupportsMultiModal], model_cls)
702
+
703
+ @property
704
+ def allowed_local_media_path(self):
705
+ return self._model_config.allowed_local_media_path
706
+
707
+ @property
708
+ def allowed_media_domains(self):
709
+ return self._model_config.allowed_media_domains
710
+
711
+ @property
712
+ def mm_registry(self):
713
+ return MULTIMODAL_REGISTRY
714
+
715
+ @cached_property
716
+ def mm_processor(self):
717
+ return self.mm_registry.create_processor(self.model_config)
718
+
719
+ def add(
720
+ self,
721
+ modality: ModalityStr,
722
+ item: _T | None,
723
+ uuid: str | None = None,
724
+ ) -> str | None:
725
+ """
726
+ Add a multi-modal item to the current prompt and returns the
727
+ placeholder string to use, if any.
728
+
729
+ An optional uuid can be added which serves as a unique identifier of the
730
+ media.
731
+ """
732
+ input_modality = modality.replace("_embeds", "")
733
+ num_items = len(self._items_by_modality[modality]) + 1
734
+
735
+ self.mm_processor.validate_num_items(input_modality, num_items)
736
+
737
+ self._items_by_modality[modality].append(item)
738
+ self._uuids_by_modality[modality].append(uuid)
739
+
740
+ return self.model_cls.get_placeholder_str(modality, num_items)
741
+
742
+ def all_mm_uuids(self) -> MultiModalUUIDDict | None:
743
+ if not self._items_by_modality:
744
+ return None
745
+
746
+ uuids_by_modality = dict(self._uuids_by_modality)
747
+ if "image" in uuids_by_modality and "image_embeds" in uuids_by_modality:
748
+ raise ValueError("Mixing raw image and embedding inputs is not allowed")
749
+ if "audio" in uuids_by_modality and "audio_embeds" in uuids_by_modality:
750
+ raise ValueError("Mixing raw audio and embedding inputs is not allowed")
751
+
752
+ mm_uuids = {}
753
+ if "image_embeds" in uuids_by_modality:
754
+ mm_uuids["image"] = uuids_by_modality["image_embeds"]
755
+ if "image" in uuids_by_modality:
756
+ mm_uuids["image"] = uuids_by_modality["image"] # UUIDs of images
757
+ if "audio_embeds" in uuids_by_modality:
758
+ mm_uuids["audio"] = uuids_by_modality["audio_embeds"]
759
+ if "audio" in uuids_by_modality:
760
+ mm_uuids["audio"] = uuids_by_modality["audio"] # UUIDs of audios
761
+ if "video" in uuids_by_modality:
762
+ mm_uuids["video"] = uuids_by_modality["video"] # UUIDs of videos
763
+
764
+ return mm_uuids
765
+
766
+ @abstractmethod
767
+ def create_parser(self) -> "BaseMultiModalContentParser":
768
+ raise NotImplementedError
769
+
770
+
771
+ class MultiModalItemTracker(BaseMultiModalItemTracker[object]):
772
+ def all_mm_data(self) -> MultiModalDataDict | None:
773
+ if not self._items_by_modality:
774
+ return None
775
+
776
+ items_by_modality = dict(self._items_by_modality)
777
+ if "image" in items_by_modality and "image_embeds" in items_by_modality:
778
+ raise ValueError("Mixing raw image and embedding inputs is not allowed")
779
+ if "audio" in items_by_modality and "audio_embeds" in items_by_modality:
780
+ raise ValueError("Mixing raw audio and embedding inputs is not allowed")
781
+
782
+ mm_inputs = {}
783
+ if "image_embeds" in items_by_modality:
784
+ mm_inputs["image"] = _get_embeds_data(items_by_modality, "image")
785
+ if "image" in items_by_modality:
786
+ mm_inputs["image"] = items_by_modality["image"] # A list of images
787
+ if "audio_embeds" in items_by_modality:
788
+ mm_inputs["audio"] = _get_embeds_data(items_by_modality, "audio")
789
+ if "audio" in items_by_modality:
790
+ mm_inputs["audio"] = items_by_modality["audio"] # A list of audios
791
+ if "video" in items_by_modality:
792
+ mm_inputs["video"] = items_by_modality["video"] # A list of videos
793
+
794
+ return mm_inputs
795
+
796
+ def create_parser(self) -> "BaseMultiModalContentParser":
797
+ return MultiModalContentParser(self)
798
+
799
+
800
+ class AsyncMultiModalItemTracker(BaseMultiModalItemTracker[Awaitable[object]]):
801
+ async def all_mm_data(self) -> MultiModalDataDict | None:
802
+ if not self._items_by_modality:
803
+ return None
804
+
805
+ coros_by_modality = {
806
+ modality: [item or asyncio.sleep(0) for item in items]
807
+ for modality, items in self._items_by_modality.items()
808
+ }
809
+ items_by_modality: dict[str, list[object | None]] = {
810
+ modality: await asyncio.gather(*coros)
811
+ for modality, coros in coros_by_modality.items()
812
+ }
813
+ if "image" in items_by_modality and "image_embeds" in items_by_modality:
814
+ raise ValueError("Mixing raw image and embedding inputs is not allowed")
815
+ if "audio" in items_by_modality and "audio_embeds" in items_by_modality:
816
+ raise ValueError("Mixing raw audio and embedding inputs is not allowed")
817
+
818
+ mm_inputs = {}
819
+ if "image_embeds" in items_by_modality:
820
+ mm_inputs["image"] = _get_embeds_data(items_by_modality, "image")
821
+ if "image" in items_by_modality:
822
+ mm_inputs["image"] = items_by_modality["image"] # A list of images
823
+ if "audio_embeds" in items_by_modality:
824
+ mm_inputs["audio"] = _get_embeds_data(items_by_modality, "audio")
825
+ if "audio" in items_by_modality:
826
+ mm_inputs["audio"] = items_by_modality["audio"] # A list of audios
827
+ if "video" in items_by_modality:
828
+ mm_inputs["video"] = items_by_modality["video"] # A list of videos
829
+
830
+ return mm_inputs
831
+
832
+ def create_parser(self) -> "BaseMultiModalContentParser":
833
+ return AsyncMultiModalContentParser(self)
834
+
835
+
836
+ class BaseMultiModalContentParser(ABC):
837
+ def __init__(self) -> None:
838
+ super().__init__()
839
+
840
+ # stores model placeholders list with corresponding
841
+ # general MM placeholder:
842
+ # {
843
+ # "<##IMAGE##>": ["<image>", "<image>", "<image>"],
844
+ # "<##AUDIO##>": ["<audio>", "<audio>"]
845
+ # }
846
+ self._placeholder_storage: dict[str, list] = defaultdict(list)
847
+
848
+ def _add_placeholder(self, modality: ModalityStr, placeholder: str | None):
849
+ mod_placeholder = MODALITY_PLACEHOLDERS_MAP[modality]
850
+ if placeholder:
851
+ self._placeholder_storage[mod_placeholder].append(placeholder)
852
+
853
+ def mm_placeholder_storage(self) -> dict[str, list]:
854
+ return dict(self._placeholder_storage)
855
+
856
+ @abstractmethod
857
+ def parse_image(self, image_url: str | None, uuid: str | None = None) -> None:
858
+ raise NotImplementedError
859
+
860
+ @abstractmethod
861
+ def parse_image_embeds(
862
+ self,
863
+ image_embeds: str | dict[str, str] | None,
864
+ uuid: str | None = None,
865
+ ) -> None:
866
+ raise NotImplementedError
867
+
868
+ @abstractmethod
869
+ def parse_image_pil(
870
+ self, image_pil: Image.Image | None, uuid: str | None = None
871
+ ) -> None:
872
+ raise NotImplementedError
873
+
874
+ @abstractmethod
875
+ def parse_audio(self, audio_url: str | None, uuid: str | None = None) -> None:
876
+ raise NotImplementedError
877
+
878
+ @abstractmethod
879
+ def parse_input_audio(
880
+ self, input_audio: InputAudio | None, uuid: str | None = None
881
+ ) -> None:
882
+ raise NotImplementedError
883
+
884
+ @abstractmethod
885
+ def parse_audio_embeds(
886
+ self,
887
+ audio_embeds: str | dict[str, str] | None,
888
+ uuid: str | None = None,
889
+ ) -> None:
890
+ raise NotImplementedError
891
+
892
+ @abstractmethod
893
+ def parse_video(self, video_url: str | None, uuid: str | None = None) -> None:
894
+ raise NotImplementedError
895
+
896
+
897
+ class MultiModalContentParser(BaseMultiModalContentParser):
898
+ def __init__(self, tracker: MultiModalItemTracker) -> None:
899
+ super().__init__()
900
+
901
+ self._tracker = tracker
902
+ multimodal_config = self._tracker.model_config.multimodal_config
903
+ media_io_kwargs = getattr(multimodal_config, "media_io_kwargs", None)
904
+
905
+ self._connector: MediaConnector = MEDIA_CONNECTOR_REGISTRY.load(
906
+ envs.VLLM_MEDIA_CONNECTOR,
907
+ media_io_kwargs=media_io_kwargs,
908
+ allowed_local_media_path=tracker.allowed_local_media_path,
909
+ allowed_media_domains=tracker.allowed_media_domains,
910
+ )
911
+
912
+ @property
913
+ def model_config(self) -> ModelConfig:
914
+ return self._tracker.model_config
915
+
916
+ def parse_image(self, image_url: str | None, uuid: str | None = None) -> None:
917
+ image = self._connector.fetch_image(image_url) if image_url else None
918
+
919
+ placeholder = self._tracker.add("image", image, uuid)
920
+ self._add_placeholder("image", placeholder)
921
+
922
+ def parse_image_embeds(
923
+ self,
924
+ image_embeds: str | dict[str, str] | None,
925
+ uuid: str | None = None,
926
+ ) -> None:
927
+ mm_config = self.model_config.get_multimodal_config()
928
+ if not mm_config.enable_mm_embeds:
929
+ raise ValueError(
930
+ "You must set `--enable-mm-embeds` to input `image_embeds`"
931
+ )
932
+
933
+ if isinstance(image_embeds, dict):
934
+ embeds = {
935
+ k: self._connector.fetch_image_embedding(v)
936
+ for k, v in image_embeds.items()
937
+ }
938
+ placeholder = self._tracker.add("image_embeds", embeds, uuid)
939
+
940
+ if isinstance(image_embeds, str):
941
+ embedding = self._connector.fetch_image_embedding(image_embeds)
942
+ placeholder = self._tracker.add("image_embeds", embedding, uuid)
943
+
944
+ if image_embeds is None:
945
+ placeholder = self._tracker.add("image_embeds", None, uuid)
946
+
947
+ self._add_placeholder("image", placeholder)
948
+
949
+ def parse_audio_embeds(
950
+ self,
951
+ audio_embeds: str | dict[str, str] | None,
952
+ uuid: str | None = None,
953
+ ) -> None:
954
+ mm_config = self.model_config.get_multimodal_config()
955
+ if not mm_config.enable_mm_embeds:
956
+ raise ValueError(
957
+ "You must set `--enable-mm-embeds` to input `audio_embeds`"
958
+ )
959
+
960
+ if isinstance(audio_embeds, dict):
961
+ embeds = {
962
+ k: self._connector.fetch_audio_embedding(v)
963
+ for k, v in audio_embeds.items()
964
+ }
965
+ placeholder = self._tracker.add("audio_embeds", embeds, uuid)
966
+ elif isinstance(audio_embeds, str):
967
+ embedding = self._connector.fetch_audio_embedding(audio_embeds)
968
+ placeholder = self._tracker.add("audio_embeds", embedding, uuid)
969
+ else:
970
+ placeholder = self._tracker.add("audio_embeds", None, uuid)
971
+
972
+ self._add_placeholder("audio", placeholder)
973
+
974
+ def parse_image_pil(
975
+ self, image_pil: Image.Image | None, uuid: str | None = None
976
+ ) -> None:
977
+ placeholder = self._tracker.add("image", image_pil, uuid)
978
+ self._add_placeholder("image", placeholder)
979
+
980
+ def parse_audio(self, audio_url: str | None, uuid: str | None = None) -> None:
981
+ audio = self._connector.fetch_audio(audio_url) if audio_url else None
982
+
983
+ placeholder = self._tracker.add("audio", audio, uuid)
984
+ self._add_placeholder("audio", placeholder)
985
+
986
+ def parse_input_audio(
987
+ self, input_audio: InputAudio | None, uuid: str | None = None
988
+ ) -> None:
989
+ if input_audio:
990
+ audio_data = input_audio.get("data", "")
991
+ audio_format = input_audio.get("format", "")
992
+ if audio_data:
993
+ audio_url = f"data:audio/{audio_format};base64,{audio_data}"
994
+ else:
995
+ # If a UUID is provided, audio data may be empty.
996
+ audio_url = None
997
+ else:
998
+ audio_url = None
999
+
1000
+ return self.parse_audio(audio_url, uuid)
1001
+
1002
+ def parse_video(self, video_url: str | None, uuid: str | None = None) -> None:
1003
+ video = self._connector.fetch_video(video_url=video_url) if video_url else None
1004
+
1005
+ placeholder = self._tracker.add("video", video, uuid)
1006
+ self._add_placeholder("video", placeholder)
1007
+
1008
+
1009
+ class AsyncMultiModalContentParser(BaseMultiModalContentParser):
1010
+ def __init__(self, tracker: AsyncMultiModalItemTracker) -> None:
1011
+ super().__init__()
1012
+
1013
+ self._tracker = tracker
1014
+ multimodal_config = self._tracker.model_config.multimodal_config
1015
+ media_io_kwargs = getattr(multimodal_config, "media_io_kwargs", None)
1016
+ self._connector: MediaConnector = MEDIA_CONNECTOR_REGISTRY.load(
1017
+ envs.VLLM_MEDIA_CONNECTOR,
1018
+ media_io_kwargs=media_io_kwargs,
1019
+ allowed_local_media_path=tracker.allowed_local_media_path,
1020
+ allowed_media_domains=tracker.allowed_media_domains,
1021
+ )
1022
+
1023
+ @property
1024
+ def model_config(self) -> ModelConfig:
1025
+ return self._tracker.model_config
1026
+
1027
+ def parse_image(self, image_url: str | None, uuid: str | None = None) -> None:
1028
+ image_coro = self._connector.fetch_image_async(image_url) if image_url else None
1029
+
1030
+ placeholder = self._tracker.add("image", image_coro, uuid)
1031
+ self._add_placeholder("image", placeholder)
1032
+
1033
+ def parse_image_embeds(
1034
+ self,
1035
+ image_embeds: str | dict[str, str] | None,
1036
+ uuid: str | None = None,
1037
+ ) -> None:
1038
+ mm_config = self.model_config.get_multimodal_config()
1039
+ if not mm_config.enable_mm_embeds:
1040
+ raise ValueError(
1041
+ "You must set `--enable-mm-embeds` to input `image_embeds`"
1042
+ )
1043
+
1044
+ future: asyncio.Future[str | dict[str, str] | None] = asyncio.Future()
1045
+
1046
+ if isinstance(image_embeds, dict):
1047
+ embeds = {
1048
+ k: self._connector.fetch_image_embedding(v)
1049
+ for k, v in image_embeds.items()
1050
+ }
1051
+ future.set_result(embeds)
1052
+
1053
+ if isinstance(image_embeds, str):
1054
+ embedding = self._connector.fetch_image_embedding(image_embeds)
1055
+ future.set_result(embedding)
1056
+
1057
+ if image_embeds is None:
1058
+ future.set_result(None)
1059
+
1060
+ placeholder = self._tracker.add("image_embeds", future, uuid)
1061
+ self._add_placeholder("image", placeholder)
1062
+
1063
+ def parse_audio_embeds(
1064
+ self,
1065
+ audio_embeds: str | dict[str, str] | None,
1066
+ uuid: str | None = None,
1067
+ ) -> None:
1068
+ mm_config = self.model_config.get_multimodal_config()
1069
+ if not mm_config.enable_mm_embeds:
1070
+ raise ValueError(
1071
+ "You must set `--enable-mm-embeds` to input `audio_embeds`"
1072
+ )
1073
+
1074
+ logger.info(
1075
+ "🎵 Parsing audio_embeds: type=%s, uuid=%s, is_dict=%s, "
1076
+ "is_str=%s, is_none=%s",
1077
+ type(audio_embeds).__name__,
1078
+ uuid,
1079
+ isinstance(audio_embeds, dict),
1080
+ isinstance(audio_embeds, str),
1081
+ audio_embeds is None,
1082
+ )
1083
+
1084
+ future: asyncio.Future[str | dict[str, str] | None] = asyncio.Future()
1085
+
1086
+ if isinstance(audio_embeds, dict):
1087
+ logger.info(
1088
+ "🎵 Processing dict audio_embeds with %d entries",
1089
+ len(audio_embeds),
1090
+ )
1091
+ embeds = {
1092
+ k: self._connector.fetch_audio_embedding(v)
1093
+ for k, v in audio_embeds.items()
1094
+ }
1095
+ future.set_result(embeds)
1096
+ logger.info(
1097
+ "🎵 Successfully loaded %d audio embeddings from dict",
1098
+ len(embeds),
1099
+ )
1100
+
1101
+ if isinstance(audio_embeds, str):
1102
+ base64_size = len(audio_embeds)
1103
+ logger.info(
1104
+ "🎵 Processing base64 audio_embeds: %d chars (%.2f KB)",
1105
+ base64_size,
1106
+ base64_size / 1024,
1107
+ )
1108
+ embedding = self._connector.fetch_audio_embedding(audio_embeds)
1109
+ future.set_result(embedding)
1110
+ logger.info(
1111
+ "🎵 Successfully loaded audio embedding tensor: shape=%s, dtype=%s",
1112
+ embedding.shape,
1113
+ embedding.dtype,
1114
+ )
1115
+
1116
+ if audio_embeds is None:
1117
+ logger.info("🎵 Audio embeds is None (UUID-only reference)")
1118
+ future.set_result(None)
1119
+
1120
+ placeholder = self._tracker.add("audio_embeds", future, uuid)
1121
+ self._add_placeholder("audio", placeholder)
1122
+ logger.info("🎵 Added audio_embeds placeholder with uuid=%s", uuid)
1123
+
1124
+ def parse_image_pil(
1125
+ self, image_pil: Image.Image | None, uuid: str | None = None
1126
+ ) -> None:
1127
+ future: asyncio.Future[Image.Image | None] = asyncio.Future()
1128
+ if image_pil:
1129
+ future.set_result(image_pil)
1130
+ else:
1131
+ future.set_result(None)
1132
+
1133
+ placeholder = self._tracker.add("image", future, uuid)
1134
+ self._add_placeholder("image", placeholder)
1135
+
1136
+ def parse_audio(self, audio_url: str | None, uuid: str | None = None) -> None:
1137
+ audio_coro = self._connector.fetch_audio_async(audio_url) if audio_url else None
1138
+
1139
+ placeholder = self._tracker.add("audio", audio_coro, uuid)
1140
+ self._add_placeholder("audio", placeholder)
1141
+
1142
+ def parse_input_audio(
1143
+ self, input_audio: InputAudio | None, uuid: str | None = None
1144
+ ) -> None:
1145
+ if input_audio:
1146
+ audio_data = input_audio.get("data", "")
1147
+ audio_format = input_audio.get("format", "")
1148
+ if audio_data:
1149
+ audio_url = f"data:audio/{audio_format};base64,{audio_data}"
1150
+ else:
1151
+ # If a UUID is provided, audio data may be empty.
1152
+ audio_url = None
1153
+ else:
1154
+ audio_url = None
1155
+
1156
+ return self.parse_audio(audio_url, uuid)
1157
+
1158
+ def parse_video(self, video_url: str | None, uuid: str | None = None) -> None:
1159
+ video = (
1160
+ self._connector.fetch_video_async(video_url=video_url)
1161
+ if video_url
1162
+ else None
1163
+ )
1164
+
1165
+ placeholder = self._tracker.add("video", video, uuid)
1166
+ self._add_placeholder("video", placeholder)
1167
+
1168
+
1169
+ def validate_chat_template(chat_template: Path | str | None):
1170
+ """Raises if the provided chat template appears invalid."""
1171
+ if chat_template is None:
1172
+ return
1173
+
1174
+ elif isinstance(chat_template, Path) and not chat_template.exists():
1175
+ raise FileNotFoundError("the supplied chat template path doesn't exist")
1176
+
1177
+ elif isinstance(chat_template, str):
1178
+ JINJA_CHARS = "{}\n"
1179
+ if (
1180
+ not any(c in chat_template for c in JINJA_CHARS)
1181
+ and not Path(chat_template).exists()
1182
+ ):
1183
+ # Try to find the template in the built-in templates directory
1184
+ from vllm.transformers_utils.chat_templates.registry import (
1185
+ CHAT_TEMPLATES_DIR,
1186
+ )
1187
+
1188
+ builtin_template_path = CHAT_TEMPLATES_DIR / chat_template
1189
+ if not builtin_template_path.exists():
1190
+ raise ValueError(
1191
+ f"The supplied chat template string ({chat_template}) "
1192
+ f"appears path-like, but doesn't exist! "
1193
+ f"Tried: {chat_template} and {builtin_template_path}"
1194
+ )
1195
+
1196
+ else:
1197
+ raise TypeError(f"{type(chat_template)} is not a valid chat template type")
1198
+
1199
+
1200
+ def _load_chat_template(
1201
+ chat_template: Path | str | None,
1202
+ *,
1203
+ is_literal: bool = False,
1204
+ ) -> str | None:
1205
+ if chat_template is None:
1206
+ return None
1207
+
1208
+ if is_literal:
1209
+ if isinstance(chat_template, Path):
1210
+ raise TypeError(
1211
+ "chat_template is expected to be read directly from its value"
1212
+ )
1213
+
1214
+ return chat_template
1215
+
1216
+ try:
1217
+ with open(chat_template) as f:
1218
+ return f.read()
1219
+ except OSError as e:
1220
+ if isinstance(chat_template, Path):
1221
+ raise
1222
+
1223
+ JINJA_CHARS = "{}\n"
1224
+ if not any(c in chat_template for c in JINJA_CHARS):
1225
+ # Try to load from the built-in templates directory
1226
+ from vllm.transformers_utils.chat_templates.registry import (
1227
+ CHAT_TEMPLATES_DIR,
1228
+ )
1229
+
1230
+ builtin_template_path = CHAT_TEMPLATES_DIR / chat_template
1231
+ try:
1232
+ with open(builtin_template_path) as f:
1233
+ return f.read()
1234
+ except OSError:
1235
+ msg = (
1236
+ f"The supplied chat template ({chat_template}) "
1237
+ f"looks like a file path, but it failed to be opened. "
1238
+ f"Tried: {chat_template} and {builtin_template_path}. "
1239
+ f"Reason: {e}"
1240
+ )
1241
+ raise ValueError(msg) from e
1242
+
1243
+ # If opening a file fails, set chat template to be args to
1244
+ # ensure we decode so our escape are interpreted correctly
1245
+ return _load_chat_template(chat_template, is_literal=True)
1246
+
1247
+
1248
+ _cached_load_chat_template = lru_cache(_load_chat_template)
1249
+
1250
+
1251
+ def load_chat_template(
1252
+ chat_template: Path | str | None,
1253
+ *,
1254
+ is_literal: bool = False,
1255
+ ) -> str | None:
1256
+ return _cached_load_chat_template(chat_template, is_literal=is_literal)
1257
+
1258
+
1259
+ def _get_interleaved_text_prompt(
1260
+ placeholder_storage: dict[str, list], texts: list[str]
1261
+ ) -> str:
1262
+ for idx, elem in enumerate(texts):
1263
+ if elem in placeholder_storage:
1264
+ texts[idx] = placeholder_storage[elem].pop(0)
1265
+
1266
+ return "\n".join(texts)
1267
+
1268
+
1269
+ # TODO: Let user specify how to insert multimodal tokens into prompt
1270
+ # (similar to chat template)
1271
+ def _get_full_multimodal_text_prompt(
1272
+ placeholder_storage: dict[str, list],
1273
+ texts: list[str],
1274
+ interleave_strings: bool,
1275
+ ) -> str:
1276
+ """Combine multimodal prompts for a multimodal language model."""
1277
+
1278
+ # flatten storage to make it looks like
1279
+ # {
1280
+ # "<|image|>": 2,
1281
+ # "<|audio|>": 1
1282
+ # }
1283
+ placeholder_counts = Counter(
1284
+ [v for elem in placeholder_storage.values() for v in elem]
1285
+ )
1286
+
1287
+ if interleave_strings:
1288
+ text_prompt = _get_interleaved_text_prompt(placeholder_storage, texts)
1289
+ else:
1290
+ text_prompt = "\n".join(texts)
1291
+
1292
+ # Pass interleaved text further in case the user used image placeholders
1293
+ # himself, but forgot to disable the 'interleave_strings' flag
1294
+
1295
+ # Look through the text prompt to check for missing placeholders
1296
+ missing_placeholders: list[str] = []
1297
+ for placeholder in placeholder_counts:
1298
+ # For any existing placeholder in the text prompt, we leave it as is
1299
+ placeholder_counts[placeholder] -= text_prompt.count(placeholder)
1300
+
1301
+ if placeholder_counts[placeholder] < 0:
1302
+ logger.error(
1303
+ "Placeholder count is negative! "
1304
+ "Ensure that the 'interleave_strings' flag is disabled "
1305
+ "(current value: %s) "
1306
+ "when manually placing image placeholders.",
1307
+ interleave_strings,
1308
+ )
1309
+ logger.debug("Input prompt: %s", text_prompt)
1310
+ raise ValueError(
1311
+ f"Found more '{placeholder}' placeholders in input prompt than "
1312
+ "actual multimodal data items."
1313
+ )
1314
+
1315
+ missing_placeholders.extend([placeholder] * placeholder_counts[placeholder])
1316
+
1317
+ # NOTE: Default behaviour: we always add missing placeholders
1318
+ # at the front of the prompt, if interleave_strings=False
1319
+ return "\n".join(missing_placeholders + [text_prompt])
1320
+
1321
+
1322
+ # No need to validate using Pydantic again
1323
+ _TextParser = partial(cast, ChatCompletionContentPartTextParam)
1324
+ _ImageEmbedsParser = partial(cast, ChatCompletionContentPartImageEmbedsParam)
1325
+ _AudioEmbedsParser = partial(cast, ChatCompletionContentPartAudioEmbedsParam)
1326
+ _InputAudioParser = partial(cast, ChatCompletionContentPartInputAudioParam)
1327
+ _RefusalParser = partial(cast, ChatCompletionContentPartRefusalParam)
1328
+ _PILImageParser = partial(cast, CustomChatCompletionContentPILImageParam)
1329
+ _ThinkParser = partial(cast, CustomThinkCompletionContentParam)
1330
+ # Need to validate url objects
1331
+ _ImageParser = TypeAdapter(ChatCompletionContentPartImageParam).validate_python
1332
+ _AudioParser = TypeAdapter(ChatCompletionContentPartAudioParam).validate_python
1333
+ _VideoParser = TypeAdapter(ChatCompletionContentPartVideoParam).validate_python
1334
+
1335
+ _ResponsesInputImageParser = TypeAdapter(ResponseInputImageParam).validate_python
1336
+ _ContentPart: TypeAlias = str | dict[str, str] | InputAudio | PILImage
1337
+
1338
+ # Define a mapping from part types to their corresponding parsing functions.
1339
+ MM_PARSER_MAP: dict[
1340
+ str,
1341
+ Callable[[ChatCompletionContentPartParam], _ContentPart],
1342
+ ] = {
1343
+ "text": lambda part: _TextParser(part).get("text", None),
1344
+ "thinking": lambda part: _ThinkParser(part).get("thinking", None),
1345
+ "input_text": lambda part: _TextParser(part).get("text", None),
1346
+ "output_text": lambda part: _TextParser(part).get("text", None),
1347
+ "input_image": lambda part: _ResponsesInputImageParser(part).get("image_url", None),
1348
+ "image_url": lambda part: _ImageParser(part).get("image_url", {}).get("url", None),
1349
+ "image_embeds": lambda part: _ImageEmbedsParser(part).get("image_embeds", None),
1350
+ "audio_embeds": lambda part: _AudioEmbedsParser(part).get("audio_embeds", None),
1351
+ "image_pil": lambda part: _PILImageParser(part).get("image_pil", None),
1352
+ "audio_url": lambda part: _AudioParser(part).get("audio_url", {}).get("url", None),
1353
+ "input_audio": lambda part: _InputAudioParser(part).get("input_audio", None),
1354
+ "refusal": lambda part: _RefusalParser(part).get("refusal", None),
1355
+ "video_url": lambda part: _VideoParser(part).get("video_url", {}).get("url", None),
1356
+ }
1357
+
1358
+
1359
+ def _parse_chat_message_content_mm_part(
1360
+ part: ChatCompletionContentPartParam,
1361
+ ) -> tuple[str, _ContentPart]:
1362
+ """
1363
+ Parses a given multi-modal content part based on its type.
1364
+
1365
+ Args:
1366
+ part: A dict containing the content part, with a potential 'type' field.
1367
+
1368
+ Returns:
1369
+ A tuple (part_type, content) where:
1370
+ - part_type: Type of the part (e.g., 'text', 'image_url').
1371
+ - content: Parsed content (e.g., text, image URL).
1372
+
1373
+ Raises:
1374
+ ValueError: If the 'type' field is missing and no direct URL is found.
1375
+ """
1376
+ assert isinstance(
1377
+ part, dict
1378
+ ) # This is needed to avoid mypy errors: part.get() from str
1379
+ part_type = part.get("type", None)
1380
+ uuid = part.get("uuid", None)
1381
+
1382
+ if isinstance(part_type, str) and part_type in MM_PARSER_MAP and uuid is None: # noqa: E501
1383
+ content = MM_PARSER_MAP[part_type](part)
1384
+
1385
+ # Special case for 'image_url.detail'
1386
+ # We only support 'auto', which is the default
1387
+ if part_type == "image_url" and part.get("detail", "auto") != "auto":
1388
+ logger.warning(
1389
+ "'image_url.detail' is currently not supported and will be ignored."
1390
+ )
1391
+
1392
+ return part_type, content
1393
+
1394
+ # Handle missing 'type' but provided direct URL fields.
1395
+ # 'type' is required field by pydantic
1396
+ if part_type is None or uuid is not None:
1397
+ if "image_url" in part:
1398
+ image_params = cast(CustomChatCompletionContentSimpleImageParam, part)
1399
+ image_url = image_params.get("image_url", None)
1400
+ if isinstance(image_url, dict):
1401
+ # Can potentially happen if user provides a uuid
1402
+ # with url as a dict of {"url": url}
1403
+ image_url = image_url.get("url", None)
1404
+ return "image_url", image_url
1405
+ if "image_pil" in part:
1406
+ # "image_pil" could be None if UUID is provided.
1407
+ image_params = cast( # type: ignore
1408
+ CustomChatCompletionContentPILImageParam, part
1409
+ )
1410
+ image_pil = image_params.get("image_pil", None)
1411
+ return "image_pil", image_pil
1412
+ if "image_embeds" in part:
1413
+ # "image_embeds" could be None if UUID is provided.
1414
+ image_params = cast( # type: ignore
1415
+ ChatCompletionContentPartImageEmbedsParam, part
1416
+ )
1417
+ image_embeds = image_params.get("image_embeds", None)
1418
+ return "image_embeds", image_embeds
1419
+ if "audio_embeds" in part:
1420
+ # "audio_embeds" could be None if UUID is provided.
1421
+ audio_params = cast( # type: ignore[assignment]
1422
+ ChatCompletionContentPartAudioEmbedsParam, part
1423
+ )
1424
+ audio_embeds = audio_params.get("audio_embeds", None)
1425
+ return "audio_embeds", audio_embeds
1426
+ if "audio_url" in part:
1427
+ audio_params = cast( # type: ignore[assignment]
1428
+ CustomChatCompletionContentSimpleAudioParam, part
1429
+ )
1430
+ audio_url = audio_params.get("audio_url", None)
1431
+ if isinstance(audio_url, dict):
1432
+ # Can potentially happen if user provides a uuid
1433
+ # with url as a dict of {"url": url}
1434
+ audio_url = audio_url.get("url", None)
1435
+ return "audio_url", audio_url
1436
+ if part.get("input_audio") is not None:
1437
+ input_audio_params = cast(dict[str, str], part)
1438
+ return "input_audio", input_audio_params
1439
+ if "video_url" in part:
1440
+ video_params = cast(CustomChatCompletionContentSimpleVideoParam, part)
1441
+ video_url = video_params.get("video_url", None)
1442
+ if isinstance(video_url, dict):
1443
+ # Can potentially happen if user provides a uuid
1444
+ # with url as a dict of {"url": url}
1445
+ video_url = video_url.get("url", None)
1446
+ return "video_url", video_url
1447
+ # Raise an error if no 'type' or direct URL is found.
1448
+ raise ValueError("Missing 'type' field in multimodal part.")
1449
+
1450
+ if not isinstance(part_type, str):
1451
+ raise ValueError("Invalid 'type' field in multimodal part.")
1452
+ return part_type, "unknown part_type content"
1453
+
1454
+
1455
+ PART_TYPES_TO_SKIP_NONE_CONTENT = (
1456
+ "text",
1457
+ "refusal",
1458
+ )
1459
+
1460
+
1461
+ def _parse_chat_message_content_parts(
1462
+ role: str,
1463
+ parts: Iterable[ChatCompletionContentPartParam],
1464
+ mm_tracker: BaseMultiModalItemTracker,
1465
+ *,
1466
+ wrap_dicts: bool,
1467
+ interleave_strings: bool,
1468
+ ) -> list[ConversationMessage]:
1469
+ content = list[_ContentPart]()
1470
+
1471
+ mm_parser = mm_tracker.create_parser()
1472
+
1473
+ for part in parts:
1474
+ parse_res = _parse_chat_message_content_part(
1475
+ part,
1476
+ mm_parser,
1477
+ wrap_dicts=wrap_dicts,
1478
+ interleave_strings=interleave_strings,
1479
+ )
1480
+ if parse_res:
1481
+ content.append(parse_res)
1482
+
1483
+ if wrap_dicts:
1484
+ # Parsing wraps images and texts as interleaved dictionaries
1485
+ return [ConversationMessage(role=role, content=content)] # type: ignore
1486
+ texts = cast(list[str], content)
1487
+ mm_placeholder_storage = mm_parser.mm_placeholder_storage()
1488
+ if mm_placeholder_storage:
1489
+ text_prompt = _get_full_multimodal_text_prompt(
1490
+ mm_placeholder_storage, texts, interleave_strings
1491
+ )
1492
+ else:
1493
+ text_prompt = "\n".join(texts)
1494
+
1495
+ return [ConversationMessage(role=role, content=text_prompt)]
1496
+
1497
+
1498
+ def _parse_chat_message_content_part(
1499
+ part: ChatCompletionContentPartParam,
1500
+ mm_parser: BaseMultiModalContentParser,
1501
+ *,
1502
+ wrap_dicts: bool,
1503
+ interleave_strings: bool,
1504
+ ) -> _ContentPart | None:
1505
+ """Parses a single part of a conversation. If wrap_dicts is True,
1506
+ structured dictionary pieces for texts and images will be
1507
+ wrapped in dictionaries, i.e., {"type": "text", "text", ...} and
1508
+ {"type": "image"}, respectively. Otherwise multimodal data will be
1509
+ handled by mm_parser, and texts will be returned as strings to be joined
1510
+ with multimodal placeholders.
1511
+ """
1512
+ if isinstance(part, str): # Handle plain text parts
1513
+ return part
1514
+ # Handle structured dictionary parts
1515
+ part_type, content = _parse_chat_message_content_mm_part(part)
1516
+ # if part_type is text/refusal/image_url/audio_url/video_url/input_audio but
1517
+ # content is None, log a warning and skip
1518
+ if part_type in PART_TYPES_TO_SKIP_NONE_CONTENT and content is None:
1519
+ logger.warning(
1520
+ "Skipping multimodal part '%s' (type: '%s') "
1521
+ "with empty / unparsable content.",
1522
+ part,
1523
+ part_type,
1524
+ )
1525
+ return None
1526
+
1527
+ if part_type in ("text", "input_text", "output_text", "refusal", "thinking"):
1528
+ str_content = cast(str, content)
1529
+ if wrap_dicts:
1530
+ return {"type": "text", "text": str_content}
1531
+ else:
1532
+ return str_content
1533
+
1534
+ # For media items, if a user has provided one, use it. Otherwise, insert
1535
+ # a placeholder empty uuid.
1536
+ uuid = part.get("uuid", None)
1537
+ if uuid is not None:
1538
+ uuid = str(uuid)
1539
+
1540
+ modality = None
1541
+ if part_type == "image_pil":
1542
+ image_content = cast(Image.Image, content) if content is not None else None
1543
+ mm_parser.parse_image_pil(image_content, uuid)
1544
+ modality = "image"
1545
+ elif part_type in ("image_url", "input_image"):
1546
+ str_content = cast(str, content)
1547
+ mm_parser.parse_image(str_content, uuid)
1548
+ modality = "image"
1549
+ elif part_type == "image_embeds":
1550
+ content = cast(str | dict[str, str], content) if content is not None else None
1551
+ mm_parser.parse_image_embeds(content, uuid)
1552
+ modality = "image"
1553
+ elif part_type == "audio_embeds":
1554
+ content = cast(str | dict[str, str], content) if content is not None else None
1555
+ mm_parser.parse_audio_embeds(content, uuid)
1556
+ modality = "audio"
1557
+ elif part_type == "audio_url":
1558
+ str_content = cast(str, content)
1559
+ mm_parser.parse_audio(str_content, uuid)
1560
+ modality = "audio"
1561
+ elif part_type == "input_audio":
1562
+ dict_content = cast(InputAudio, content)
1563
+ mm_parser.parse_input_audio(dict_content, uuid)
1564
+ modality = "audio"
1565
+ elif part_type == "video_url":
1566
+ str_content = cast(str, content)
1567
+ mm_parser.parse_video(str_content, uuid)
1568
+ modality = "video"
1569
+ else:
1570
+ raise NotImplementedError(f"Unknown part type: {part_type}")
1571
+
1572
+ return (
1573
+ {"type": modality}
1574
+ if wrap_dicts
1575
+ else (MODALITY_PLACEHOLDERS_MAP[modality] if interleave_strings else None)
1576
+ )
1577
+
1578
+
1579
+ # No need to validate using Pydantic again
1580
+ _AssistantParser = partial(cast, ChatCompletionAssistantMessageParam)
1581
+ _ToolParser = partial(cast, ChatCompletionToolMessageParam)
1582
+
1583
+
1584
+ def _parse_chat_message_content(
1585
+ message: ChatCompletionMessageParam,
1586
+ mm_tracker: BaseMultiModalItemTracker,
1587
+ content_format: _ChatTemplateContentFormat,
1588
+ interleave_strings: bool,
1589
+ ) -> list[ConversationMessage]:
1590
+ role = message["role"]
1591
+ content = message.get("content")
1592
+ reasoning = message.get("reasoning") or message.get("reasoning_content")
1593
+
1594
+ if content is None:
1595
+ content = []
1596
+ elif isinstance(content, str):
1597
+ content = [ChatCompletionContentPartTextParam(type="text", text=content)]
1598
+ result = _parse_chat_message_content_parts(
1599
+ role,
1600
+ content, # type: ignore
1601
+ mm_tracker,
1602
+ wrap_dicts=(content_format == "openai"),
1603
+ interleave_strings=interleave_strings,
1604
+ )
1605
+
1606
+ for result_msg in result:
1607
+ if role == "assistant":
1608
+ parsed_msg = _AssistantParser(message)
1609
+
1610
+ # The 'tool_calls' is not None check ensures compatibility.
1611
+ # It's needed only if downstream code doesn't strictly
1612
+ # follow the OpenAI spec.
1613
+ if "tool_calls" in parsed_msg and parsed_msg["tool_calls"] is not None:
1614
+ result_msg["tool_calls"] = list(parsed_msg["tool_calls"])
1615
+ # Include reasoning if present for interleaved thinking.
1616
+ if reasoning is not None:
1617
+ result_msg["reasoning"] = cast(str, reasoning)
1618
+ result_msg["reasoning_content"] = cast(
1619
+ str, reasoning
1620
+ ) # keep compatibility
1621
+ elif role == "tool":
1622
+ parsed_msg = _ToolParser(message)
1623
+ if "tool_call_id" in parsed_msg:
1624
+ result_msg["tool_call_id"] = parsed_msg["tool_call_id"]
1625
+
1626
+ if "name" in message and isinstance(message["name"], str):
1627
+ result_msg["name"] = message["name"]
1628
+
1629
+ if role == "developer":
1630
+ result_msg["tools"] = message.get("tools", None)
1631
+ return result
1632
+
1633
+
1634
+ def _postprocess_messages(messages: list[ConversationMessage]) -> None:
1635
+ # per the Transformers docs & maintainers, tool call arguments in
1636
+ # assistant-role messages with tool_calls need to be dicts not JSON str -
1637
+ # this is how tool-use chat templates will expect them moving forwards
1638
+ # so, for messages that have tool_calls, parse the string (which we get
1639
+ # from openAI format) to dict
1640
+ for message in messages:
1641
+ if message["role"] == "assistant" and "tool_calls" in message:
1642
+ tool_calls = message.get("tool_calls")
1643
+ if not isinstance(tool_calls, list):
1644
+ continue
1645
+
1646
+ if len(tool_calls) == 0:
1647
+ # Drop empty tool_calls to keep templates on the normal assistant path.
1648
+ message.pop("tool_calls", None)
1649
+ continue
1650
+
1651
+ for item in tool_calls:
1652
+ # if arguments is None or empty string, set to {}
1653
+ if content := item["function"].get("arguments"):
1654
+ if not isinstance(content, (dict, list)):
1655
+ item["function"]["arguments"] = json.loads(content)
1656
+ else:
1657
+ item["function"]["arguments"] = {}
1658
+
1659
+
1660
+ def parse_chat_messages(
1661
+ messages: list[ChatCompletionMessageParam],
1662
+ model_config: ModelConfig,
1663
+ content_format: _ChatTemplateContentFormat,
1664
+ ) -> tuple[
1665
+ list[ConversationMessage],
1666
+ MultiModalDataDict | None,
1667
+ MultiModalUUIDDict | None,
1668
+ ]:
1669
+ conversation: list[ConversationMessage] = []
1670
+ mm_tracker = MultiModalItemTracker(model_config)
1671
+
1672
+ for msg in messages:
1673
+ sub_messages = _parse_chat_message_content(
1674
+ msg,
1675
+ mm_tracker,
1676
+ content_format,
1677
+ interleave_strings=(
1678
+ content_format == "string"
1679
+ and model_config.multimodal_config is not None
1680
+ and model_config.multimodal_config.interleave_mm_strings
1681
+ ),
1682
+ )
1683
+
1684
+ conversation.extend(sub_messages)
1685
+
1686
+ _postprocess_messages(conversation)
1687
+
1688
+ return conversation, mm_tracker.all_mm_data(), mm_tracker.all_mm_uuids()
1689
+
1690
+
1691
+ def parse_chat_messages_futures(
1692
+ messages: list[ChatCompletionMessageParam],
1693
+ model_config: ModelConfig,
1694
+ content_format: _ChatTemplateContentFormat,
1695
+ ) -> tuple[
1696
+ list[ConversationMessage],
1697
+ Awaitable[MultiModalDataDict | None],
1698
+ MultiModalUUIDDict | None,
1699
+ ]:
1700
+ conversation: list[ConversationMessage] = []
1701
+ mm_tracker = AsyncMultiModalItemTracker(model_config)
1702
+
1703
+ for msg in messages:
1704
+ sub_messages = _parse_chat_message_content(
1705
+ msg,
1706
+ mm_tracker,
1707
+ content_format,
1708
+ interleave_strings=(
1709
+ content_format == "string"
1710
+ and model_config.multimodal_config is not None
1711
+ and model_config.multimodal_config.interleave_mm_strings
1712
+ ),
1713
+ )
1714
+
1715
+ conversation.extend(sub_messages)
1716
+
1717
+ _postprocess_messages(conversation)
1718
+
1719
+ return conversation, mm_tracker.all_mm_data(), mm_tracker.all_mm_uuids()
1720
+
1721
+
1722
+ # adapted from https://github.com/huggingface/transformers/blob/v4.56.2/src/transformers/utils/chat_template_utils.py#L398-L412
1723
+ # only preserve the parse function used to resolve chat template kwargs
1724
+ class AssistantTracker(jinja2.ext.Extension):
1725
+ tags = {"generation"}
1726
+
1727
+ def parse(self, parser: jinja2.parser.Parser) -> jinja2.nodes.CallBlock:
1728
+ lineno = next(parser.stream).lineno
1729
+ body = parser.parse_statements(["name:endgeneration"], drop_needle=True)
1730
+ call = self.call_method("_generation_support")
1731
+ call_block = jinja2.nodes.CallBlock(call, [], [], body)
1732
+ return call_block.set_lineno(lineno)
1733
+
1734
+
1735
+ def _resolve_chat_template_kwargs(
1736
+ chat_template: str,
1737
+ ):
1738
+ env = jinja2.sandbox.ImmutableSandboxedEnvironment(
1739
+ trim_blocks=True,
1740
+ lstrip_blocks=True,
1741
+ extensions=[AssistantTracker, jinja2.ext.loopcontrols],
1742
+ )
1743
+ parsed_content = env.parse(chat_template)
1744
+ template_vars = jinja2.meta.find_undeclared_variables(parsed_content)
1745
+ return template_vars
1746
+
1747
+
1748
+ _cached_resolve_chat_template_kwargs = lru_cache(_resolve_chat_template_kwargs)
1749
+
1750
+
1751
+ @lru_cache
1752
+ def _get_hf_base_chat_template_params() -> frozenset[str]:
1753
+ # Get standard parameters from HuggingFace's base tokenizer class.
1754
+ # This dynamically extracts parameters from PreTrainedTokenizer's
1755
+ # apply_chat_template method, ensuring compatibility with tokenizers
1756
+ # that use **kwargs to receive standard parameters.
1757
+
1758
+ # Read signature from HF's base class - the single source of truth
1759
+ base_sig = inspect.signature(PreTrainedTokenizer.apply_chat_template)
1760
+ # Exclude VAR_KEYWORD (**kwargs) and VAR_POSITIONAL (*args) placeholders
1761
+ return frozenset(
1762
+ p.name
1763
+ for p in base_sig.parameters.values()
1764
+ if p.kind
1765
+ not in (inspect.Parameter.VAR_KEYWORD, inspect.Parameter.VAR_POSITIONAL)
1766
+ )
1767
+
1768
+
1769
+ def resolve_chat_template_kwargs(
1770
+ tokenizer: PreTrainedTokenizer | PreTrainedTokenizerFast,
1771
+ chat_template: str,
1772
+ chat_template_kwargs: dict[str, Any],
1773
+ raise_on_unexpected: bool = True,
1774
+ ) -> dict[str, Any]:
1775
+ # We exclude chat_template from kwargs here, because
1776
+ # chat template has been already resolved at this stage
1777
+ unexpected_vars = {"chat_template", "tokenize"}
1778
+ if raise_on_unexpected and (
1779
+ unexpected_in_kwargs := unexpected_vars & chat_template_kwargs.keys()
1780
+ ):
1781
+ raise ValueError(
1782
+ "Found unexpected chat template kwargs from request: "
1783
+ f"{unexpected_in_kwargs}"
1784
+ )
1785
+
1786
+ fn_kw = {
1787
+ k
1788
+ for k in chat_template_kwargs
1789
+ if supports_kw(tokenizer.apply_chat_template, k, allow_var_kwargs=False)
1790
+ }
1791
+ template_vars = _cached_resolve_chat_template_kwargs(chat_template)
1792
+
1793
+ # Allow standard HF parameters even if tokenizer uses **kwargs to receive them
1794
+ hf_base_params = _get_hf_base_chat_template_params()
1795
+
1796
+ accept_vars = (fn_kw | template_vars | hf_base_params) - unexpected_vars
1797
+ return {k: v for k, v in chat_template_kwargs.items() if k in accept_vars}
1798
+
1799
+
1800
+ def apply_hf_chat_template(
1801
+ tokenizer: PreTrainedTokenizer | PreTrainedTokenizerFast,
1802
+ conversation: list[ConversationMessage],
1803
+ chat_template: str | None,
1804
+ tools: list[dict[str, Any]] | None,
1805
+ *,
1806
+ model_config: ModelConfig,
1807
+ **kwargs: Any,
1808
+ ) -> str:
1809
+ hf_chat_template = resolve_hf_chat_template(
1810
+ tokenizer,
1811
+ chat_template=chat_template,
1812
+ tools=tools,
1813
+ model_config=model_config,
1814
+ )
1815
+
1816
+ if hf_chat_template is None:
1817
+ raise ValueError(
1818
+ "As of transformers v4.44, default chat template is no longer "
1819
+ "allowed, so you must provide a chat template if the tokenizer "
1820
+ "does not define one."
1821
+ )
1822
+
1823
+ resolved_kwargs = resolve_chat_template_kwargs(
1824
+ tokenizer=tokenizer,
1825
+ chat_template=hf_chat_template,
1826
+ chat_template_kwargs=kwargs,
1827
+ )
1828
+
1829
+ try:
1830
+ return tokenizer.apply_chat_template(
1831
+ conversation=conversation, # type: ignore[arg-type]
1832
+ tools=tools, # type: ignore[arg-type]
1833
+ chat_template=hf_chat_template,
1834
+ tokenize=False,
1835
+ **resolved_kwargs,
1836
+ )
1837
+
1838
+ # External library exceptions can sometimes occur despite the framework's
1839
+ # internal exception management capabilities.
1840
+ except Exception as e:
1841
+ # Log and report any library-related exceptions for further
1842
+ # investigation.
1843
+ logger.exception(
1844
+ "An error occurred in `transformers` while applying chat template"
1845
+ )
1846
+ raise ValueError(str(e)) from e
1847
+
1848
+
1849
+ def apply_mistral_chat_template(
1850
+ tokenizer: "MistralTokenizer",
1851
+ messages: list[ChatCompletionMessageParam],
1852
+ chat_template: str | None,
1853
+ tools: list[dict[str, Any]] | None,
1854
+ **kwargs: Any,
1855
+ ) -> list[int]:
1856
+ from mistral_common.exceptions import MistralCommonException
1857
+
1858
+ # The return value of resolve_mistral_chat_template is always None,
1859
+ # and we won't use it.
1860
+ resolve_mistral_chat_template(
1861
+ chat_template=chat_template,
1862
+ **kwargs,
1863
+ )
1864
+
1865
+ try:
1866
+ return tokenizer.apply_chat_template(
1867
+ messages=messages,
1868
+ tools=tools,
1869
+ **kwargs,
1870
+ )
1871
+ # mistral-common uses assert statements to stop processing of input
1872
+ # if input does not comply with the expected format.
1873
+ # We convert those assertion errors to ValueErrors so they can be
1874
+ # properly caught in the preprocessing_input step
1875
+ except (AssertionError, MistralCommonException) as e:
1876
+ raise ValueError(str(e)) from e
1877
+
1878
+ # External library exceptions can sometimes occur despite the framework's
1879
+ # internal exception management capabilities.
1880
+ except Exception as e:
1881
+ # Log and report any library-related exceptions for further
1882
+ # investigation.
1883
+ logger.exception(
1884
+ "An error occurred in `mistral_common` while applying chat template"
1885
+ )
1886
+ raise ValueError(str(e)) from e
1887
+
1888
+
1889
+ def get_history_tool_calls_cnt(conversation: list[ConversationMessage]):
1890
+ idx = 0
1891
+ for msg in conversation:
1892
+ if msg["role"] == "assistant":
1893
+ tool_calls = msg.get("tool_calls")
1894
+ idx += len(list(tool_calls)) if tool_calls is not None else 0 # noqa
1895
+ return idx
1896
+
1897
+
1898
+ def make_tool_call_id(id_type: str = "random", func_name=None, idx=None):
1899
+ if id_type == "kimi_k2":
1900
+ return f"functions.{func_name}:{idx}"
1901
+ else:
1902
+ # by default return random
1903
+ return f"chatcmpl-tool-{random_uuid()}"