PyPI - sglang - Versions diffs - 0.4.8__py3-none-any.whl → 0.4.9__py3-none-any.whl - Mend

sglang 0.4.8py3-none-any.whl → 0.4.9py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (150) hide show

sglang/bench_one_batch_server.py +17 -2
sglang/bench_serving.py +168 -22
sglang/srt/configs/internvl.py +4 -2
sglang/srt/configs/janus_pro.py +1 -1
sglang/srt/configs/model_config.py +49 -0
sglang/srt/configs/update_config.py +119 -0
sglang/srt/conversation.py +35 -0
sglang/srt/custom_op.py +7 -1
sglang/srt/disaggregation/base/conn.py +2 -0
sglang/srt/disaggregation/decode.py +22 -6
sglang/srt/disaggregation/mooncake/conn.py +289 -48
sglang/srt/disaggregation/mooncake/transfer_engine.py +31 -1
sglang/srt/disaggregation/nixl/conn.py +100 -52
sglang/srt/disaggregation/prefill.py +5 -4
sglang/srt/disaggregation/utils.py +13 -12
sglang/srt/distributed/parallel_state.py +44 -17
sglang/srt/entrypoints/EngineBase.py +8 -0
sglang/srt/entrypoints/engine.py +45 -9
sglang/srt/entrypoints/http_server.py +111 -24
sglang/srt/entrypoints/openai/protocol.py +51 -6
sglang/srt/entrypoints/openai/serving_chat.py +52 -76
sglang/srt/entrypoints/openai/serving_completions.py +1 -0
sglang/srt/entrypoints/openai/serving_embedding.py +1 -0
sglang/srt/eplb/__init__.py +0 -0
sglang/srt/{managers → eplb}/eplb_algorithms/__init__.py +1 -1
sglang/srt/{managers → eplb}/eplb_manager.py +2 -4
sglang/srt/{eplb_simulator → eplb/eplb_simulator}/reader.py +1 -1
sglang/srt/{managers → eplb}/expert_distribution.py +18 -1
sglang/srt/{managers → eplb}/expert_location.py +1 -1
sglang/srt/{managers → eplb}/expert_location_dispatch.py +1 -1
sglang/srt/{model_executor → eplb}/expert_location_updater.py +17 -1
sglang/srt/hf_transformers_utils.py +2 -1
sglang/srt/layers/activation.py +7 -0
sglang/srt/layers/amx_utils.py +86 -0
sglang/srt/layers/attention/ascend_backend.py +219 -0
sglang/srt/layers/attention/flashattention_backend.py +56 -23
sglang/srt/layers/attention/tbo_backend.py +37 -9
sglang/srt/layers/communicator.py +18 -2
sglang/srt/layers/dp_attention.py +9 -3
sglang/srt/layers/elementwise.py +76 -12
sglang/srt/layers/flashinfer_comm_fusion.py +202 -0
sglang/srt/layers/layernorm.py +41 -0
sglang/srt/layers/linear.py +99 -12
sglang/srt/layers/logits_processor.py +15 -6
sglang/srt/layers/moe/ep_moe/kernels.py +23 -8
sglang/srt/layers/moe/ep_moe/layer.py +115 -25
sglang/srt/layers/moe/ep_moe/token_dispatcher.py +42 -19
sglang/srt/layers/moe/fused_moe_native.py +7 -0
sglang/srt/layers/moe/fused_moe_triton/fused_moe.py +8 -4
sglang/srt/layers/moe/fused_moe_triton/layer.py +129 -10
sglang/srt/layers/moe/router.py +60 -22
sglang/srt/layers/moe/topk.py +36 -28
sglang/srt/layers/parameter.py +67 -7
sglang/srt/layers/quantization/compressed_tensors/schemes/compressed_tensors_w8a16_fp8.py +1 -1
sglang/srt/layers/quantization/fp8.py +44 -0
sglang/srt/layers/quantization/fp8_kernel.py +1 -1
sglang/srt/layers/quantization/fp8_utils.py +6 -6
sglang/srt/layers/quantization/gptq.py +5 -1
sglang/srt/layers/quantization/moe_wna16.py +1 -1
sglang/srt/layers/quantization/quant_utils.py +166 -0
sglang/srt/layers/quantization/w8a8_int8.py +52 -1
sglang/srt/layers/rotary_embedding.py +105 -13
sglang/srt/layers/vocab_parallel_embedding.py +19 -2
sglang/srt/lora/lora.py +4 -5
sglang/srt/lora/lora_manager.py +73 -20
sglang/srt/managers/configure_logging.py +1 -1
sglang/srt/managers/io_struct.py +60 -15
sglang/srt/managers/mm_utils.py +73 -59
sglang/srt/managers/multimodal_processor.py +2 -6
sglang/srt/managers/multimodal_processors/qwen_audio.py +94 -0
sglang/srt/managers/schedule_batch.py +80 -79
sglang/srt/managers/scheduler.py +153 -63
sglang/srt/managers/scheduler_output_processor_mixin.py +8 -2
sglang/srt/managers/session_controller.py +12 -3
sglang/srt/managers/tokenizer_manager.py +314 -103
sglang/srt/managers/tp_worker.py +13 -1
sglang/srt/managers/tp_worker_overlap_thread.py +8 -0
sglang/srt/mem_cache/allocator.py +290 -0
sglang/srt/mem_cache/chunk_cache.py +34 -2
sglang/srt/mem_cache/memory_pool.py +289 -3
sglang/srt/mem_cache/multimodal_cache.py +3 -0
sglang/srt/model_executor/cuda_graph_runner.py +3 -2
sglang/srt/model_executor/forward_batch_info.py +17 -4
sglang/srt/model_executor/model_runner.py +302 -58
sglang/srt/model_loader/loader.py +86 -10
sglang/srt/model_loader/weight_utils.py +160 -3
sglang/srt/models/deepseek_nextn.py +5 -4
sglang/srt/models/deepseek_v2.py +305 -26
sglang/srt/models/deepseek_vl2.py +3 -5
sglang/srt/models/gemma3_causal.py +1 -2
sglang/srt/models/gemma3n_audio.py +949 -0
sglang/srt/models/gemma3n_causal.py +1010 -0
sglang/srt/models/gemma3n_mm.py +495 -0
sglang/srt/models/hunyuan.py +771 -0
sglang/srt/models/kimi_vl.py +1 -2
sglang/srt/models/llama.py +10 -4
sglang/srt/models/llama4.py +32 -45
sglang/srt/models/llama_eagle3.py +61 -11
sglang/srt/models/llava.py +5 -5
sglang/srt/models/minicpmo.py +2 -2
sglang/srt/models/mistral.py +1 -1
sglang/srt/models/mllama4.py +43 -11
sglang/srt/models/phi4mm.py +1 -3
sglang/srt/models/pixtral.py +3 -7
sglang/srt/models/qwen2.py +31 -3
sglang/srt/models/qwen2_5_vl.py +1 -3
sglang/srt/models/qwen2_audio.py +200 -0
sglang/srt/models/qwen2_moe.py +32 -6
sglang/srt/models/qwen2_vl.py +1 -4
sglang/srt/models/qwen3.py +94 -25
sglang/srt/models/qwen3_moe.py +68 -21
sglang/srt/models/vila.py +3 -8
sglang/srt/{managers/multimodal_processors → multimodal/processors}/base_processor.py +150 -133
sglang/srt/{managers/multimodal_processors → multimodal/processors}/clip.py +2 -13
sglang/srt/{managers/multimodal_processors → multimodal/processors}/deepseek_vl_v2.py +4 -11
sglang/srt/{managers/multimodal_processors → multimodal/processors}/gemma3.py +3 -10
sglang/srt/multimodal/processors/gemma3n.py +82 -0
sglang/srt/{managers/multimodal_processors → multimodal/processors}/internvl.py +3 -10
sglang/srt/{managers/multimodal_processors → multimodal/processors}/janus_pro.py +3 -9
sglang/srt/{managers/multimodal_processors → multimodal/processors}/kimi_vl.py +6 -13
sglang/srt/{managers/multimodal_processors → multimodal/processors}/llava.py +2 -10
sglang/srt/{managers/multimodal_processors → multimodal/processors}/minicpm.py +5 -12
sglang/srt/{managers/multimodal_processors → multimodal/processors}/mlama.py +2 -14
sglang/srt/{managers/multimodal_processors → multimodal/processors}/mllama4.py +3 -6
sglang/srt/{managers/multimodal_processors → multimodal/processors}/phi4mm.py +4 -14
sglang/srt/{managers/multimodal_processors → multimodal/processors}/pixtral.py +3 -9
sglang/srt/{managers/multimodal_processors → multimodal/processors}/qwen_vl.py +8 -14
sglang/srt/{managers/multimodal_processors → multimodal/processors}/vila.py +13 -31
sglang/srt/operations_strategy.py +6 -2
sglang/srt/reasoning_parser.py +26 -0
sglang/srt/sampling/sampling_batch_info.py +39 -1
sglang/srt/server_args.py +85 -24
sglang/srt/speculative/build_eagle_tree.py +57 -18
sglang/srt/speculative/eagle_worker.py +6 -4
sglang/srt/two_batch_overlap.py +204 -28
sglang/srt/utils.py +369 -138
sglang/srt/warmup.py +12 -3
sglang/test/runners.py +10 -1
sglang/test/test_utils.py +15 -3
sglang/version.py +1 -1
{sglang-0.4.8.dist-info → sglang-0.4.9.dist-info}/METADATA +9 -6
{sglang-0.4.8.dist-info → sglang-0.4.9.dist-info}/RECORD +149 -137
sglang/math_utils.py +0 -8
/sglang/srt/{managers → eplb}/eplb_algorithms/deepseek.py +0 -0
/sglang/srt/{managers → eplb}/eplb_algorithms/deepseek_vec.py +0 -0
/sglang/srt/{eplb_simulator → eplb/eplb_simulator}/__init__.py +0 -0
/sglang/srt/{mm_utils.py → multimodal/mm_utils.py} +0 -0
{sglang-0.4.8.dist-info → sglang-0.4.9.dist-info}/WHEEL +0 -0
{sglang-0.4.8.dist-info → sglang-0.4.9.dist-info}/licenses/LICENSE +0 -0
{sglang-0.4.8.dist-info → sglang-0.4.9.dist-info}/top_level.txt +0 -0

sglang/srt/models/qwen3_moe.py CHANGED Viewed

@@ -18,7 +18,7 @@
 """Inference-only Qwen3MoE model compatible with HuggingFace weights."""
 import logging
-from typing import Any, Dict, Iterable, Optional, Tuple
+from typing import Any, Dict, Iterable, List, Optional, Tuple
 import torch
 from torch import nn
@@ -32,6 +32,9 @@ from sglang.srt.distributed import (
     tensor_model_parallel_all_gather,
     tensor_model_parallel_all_reduce,
 )
+from sglang.srt.eplb.expert_distribution import get_global_expert_distribution_recorder
+from sglang.srt.eplb.expert_location import ModelConfigForExpertLocation
+from sglang.srt.eplb.expert_location_dispatch import ExpertLocationDispatchInfo
 from sglang.srt.layers.activation import SiluAndMul
 from sglang.srt.layers.communicator import LayerCommunicator, LayerScatterModes
 from sglang.srt.layers.dp_attention import (
@@ -63,12 +66,8 @@ from sglang.srt.layers.vocab_parallel_embedding import (
     ParallelLMHead,
     VocabParallelEmbedding,
 )
-from sglang.srt.managers.expert_distribution import (
-    get_global_expert_distribution_recorder,
-)
-from sglang.srt.managers.expert_location import ModelConfigForExpertLocation
-from sglang.srt.managers.expert_location_dispatch import ExpertLocationDispatchInfo
 from sglang.srt.managers.schedule_batch import global_server_args_dict
+from sglang.srt.model_executor.cuda_graph_runner import get_is_capture_mode
 from sglang.srt.model_executor.forward_batch_info import (
     ForwardBatch,
     ForwardMode,
@@ -78,11 +77,12 @@ from sglang.srt.model_loader.weight_utils import default_weight_loader
 from sglang.srt.models.qwen2_moe import Qwen2MoeMLP as Qwen3MoeMLP
 from sglang.srt.models.qwen2_moe import Qwen2MoeModel
 from sglang.srt.two_batch_overlap import MaybeTboDeepEPDispatcher
-from sglang.srt.utils import DeepEPMode, add_prefix, is_non_idle_and_non_empty
+from sglang.srt.utils import DeepEPMode, add_prefix, is_cuda, is_non_idle_and_non_empty
 Qwen3MoeConfig = None
 logger = logging.getLogger(__name__)
+_is_cuda = is_cuda()
 class Qwen3MoeSparseMoeBlock(nn.Module):
@@ -117,6 +117,15 @@ class Qwen3MoeSparseMoeBlock(nn.Module):
                 if global_server_args_dict["enable_deepep_moe"]
                 else {}
             ),
+            # Additional args for FusedMoE
+            **(
+                dict(
+                    enable_flashinfer_moe=True,
+                    enable_ep_moe=global_server_args_dict["enable_ep_moe"],
+                )
+                if global_server_args_dict["enable_flashinfer_moe"]
+                else {}
+            ),
         )
         self.gate = ReplicatedLinear(
@@ -220,7 +229,7 @@ class Qwen3MoeSparseMoeBlock(nn.Module):
                 hidden_states=hidden_states,
                 topk_idx=topk_idx,
                 topk_weights=topk_weights,
-                forward_mode=forward_mode,
+                forward_batch=forward_batch,
             )
         final_hidden_states = self.experts(
             hidden_states=hidden_states,
@@ -231,14 +240,14 @@ class Qwen3MoeSparseMoeBlock(nn.Module):
             masked_m=masked_m,
             expected_m=expected_m,
             num_recv_tokens_per_expert=num_recv_tokens_per_expert,
-            forward_mode=forward_mode,
+            forward_batch=forward_batch,
         )
         if self.ep_size > 1:
             final_hidden_states = self.deepep_dispatcher.combine(
                 hidden_states=final_hidden_states,
                 topk_idx=topk_idx,
                 topk_weights=topk_weights,
-                forward_mode=forward_mode,
+                forward_batch=forward_batch,
             )
         return final_hidden_states
@@ -284,7 +293,7 @@ class Qwen3MoeSparseMoeBlock(nn.Module):
                 hidden_states=state.pop("hidden_states_mlp_input"),
                 topk_idx=state.pop("topk_idx_local"),
                 topk_weights=state.pop("topk_weights_local"),
-                forward_mode=state.forward_batch.forward_mode,
+                forward_batch=state.forward_batch,
                 tbo_subbatch_index=state.get("tbo_subbatch_index"),
             )
@@ -316,7 +325,7 @@ class Qwen3MoeSparseMoeBlock(nn.Module):
             masked_m=state.pop("masked_m"),
             expected_m=state.pop("expected_m"),
             num_recv_tokens_per_expert=state.pop("num_recv_tokens_per_expert"),
-            forward_mode=state.forward_batch.forward_mode,
+            forward_batch=state.forward_batch,
         )
     def op_combine_a(self, state):
@@ -325,7 +334,7 @@ class Qwen3MoeSparseMoeBlock(nn.Module):
                 hidden_states=state.pop("hidden_states_experts_output"),
                 topk_idx=state.pop("topk_idx_dispatched"),
                 topk_weights=state.pop("topk_weights_dispatched"),
-                forward_mode=state.forward_batch.forward_mode,
+                forward_batch=state.forward_batch,
                 tbo_subbatch_index=state.get("tbo_subbatch_index"),
             )
@@ -354,6 +363,7 @@ class Qwen3MoeAttention(nn.Module):
         attention_bias: bool = False,
         quant_config: Optional[QuantizationConfig] = None,
         prefix: str = "",
+        alt_stream: Optional[torch.cuda.Stream] = None,
     ) -> None:
         super().__init__()
         self.hidden_size = hidden_size
@@ -423,15 +433,27 @@ class Qwen3MoeAttention(nn.Module):
         self.q_norm = RMSNorm(self.head_dim, eps=rms_norm_eps)
         self.k_norm = RMSNorm(self.head_dim, eps=rms_norm_eps)
+        self.alt_stream = alt_stream
     def _apply_qk_norm(
         self, q: torch.Tensor, k: torch.Tensor
     ) -> Tuple[torch.Tensor, torch.Tensor]:
-        q_by_head = q.reshape(-1, self.head_dim)
-        q_by_head = self.q_norm(q_by_head)
+        # overlap qk norm
+        if self.alt_stream is not None and get_is_capture_mode():
+            current_stream = torch.cuda.current_stream()
+            self.alt_stream.wait_stream(current_stream)
+            q_by_head = q.reshape(-1, self.head_dim)
+            q_by_head = self.q_norm(q_by_head)
+            with torch.cuda.stream(self.alt_stream):
+                k_by_head = k.reshape(-1, self.head_dim)
+                k_by_head = self.k_norm(k_by_head)
+            current_stream.wait_stream(self.alt_stream)
+        else:
+            q_by_head = q.reshape(-1, self.head_dim)
+            q_by_head = self.q_norm(q_by_head)
+            k_by_head = k.reshape(-1, self.head_dim)
+            k_by_head = self.k_norm(k_by_head)
         q = q_by_head.view(q.shape)
-        k_by_head = k.reshape(-1, self.head_dim)
-        k_by_head = self.k_norm(k_by_head)
         k = k_by_head.view(k.shape)
         return q, k
@@ -491,6 +513,7 @@ class Qwen3MoeDecoderLayer(nn.Module):
         layer_id: int,
         quant_config: Optional[QuantizationConfig] = None,
         prefix: str = "",
+        alt_stream: Optional[torch.cuda.Stream] = None,
     ) -> None:
         super().__init__()
         self.config = config
@@ -516,6 +539,7 @@ class Qwen3MoeDecoderLayer(nn.Module):
             attention_bias=attention_bias,
             quant_config=quant_config,
             prefix=add_prefix("self_attn", prefix),
+            alt_stream=alt_stream,
         )
         self.layer_id = layer_id
@@ -623,9 +647,7 @@ class Qwen3MoeDecoderLayer(nn.Module):
     def op_mlp(self, state):
         hidden_states = state.pop("hidden_states_mlp_input")
-        state.hidden_states_mlp_output = self.mlp(
-            hidden_states, state.forward_batch.forward_mode
-        )
+        state.hidden_states_mlp_output = self.mlp(hidden_states, state.forward_batch)
     def op_comm_postprocess_layer(self, state):
         hidden_states, residual = self.layer_communicator.postprocess_layer(
@@ -659,11 +681,13 @@ class Qwen3MoeModel(Qwen2MoeModel):
         quant_config: Optional[QuantizationConfig] = None,
         prefix: str = "",
     ) -> None:
+        alt_stream = torch.cuda.Stream() if _is_cuda else None
         super().__init__(
             config=config,
             quant_config=quant_config,
             prefix=prefix,
             decoder_layer_type=Qwen3MoeDecoderLayer,
+            alt_stream=alt_stream,
         )
@@ -691,6 +715,7 @@ class Qwen3MoeForCausalLM(nn.Module):
             use_attn_tp_group=global_server_args_dict["enable_dp_lm_head"],
         )
         self.logits_processor = LogitsProcessor(config)
+        self.capture_aux_hidden_states = False
     @torch.no_grad()
     def forward(
@@ -709,9 +734,13 @@ class Qwen3MoeForCausalLM(nn.Module):
             pp_proxy_tensors=pp_proxy_tensors,
         )
+        aux_hidden_states = None
+        if self.capture_aux_hidden_states:
+            hidden_states, aux_hidden_states = hidden_states
         if self.pp_group.is_last_rank:
             return self.logits_processor(
-                input_ids, hidden_states, self.lm_head, forward_batch
+                input_ids, hidden_states, self.lm_head, forward_batch, aux_hidden_states
             )
         else:
             return hidden_states
@@ -724,6 +753,24 @@ class Qwen3MoeForCausalLM(nn.Module):
     def end_layer(self):
         return self.model.end_layer
+    def get_embed_and_head(self):
+        return self.model.embed_tokens.weight, self.lm_head.weight
+    def set_eagle3_layers_to_capture(self, layer_ids: Optional[List[int]] = None):
+        if not self.pp_group.is_last_rank:
+            return
+        self.capture_aux_hidden_states = True
+        if layer_ids is None:
+            num_layers = self.config.num_hidden_layers
+            self.model.layers_to_capture = [
+                2,
+                num_layers // 2,
+                num_layers - 3,
+            ]  # Specific layers for EAGLE3 support
+        else:
+            self.model.layers_to_capture = [val + 1 for val in layer_ids]
     def load_weights(self, weights: Iterable[Tuple[str, torch.Tensor]]):
         stacked_params_mapping = [
             # (param_name, shard_name, shard_id)

sglang/srt/models/vila.py CHANGED Viewed

@@ -270,15 +270,10 @@ class VILAForConditionalGeneration(nn.Module):
                 weight_loader(param, loaded_weight)
     def pad_input_ids(
-        self,
-        input_ids: List[int],
-        image_inputs: MultimodalInputs,
+        self, input_ids: List[int], mm_inputs: MultimodalInputs
     ) -> List[int]:
-        pattern = MultiModalityDataPaddingPatternMultimodalTokens(
-            token_ids=[self.config.image_token_id],
-        )
-        return pattern.pad_input_tokens(input_ids, image_inputs)
+        pattern = MultiModalityDataPaddingPatternMultimodalTokens()
+        return pattern.pad_input_tokens(input_ids, mm_inputs)
     ##### BEGIN COPY modeling_vila.py #####

sglang/srt/{managers/multimodal_processors → multimodal/processors}/base_processor.py RENAMED Viewed

@@ -17,14 +17,6 @@ from sglang.srt.managers.schedule_batch import Modality, MultimodalDataItem
 from sglang.srt.utils import encode_video, load_audio, load_image
-class MultimodalInputFormat(Enum):
-    """Enum for different multimodal input formats."""
-    RAW_IMAGES = "raw_images"
-    PRECOMPUTED_FEATURES = "precomputed_features"
-    PIXEL_VALUES = "pixel_values"
 @dataclasses.dataclass
 class BaseMultiModalProcessorOutput:
     # input_text, with each frame of video/image represented with a image_token
@@ -97,6 +89,7 @@ class BaseMultimodalProcessor(ABC):
         self._processor = _processor
         self.arch = hf_config.architectures[0]
         self.server_args = server_args
         # FIXME: not accurate, model and image specific
         self.NUM_TOKEN_PER_FRAME = 330
@@ -108,18 +101,45 @@ class BaseMultimodalProcessor(ABC):
             max_workers=int(os.environ.get("SGLANG_CPU_WORKERS", os.cpu_count())),
         )
+        # Mapping from attribute names to modality types
+        self.ATTR_NAME_TO_MODALITY = {
+            # Image-related attributes
+            "pixel_values": Modality.IMAGE,
+            "image_sizes": Modality.IMAGE,
+            "image_grid_thw": Modality.IMAGE,
+            "image_emb_mask": Modality.IMAGE,
+            "image_spatial_crop": Modality.IMAGE,
+            "tgt_size": Modality.IMAGE,
+            "image_grid_hws": Modality.IMAGE,
+            "aspect_ratio_id": Modality.IMAGE,
+            "aspect_ratio_mask": Modality.IMAGE,
+            "second_per_grid_ts": Modality.IMAGE,
+            # Audio-related attributes
+            "audio_features": Modality.AUDIO,
+            "audio_feature_lens": Modality.AUDIO,
+            "input_features": Modality.AUDIO,
+            "input_features_mask": Modality.AUDIO,
+            # Video-related attributes
+            "video_grid_thws": Modality.VIDEO,
+            # Generic attributes that could apply to multiple modalities
+            # "precomputed_features" - handled specially as it can be any modality
+        }
     def process_mm_data(
         self, input_text, images=None, videos=None, audios=None, **kwargs
     ):
         """
         process multimodal data with transformers AutoProcessor
         """
-        if images is not None:
+        if images:
             kwargs["images"] = images
-        if videos is not None:
+        if videos:
             kwargs["videos"] = videos
-        if audios is not None:
+        if audios:
             kwargs["audios"] = audios
+            if self.__class__.__name__ == "Gemma3nSGLangProcessor":
+                # Note(Xinyuan): for gemma3n, ref: https://github.com/huggingface/transformers/blob/ccf2ca162e33f381e454cdb74bf4b41a51ab976d/src/transformers/models/gemma3n/processing_gemma3n.py#L107
+                kwargs["audio"] = audios
         processor = self._processor
         if hasattr(processor, "image_processor") and isinstance(
@@ -142,6 +162,7 @@ class BaseMultimodalProcessor(ABC):
     async def process_mm_data_async(
         self,
         image_data,
+        audio_data,
         input_text,
         request_obj,
         max_req_input_len,
@@ -416,141 +437,137 @@ class BaseMultimodalProcessor(ABC):
                 values[k] = v
         return values
+    def collect_mm_items_from_processor_output(
+        self, data_dict: dict
+    ) -> List[MultimodalDataItem]:
+        """Create mm_items directly from processor output."""
+        items = {}  # modality -> MultimodalDataItem
+        for attr_name, value in data_dict.items():
+            if attr_name == "input_ids":
+                continue
+            # Get modality for this attribute
+            modality = self.ATTR_NAME_TO_MODALITY.get(attr_name)
+            if not modality and attr_name == "precomputed_features":
+                modality_str = data_dict.get("modality")
+                try:
+                    modality = (
+                        Modality.from_str(modality_str)
+                        if modality_str
+                        else Modality.IMAGE
+                    )
+                except ValueError:
+                    modality = Modality.IMAGE
+            if modality:
+                # Create item if needed
+                if modality not in items:
+                    items[modality] = MultimodalDataItem(modality=modality)
+                # Set attribute
+                if hasattr(items[modality], attr_name):
+                    setattr(items[modality], attr_name, value)
+        return list(items.values())
+    def _process_and_collect_mm_items(
+        self, input_text: str, images=None, audios=None, videos=None, **kwargs
+    ) -> Tuple[List[MultimodalDataItem], torch.Tensor]:
+        """
+        Helper method to process multimodal data and create mm_items in one step.
+        Returns:
+            Tuple of (created mm_items, input_ids)
+        """
+        ret = self.process_mm_data(
+            input_text=input_text, images=images, audios=audios, videos=videos, **kwargs
+        )
+        input_ids = ret["input_ids"].flatten()
+        collected_items = self.collect_mm_items_from_processor_output(ret)
+        return collected_items, input_ids
     def process_and_combine_mm_data(
         self, base_output: BaseMultiModalProcessorOutput
-    ) -> Tuple[Optional[MultimodalDataItem], torch.Tensor]:
+    ) -> Tuple[List[MultimodalDataItem], torch.Tensor]:
         """
-        Process multimodal data and return the combined multimodal item and input_ids.
-        Handles all three input formats at the same abstraction level.
+        Process multimodal data and return the combined multimodal items and input_ids.
+        Supports mixed modalities (images and audio in the same request).
         Returns:
-            Tuple of (combined_mm_item, input_ids)
+            Tuple of (list of mm_items, input_ids)
         """
+        # Collect all items and categorize them
+        all_items = (base_output.images or []) + (base_output.audios or [])
-        def tokenize_text(input_text: str) -> torch.Tensor:
-            """Tokenize input text."""
-            return self._processor.tokenizer(
-                input_text,
+        # Handle text-only case
+        if not all_items:
+            input_ids = self._processor.tokenizer(
+                base_output.input_text,
                 return_tensors="pt",
                 add_special_tokens=True,
             ).input_ids.flatten()
+            return [], input_ids
+        dict_items, raw_images, raw_audios = [], [], []
+        for item in all_items:
+            if isinstance(item, dict):
+                dict_items.append(item)
+            elif isinstance(item, Image.Image):
+                raw_images.append(item)
+            elif isinstance(item, np.ndarray):
+                raw_audios.append(item)
+            else:
+                raise ValueError(f"Unknown multimodal item type: {type(item)}")
-        def categorize_mm_inputs(mm_inputs: List) -> MultimodalInputFormat:
-            """Categorize multimodal inputs and validate consistency."""
-            try:
-                has_image = False
-                has_pixel_values = False
-                has_precomputed_features = False
-                for mm_input in mm_inputs:
-                    if isinstance(mm_input, Image.Image):
-                        has_image = True
-                    elif isinstance(mm_input, dict):
-                        if mm_input.get("precomputed_features", None) is not None:
-                            has_precomputed_features = True
-                        elif mm_input.get("pixel_values", None) is not None:
-                            has_pixel_values = True
-                        else:
-                            raise ValueError(
-                                f"Invalid multimodal input: {mm_input}, expected dict with pixel_values or precomputed_features"
-                            )
-                    else:
-                        raise ValueError(
-                            f"Invalid multimodal input: {mm_input}, expected Image.Image or dict"
-                        )
+        # Process items and get input_ids
+        all_collected_items = []
+        input_ids = None
-                # Validate format consistency
-                format_count = sum(
-                    [has_image, has_pixel_values, has_precomputed_features]
-                )
-                if format_count > 1:
-                    raise ValueError(
-                        "Unsupported: mixture of multimodal input formats. "
-                        f"Found formats: image={has_image}, pixel_values={has_pixel_values}, "
-                        f"precomputed_features={has_precomputed_features}"
-                    )
+        # Handle dict items (already processed)
+        for dict_item in dict_items:
+            all_collected_items.extend(
+                self.collect_mm_items_from_processor_output(dict_item)
+            )
-                if has_image:
-                    return MultimodalInputFormat.RAW_IMAGES
-                elif has_precomputed_features:
-                    return MultimodalInputFormat.PRECOMPUTED_FEATURES
-                elif has_pixel_values:
-                    return MultimodalInputFormat.PIXEL_VALUES
-                else:
-                    raise ValueError("No valid multimodal input format found")
-            except Exception as e:
-                raise ValueError(f"Failed to categorize inputs: {e}")
-        def process_raw_images(
-            base_output: BaseMultiModalProcessorOutput,
-        ) -> Tuple[MultimodalDataItem, torch.Tensor]:
-            """Process raw Image.Image objects using transformers processor."""
-            ret = self.process_mm_data(
+        # Handle raw items (need processing)
+        if raw_images or raw_audios:
+            collected_items, input_ids = self._process_and_collect_mm_items(
                 input_text=base_output.input_text,
-                images=base_output.images,
+                images=raw_images,
+                audios=raw_audios,
             )
-            combined_mm_item = MultimodalDataItem(modality=Modality.IMAGE)
-            # Copy all fields from processor output except input_ids
-            for key, value in ret.items():
-                if key != "input_ids" and hasattr(combined_mm_item, key):
-                    setattr(combined_mm_item, key, value)
-            input_ids = ret["input_ids"].flatten()
-            return combined_mm_item, input_ids
-        def process_precomputed_features(
-            base_output: BaseMultiModalProcessorOutput,
-        ) -> Tuple[MultimodalDataItem, torch.Tensor]:
-            """Process inputs with precomputed features."""
-            combined_mm_item = MultimodalDataItem(modality=Modality.IMAGE)
-            combined_mm_item.precomputed_features = self._extract_processor_features(
-                base_output.images, "precomputed_features"
-            )
-            input_ids = tokenize_text(base_output.input_text)
-            return combined_mm_item, input_ids
-        def process_pixel_values(
-            base_output: BaseMultiModalProcessorOutput,
-        ) -> Tuple[MultimodalDataItem, torch.Tensor]:
-            """Process inputs with pixel values."""
-            values = self._extract_processor_features_from_all_attributes(
-                base_output.images
-            )
-            combined_mm_item = MultimodalDataItem.from_dict(values)
-            input_ids = tokenize_text(base_output.input_text)
-            return combined_mm_item, input_ids
-        def finalize_mm_item(
-            combined_mm_item: MultimodalDataItem, input_ids: torch.Tensor
-        ) -> MultimodalDataItem:
-            """Apply common post-processing to the multimodal item."""
-            combined_mm_item.image_offsets = self.get_mm_items_offset(
-                input_ids=input_ids,
-                mm_token_id=self.IM_TOKEN_ID,
-            )
-            return combined_mm_item
-        # Main logic
-        mm_inputs = base_output.images
-        if not mm_inputs:
-            # Return text-only case
-            input_ids = tokenize_text(base_output.input_text)
-            return None, input_ids
-        # Categorize input formats
-        input_format = categorize_mm_inputs(mm_inputs)
-        # Process based on format
-        if input_format == MultimodalInputFormat.RAW_IMAGES:
-            combined_mm_item, input_ids = process_raw_images(base_output)
-        elif input_format == MultimodalInputFormat.PRECOMPUTED_FEATURES:
-            combined_mm_item, input_ids = process_precomputed_features(base_output)
-        elif input_format == MultimodalInputFormat.PIXEL_VALUES:
-            combined_mm_item, input_ids = process_pixel_values(base_output)
-        else:
-            raise ValueError(f"Unknown input format: {input_format}")
+            all_collected_items.extend(collected_items)
+        # Fallback tokenization if no raw items were processed
+        if input_ids is None:
+            input_ids = self._processor.tokenizer(
+                base_output.input_text,
+                return_tensors="pt",
+                add_special_tokens=True,
+            ).input_ids.flatten()
+        # Add offsets to all items
+        for mm_item in all_collected_items:
+            if mm_item.modality in [Modality.IMAGE, Modality.MULTI_IMAGES]:
+                mm_item.image_offsets = self.get_mm_items_offset(
+                    input_ids=input_ids,
+                    mm_token_id=self.IM_TOKEN_ID,
+                )
+            elif mm_item.modality == Modality.AUDIO:
+                mm_item.audio_offsets = self.get_mm_items_offset(
+                    input_ids=input_ids,
+                    mm_token_id=self.AUDIO_TOKEN_ID,
+                )
+            elif mm_item.modality == Modality.VIDEO:
+                mm_item.video_offsets = self.get_mm_items_offset(
+                    input_ids=input_ids,
+                    mm_token_id=self.VIDEO_TOKEN_ID,
+                )
+            else:
+                raise ValueError(f"Unknown modality: {mm_item.modality}")
-        # Finalize with common processing
-        combined_mm_item = finalize_mm_item(combined_mm_item, input_ids)
-        return combined_mm_item, input_ids
+        return all_collected_items, input_ids

sglang/srt/{managers/multimodal_processors → multimodal/processors}/clip.py RENAMED Viewed

@@ -1,10 +1,8 @@
 from typing import List, Union
-from sglang.srt.managers.multimodal_processors.base_processor import (
-    BaseMultimodalProcessor,
-)
 from sglang.srt.managers.schedule_batch import Modality, MultimodalDataItem
 from sglang.srt.models.clip import CLIPModel
+from sglang.srt.multimodal.processors.base_processor import BaseMultimodalProcessor
 from sglang.srt.utils import load_image
@@ -17,20 +15,11 @@ class ClipImageProcessor(BaseMultimodalProcessor):
     async def process_mm_data_async(
         self, image_data: List[Union[str, bytes]], input_text, *args, **kwargs
     ):
-        if not image_data:
-            return None
         if isinstance(input_text, list):
             assert len(input_text) and isinstance(input_text[0], int)
             input_text = self._processor.tokenizer.decode(input_text)
-        if not isinstance(image_data, list):
-            image_data = [image_data]
-        if len(image_data) > 0:
-            images = [load_image(image)[0] for image in image_data]
-        else:
-            images = load_image(image_data[0])[0]
+        images = [load_image(image)[0] for image in image_data]
         image_inputs = self.process_mm_data(input_text=input_text, images=images)
         image_inputs["data_hashes"] = [hash(str(image_data))]

sglang/srt/{managers/multimodal_processors → multimodal/processors}/deepseek_vl_v2.py RENAMED Viewed

@@ -20,12 +20,12 @@ from typing import List, Union
 import torch
-from sglang.srt.managers.multimodal_processors.base_processor import (
+from sglang.srt.managers.schedule_batch import Modality, MultimodalDataItem
+from sglang.srt.models.deepseek_vl2 import DeepseekVL2ForCausalLM
+from sglang.srt.multimodal.processors.base_processor import (
     BaseMultimodalProcessor,
     MultimodalSpecialTokens,
 )
-from sglang.srt.managers.schedule_batch import Modality, MultimodalDataItem
-from sglang.srt.models.deepseek_vl2 import DeepseekVL2ForCausalLM
 class DeepseekVL2ImageProcessor(BaseMultimodalProcessor):
@@ -44,17 +44,10 @@ class DeepseekVL2ImageProcessor(BaseMultimodalProcessor):
         *args,
         **kwargs
     ):
-        if not image_data:
-            return None
-        if not isinstance(image_data, list):
-            image_data = [image_data]
-        image_token = self.IMAGE_TOKEN
         base_output = self.load_mm_data(
             input_text,
             image_data=image_data,
-            multimodal_tokens=MultimodalSpecialTokens(image_token=image_token),
+            multimodal_tokens=MultimodalSpecialTokens(image_token=self.IMAGE_TOKEN),
             max_req_input_len=max_req_input_len,
         )
         res = self.process_mm_data(

sglang 0.4.8__py3-none-any.whl → 0.4.9__py3-none-any.whl

sglang 0.4.8py3-none-any.whl → 0.4.9py3-none-any.whl