PyPI - sglang - Versions diffs - 0.4.5__py3-none-any.whl → 0.4.5.post1__py3-none-any.whl - Mend

sglang 0.4.5py3-none-any.whl → 0.4.5.post1py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (121) hide show

sglang/bench_one_batch.py +21 -0
sglang/bench_serving.py +10 -4
sglang/srt/configs/model_config.py +37 -5
sglang/srt/constrained/base_grammar_backend.py +26 -5
sglang/srt/constrained/llguidance_backend.py +1 -0
sglang/srt/constrained/outlines_backend.py +1 -0
sglang/srt/constrained/reasoner_grammar_backend.py +101 -0
sglang/srt/constrained/xgrammar_backend.py +1 -0
sglang/srt/disaggregation/base/__init__.py +8 -0
sglang/srt/disaggregation/base/conn.py +113 -0
sglang/srt/disaggregation/decode.py +18 -5
sglang/srt/disaggregation/mini_lb.py +53 -122
sglang/srt/disaggregation/mooncake/__init__.py +6 -0
sglang/srt/disaggregation/mooncake/conn.py +615 -0
sglang/srt/disaggregation/mooncake/transfer_engine.py +108 -0
sglang/srt/disaggregation/prefill.py +43 -19
sglang/srt/disaggregation/utils.py +31 -0
sglang/srt/entrypoints/EngineBase.py +53 -0
sglang/srt/entrypoints/engine.py +36 -8
sglang/srt/entrypoints/http_server.py +37 -8
sglang/srt/entrypoints/http_server_engine.py +142 -0
sglang/srt/entrypoints/verl_engine.py +37 -10
sglang/srt/hf_transformers_utils.py +4 -0
sglang/srt/layers/attention/flashattention_backend.py +330 -200
sglang/srt/layers/attention/flashinfer_backend.py +13 -7
sglang/srt/layers/attention/vision.py +1 -1
sglang/srt/layers/dp_attention.py +2 -4
sglang/srt/layers/elementwise.py +15 -2
sglang/srt/layers/linear.py +1 -0
sglang/srt/layers/moe/ep_moe/token_dispatcher.py +145 -118
sglang/srt/layers/moe/fused_moe_triton/configs/E=256,N=128,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128, 128].json +146 -0
sglang/srt/layers/moe/fused_moe_triton/configs/E=264,N=128,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8.json +146 -0
sglang/srt/layers/moe/fused_moe_triton/configs/{E=257,N=256,device_name=NVIDIA_H200,dtype=fp8_w8a8,block_shape=[128, 128].json → E=264,N=256,device_name=NVIDIA_H20,dtype=fp8_w8a8,block_shape=[128, 128].json } +34 -34
sglang/srt/layers/moe/fused_moe_triton/configs/E=272,N=128,device_name=NVIDIA_A800-SXM4-80GB,dtype=int8_w8a8.json +146 -0
sglang/srt/layers/moe/fused_moe_triton/configs/E=272,N=64,device_name=NVIDIA_A800-SXM4-80GB.json +146 -0
sglang/srt/layers/moe/fused_moe_triton/configs/E=288,N=64,device_name=NVIDIA_A800-SXM4-80GB.json +146 -0
sglang/srt/layers/moe/fused_moe_triton/fused_moe.py +38 -21
sglang/srt/layers/moe/router.py +7 -1
sglang/srt/layers/moe/topk.py +37 -16
sglang/srt/layers/quantization/__init__.py +12 -5
sglang/srt/layers/quantization/compressed_tensors/compressed_tensors.py +4 -0
sglang/srt/layers/quantization/compressed_tensors/compressed_tensors_moe.py +68 -45
sglang/srt/layers/quantization/fp8.py +25 -13
sglang/srt/layers/quantization/fp8_kernel.py +130 -4
sglang/srt/layers/quantization/fp8_utils.py +34 -6
sglang/srt/layers/quantization/kv_cache.py +43 -52
sglang/srt/layers/quantization/modelopt_quant.py +271 -4
sglang/srt/layers/quantization/w8a8_fp8.py +154 -4
sglang/srt/layers/quantization/w8a8_int8.py +1 -0
sglang/srt/layers/radix_attention.py +13 -1
sglang/srt/layers/rotary_embedding.py +12 -1
sglang/srt/managers/io_struct.py +254 -97
sglang/srt/managers/mm_utils.py +3 -2
sglang/srt/managers/multimodal_processors/base_processor.py +114 -77
sglang/srt/managers/multimodal_processors/janus_pro.py +3 -1
sglang/srt/managers/multimodal_processors/mllama4.py +21 -36
sglang/srt/managers/schedule_batch.py +62 -21
sglang/srt/managers/scheduler.py +71 -14
sglang/srt/managers/tokenizer_manager.py +17 -3
sglang/srt/managers/tp_worker.py +1 -0
sglang/srt/mem_cache/memory_pool.py +14 -1
sglang/srt/metrics/collector.py +9 -0
sglang/srt/model_executor/cuda_graph_runner.py +7 -4
sglang/srt/model_executor/forward_batch_info.py +234 -15
sglang/srt/model_executor/model_runner.py +48 -9
sglang/srt/model_loader/loader.py +31 -4
sglang/srt/model_loader/weight_utils.py +4 -2
sglang/srt/models/baichuan.py +2 -0
sglang/srt/models/chatglm.py +1 -0
sglang/srt/models/commandr.py +1 -0
sglang/srt/models/dbrx.py +1 -0
sglang/srt/models/deepseek.py +1 -0
sglang/srt/models/deepseek_v2.py +248 -61
sglang/srt/models/exaone.py +1 -0
sglang/srt/models/gemma.py +1 -0
sglang/srt/models/gemma2.py +1 -0
sglang/srt/models/gemma3_causal.py +1 -0
sglang/srt/models/gpt2.py +1 -0
sglang/srt/models/gpt_bigcode.py +1 -0
sglang/srt/models/granite.py +1 -0
sglang/srt/models/grok.py +1 -0
sglang/srt/models/internlm2.py +1 -0
sglang/srt/models/llama.py +1 -0
sglang/srt/models/llama4.py +101 -34
sglang/srt/models/minicpm.py +1 -0
sglang/srt/models/minicpm3.py +2 -0
sglang/srt/models/mixtral.py +1 -0
sglang/srt/models/mixtral_quant.py +1 -0
sglang/srt/models/mllama.py +51 -8
sglang/srt/models/mllama4.py +102 -29
sglang/srt/models/olmo.py +1 -0
sglang/srt/models/olmo2.py +1 -0
sglang/srt/models/olmoe.py +1 -0
sglang/srt/models/phi3_small.py +1 -0
sglang/srt/models/qwen.py +1 -0
sglang/srt/models/qwen2.py +1 -0
sglang/srt/models/qwen2_5_vl.py +35 -70
sglang/srt/models/qwen2_moe.py +1 -0
sglang/srt/models/qwen2_vl.py +27 -25
sglang/srt/models/stablelm.py +1 -0
sglang/srt/models/xverse.py +1 -0
sglang/srt/models/xverse_moe.py +1 -0
sglang/srt/openai_api/adapter.py +4 -1
sglang/srt/patch_torch.py +11 -0
sglang/srt/server_args.py +34 -0
sglang/srt/speculative/eagle_draft_cuda_graph_runner.py +4 -4
sglang/srt/speculative/eagle_utils.py +1 -11
sglang/srt/speculative/eagle_worker.py +6 -2
sglang/srt/utils.py +120 -9
sglang/test/attention/test_flashattn_backend.py +259 -221
sglang/test/attention/test_flashattn_mla_backend.py +285 -0
sglang/test/attention/test_prefix_chunk_info.py +224 -0
sglang/test/test_block_fp8.py +57 -0
sglang/test/test_utils.py +19 -8
sglang/version.py +1 -1
{sglang-0.4.5.dist-info → sglang-0.4.5.post1.dist-info}/METADATA +14 -4
{sglang-0.4.5.dist-info → sglang-0.4.5.post1.dist-info}/RECORD +120 -106
sglang/srt/disaggregation/conn.py +0 -81
{sglang-0.4.5.dist-info → sglang-0.4.5.post1.dist-info}/WHEEL +0 -0
{sglang-0.4.5.dist-info → sglang-0.4.5.post1.dist-info}/licenses/LICENSE +0 -0
{sglang-0.4.5.dist-info → sglang-0.4.5.post1.dist-info}/top_level.txt +0 -0

sglang/test/attention/test_flashattn_mla_backend.py ADDED Viewed

@@ -0,0 +1,285 @@
+import unittest
+import torch
+from sglang.srt.configs.model_config import AttentionArch
+from sglang.srt.layers.attention.flashattention_backend import FlashAttentionBackend
+from sglang.srt.layers.attention.torch_native_backend import TorchNativeAttnBackend
+from sglang.srt.layers.radix_attention import RadixAttention
+from sglang.srt.mem_cache.memory_pool import MLATokenToKVPool
+from sglang.srt.model_executor.forward_batch_info import ForwardBatch, ForwardMode
+from sglang.test.test_utils import CustomTestCase
+class MockModelRunner:
+    def __init__(
+        self,
+        kv_lora_rank,
+        qk_rope_head_dim,
+    ):
+        attention_arch = AttentionArch.MLA
+        self.device = "cuda"
+        self.dtype = torch.float16
+        context_len = 2048
+        self.model_config = type(
+            "ModelConfig",
+            (),
+            {
+                "context_len": context_len,
+                "attention_arch": attention_arch,
+            },
+        )
+        self.sliding_window_size = None
+        batch_size = 160
+        # Create a proper req_to_token_pool with the req_to_token attribute
+        self.req_to_token_pool = type(
+            "TokenPool",
+            (),
+            {
+                # A typical max_bs * max_context_len for cuda graph decode
+                "size": batch_size,
+                # Add req_to_token attribute
+                "req_to_token": torch.zeros(
+                    batch_size, context_len, dtype=torch.int32, device=self.device
+                ),
+            },
+        )
+        self.page_size = 1
+        max_total_num_tokens = batch_size * context_len
+        self.token_to_kv_pool = MLATokenToKVPool(
+            size=max_total_num_tokens,
+            page_size=self.page_size,
+            dtype=self.dtype,
+            kv_lora_rank=kv_lora_rank,
+            qk_rope_head_dim=qk_rope_head_dim,
+            layer_num=1,  # only consider layer=1 for unit test
+            device=self.device,
+            enable_memory_saver=False,
+        )
+class MockReqToTokenPool:
+    def __init__(self, batch_size, seq_len, device):
+        self.req_to_token = (
+            torch.arange(batch_size * seq_len, device=device)
+            .reshape(batch_size, seq_len)
+            .to(torch.int32)
+        )
+@unittest.skipIf(not torch.cuda.is_available(), "Test requires CUDA")
+class TestFlashAttentionMLABackend(CustomTestCase):
+    def setUp(self):
+        # Test parameters
+        self.batch_size = 2
+        self.seq_len = 360
+        self.num_heads = 2
+        self.device = "cuda"
+        self.dtype = torch.float16
+        self.kv_lora_rank = 512
+        self.q_lora_rank = 128
+        self.qk_rope_head_dim = 64
+        self.qk_head_dim = self.qk_rope_head_dim + self.kv_lora_rank
+        # Assume no rope scaling
+        self.scaling = self.qk_head_dim**-0.5
+        # Initialize model runner and backend
+        self._init_model_runner()
+        self.backend = FlashAttentionBackend(self.model_runner)
+        self.num_local_heads = 2
+    def _init_model_runner(self):
+        self.model_runner = MockModelRunner(
+            kv_lora_rank=self.kv_lora_rank,
+            qk_rope_head_dim=self.qk_rope_head_dim,
+        )
+        self.backend = FlashAttentionBackend(self.model_runner)
+    def _create_attention_layer(self):
+        """Create attention layer for testing."""
+        self.attn_mqa = RadixAttention(
+            num_heads=self.num_local_heads,
+            head_dim=self.kv_lora_rank + self.qk_rope_head_dim,
+            scaling=self.scaling,
+            num_kv_heads=1,
+            layer_id=0,
+            v_head_dim=self.kv_lora_rank,
+            prefix="attn_mqa",
+        )
+        return self.attn_mqa
+    def _run_reference_forward(
+        self, mode, q, k, v, layer, forward_batch, expected_shape
+    ):
+        """Run reference forward pass using native backend."""
+        if mode == ForwardMode.EXTEND:
+            output = self.ref_backend.forward_extend(q, k, v, layer, forward_batch)
+        else:  # ForwardMode.DECODE
+            output = self.ref_backend.forward_decode(q, k, v, layer, forward_batch)
+        return output.view(expected_shape)
+    def _verify_output(self, output, expected_shape):
+        """Verify output tensor shape, dtype, and values."""
+        self.assertEqual(
+            output.shape,
+            expected_shape,
+            f"Expected shape {expected_shape}, got {output.shape}",
+        )
+        self.assertEqual(output.dtype, self.dtype)
+        self.assertEqual(output.device.type, "cuda")
+        self.assertEqual(
+            torch.isnan(output).sum().item(), 0, "Output contains NaN values"
+        )
+    def _create_forward_batch(self, mode, q_len=None, prefix_len=0):
+        """Create a forward batch for testing based on mode and lengths."""
+        # Default to self.seq_len if not specified
+        q_len = q_len or self.seq_len
+        if mode == ForwardMode.EXTEND:
+            total_len = prefix_len + q_len
+            out_cache_start = prefix_len * self.batch_size
+            out_cache_end = total_len * self.batch_size
+            forward_batch = ForwardBatch(
+                batch_size=self.batch_size,
+                input_ids=torch.randint(
+                    0, 100, (self.batch_size, q_len), device=self.device
+                ),
+                out_cache_loc=torch.arange(
+                    out_cache_start, out_cache_end, device=self.device
+                ),
+                seq_lens_sum=self.batch_size * total_len,
+                forward_mode=mode,
+                req_pool_indices=torch.arange(self.batch_size, device=self.device),
+                seq_lens=torch.tensor(
+                    [total_len] * self.batch_size, device=self.device
+                ),
+                seq_lens_cpu=torch.tensor([total_len] * self.batch_size, device="cpu"),
+                extend_prefix_lens=torch.tensor(
+                    [prefix_len] * self.batch_size, device=self.device
+                ),
+                extend_prefix_lens_cpu=torch.tensor(
+                    [prefix_len] * self.batch_size, device="cpu"
+                ),
+                extend_seq_lens=torch.tensor(
+                    [q_len] * self.batch_size, device=self.device
+                ),
+                extend_seq_lens_cpu=torch.tensor(
+                    [q_len] * self.batch_size, device="cpu"
+                ),
+                attn_backend=self.backend,
+            )
+        else:  # ForwardMode.DECODE
+            decode_len = q_len  # typically 1 for decode mode
+            total_len = self.seq_len + decode_len
+            out_cache_start = self.batch_size * self.seq_len
+            out_cache_end = self.batch_size * total_len
+            forward_batch = ForwardBatch(
+                batch_size=self.batch_size,
+                input_ids=torch.randint(
+                    0, 100, (self.batch_size, decode_len), device=self.device
+                ),
+                out_cache_loc=torch.arange(
+                    out_cache_start, out_cache_end, device=self.device
+                ),
+                seq_lens_sum=self.batch_size * total_len,
+                forward_mode=mode,
+                req_pool_indices=torch.arange(self.batch_size, device=self.device),
+                seq_lens=torch.tensor(
+                    [total_len] * self.batch_size, device=self.device
+                ),
+                seq_lens_cpu=torch.tensor([total_len] * self.batch_size, device="cpu"),
+                attn_backend=self.backend,
+            )
+        # Add token pool from model runner to forward batch
+        forward_batch.req_to_token_pool = self.model_runner.req_to_token_pool
+        # Add KV cache from model runner to forward batch
+        forward_batch.token_to_kv_pool = self.model_runner.token_to_kv_pool
+        return forward_batch
+    def _setup_kv_cache(self, forward_batch, layer, cache_len):
+        """Set up KV cache with prefix tokens."""
+        if cache_len <= 0:
+            return
+        # Create constant values for the prefix cache for easy debugging
+        latent_cache = torch.ones(
+            self.batch_size * cache_len,
+            1,  # latent cache has only one head in MQA
+            self.kv_lora_rank + self.qk_rope_head_dim,
+            dtype=self.dtype,
+            device=self.device,
+        )
+        # Set the prefix KV cache
+        forward_batch.token_to_kv_pool.set_kv_buffer(
+            layer,
+            torch.arange(self.batch_size * cache_len, device=self.device),
+            latent_cache,
+            None,
+        )
+    def _run_attention_test(self, mode, q_len, prefix_len=0):
+        """
+            Run an attention test with the specified parameters.
+        Args:
+            mode: ForwardMode.EXTEND or ForwardMode.DECODE
+            q_len: Length of the query sequence. For decode mode, q_len is 1.
+            prefix_len: Length of the prefix sequence for extend mode
+        """
+        layer = self._create_attention_layer()
+        # Create forward batch and set up
+        forward_batch = self._create_forward_batch(mode, q_len, prefix_len)
+        # Create q, kv_compressed for testing
+        q_shape = (self.batch_size * q_len, self.num_heads, self.qk_head_dim)
+        kv_shape = (self.batch_size * q_len, self.qk_head_dim)
+        q = torch.randn(q_shape, dtype=self.dtype, device=self.device)
+        kv_compressed = torch.randn(kv_shape, dtype=self.dtype, device=self.device)
+        # v is not used for mqa, all values passed in through k
+        k = kv_compressed.unsqueeze(1)
+        v = torch.randn((1), dtype=self.dtype, device=self.device)
+        self._setup_kv_cache(forward_batch, layer, prefix_len)
+        self.backend.init_forward_metadata(forward_batch)
+        expected_shape = (
+            self.batch_size * q_len,
+            self.num_heads * self.kv_lora_rank,
+        )
+        if mode == ForwardMode.EXTEND:
+            output = self.backend.forward_extend(q, k, v, layer, forward_batch)
+        else:
+            output = self.backend.forward_decode(q, k, v, layer, forward_batch)
+        self._verify_output(output, expected_shape)
+        return output
+    def test_forward_extend(self):
+        """Test the standard extend operation."""
+        self._run_attention_test(ForwardMode.EXTEND, q_len=self.seq_len)
+    def test_forward_decode(self):
+        """Test the decode operation with cached tokens."""
+        self._run_attention_test(ForwardMode.DECODE, q_len=1)
+    def test_forward_extend_with_prefix(self):
+        """Test extending from cached prefix tokens."""
+        prefix_len = self.seq_len // 2
+        extend_len = self.seq_len - prefix_len
+        self._run_attention_test(
+            ForwardMode.EXTEND, q_len=extend_len, prefix_len=prefix_len
+        )
+if __name__ == "__main__":
+    unittest.main()

sglang/test/attention/test_prefix_chunk_info.py ADDED Viewed

@@ -0,0 +1,224 @@
+import unittest
+import torch
+from sglang.srt.mem_cache.memory_pool import MLATokenToKVPool
+from sglang.srt.model_executor.forward_batch_info import ForwardBatch, ForwardMode
+from sglang.test.test_utils import CustomTestCase
+TEST_CASES = [
+    # Sequence with same prefix lens
+    {
+        "batch_size": 3,
+        "prefix_lens": [64, 64, 64],
+        "max_chunk_capacity": 48,
+        "prefix_chunk_len": 16,
+        "num_prefix_chunks": 4,
+        "prefix_chunk_starts": torch.tensor(
+            [
+                [0, 0, 0],
+                [16, 16, 16],
+                [32, 32, 32],
+                [48, 48, 48],
+            ],
+            dtype=torch.int32,
+        ),
+        "prefix_chunk_seq_lens": torch.tensor(
+            [
+                [16, 16, 16],
+                [16, 16, 16],
+                [16, 16, 16],
+                [16, 16, 16],
+            ],
+            dtype=torch.int32,
+        ),
+    },
+    # Sequence with different prefix lens
+    {
+        "batch_size": 4,
+        "prefix_lens": [16, 32, 48, 64],
+        "max_chunk_capacity": 64,
+        "prefix_chunk_len": 16,
+        "num_prefix_chunks": 4,
+        "prefix_chunk_starts": torch.tensor(
+            [
+                [0, 0, 0, 0],
+                [16, 16, 16, 16],
+                [32, 32, 32, 32],
+                [48, 48, 48, 48],
+            ],
+            dtype=torch.int32,
+        ),
+        "prefix_chunk_seq_lens": torch.tensor(
+            [
+                [16, 16, 16, 16],
+                [0, 16, 16, 16],
+                [0, 0, 16, 16],
+                [0, 0, 0, 16],
+            ],
+            dtype=torch.int32,
+        ),
+    },
+    # Sequence with irregular shapes
+    {
+        "batch_size": 2,
+        "prefix_lens": [1, 64],
+        "max_chunk_capacity": 31,
+        "prefix_chunk_len": 15,
+        "num_prefix_chunks": 5,
+        "prefix_chunk_starts": torch.tensor(
+            [
+                [0, 0],
+                [15, 15],
+                [30, 30],
+                [45, 45],
+                [60, 60],
+            ],
+            dtype=torch.int32,
+        ),
+        "prefix_chunk_seq_lens": torch.tensor(
+            [
+                [1, 15],
+                [0, 15],
+                [0, 15],
+                [0, 15],
+                [0, 4],
+            ],
+            dtype=torch.int32,
+        ),
+    },
+]
+class MockForwardBatch(ForwardBatch):
+    def __init__(self, max_chunk_capacity: int, *args, **kwargs):
+        super().__init__(*args, **kwargs)
+        self.max_chunk_capacity = max_chunk_capacity
+    def get_max_chunk_capacity(self):
+        return self.max_chunk_capacity
+class MockReqToTokenPool:
+    def __init__(self, batch_size, seq_len, device):
+        self.req_to_token = (
+            torch.arange(batch_size * seq_len, device=device)
+            .reshape(batch_size, seq_len)
+            .to(torch.int32)
+        )
+# Test correctness of triton kernel for computing kv indices
+def check_kv_indices(forward_batch):
+    for i in range(forward_batch.num_prefix_chunks):
+        computed_kv_indices = forward_batch.prefix_chunk_kv_indices[i]
+        req_to_token = forward_batch.req_to_token_pool.req_to_token[
+            : forward_batch.batch_size, :
+        ]
+        ref_kv_indices = torch.empty(
+            forward_batch.prefix_chunk_num_tokens[i],
+            dtype=torch.int32,
+            device=computed_kv_indices.device,
+        )
+        running_ptr = 0
+        for j in range(forward_batch.batch_size):
+            seq_start = forward_batch.prefix_chunk_starts[i, j].item()
+            seq_len = forward_batch.prefix_chunk_seq_lens[i, j].item()
+            ref_kv_indices[running_ptr : running_ptr + seq_len].copy_(
+                req_to_token[j, seq_start : seq_start + seq_len]
+            )
+            running_ptr += seq_len
+        assert torch.allclose(computed_kv_indices, ref_kv_indices)
+@unittest.skipIf(not torch.cuda.is_available(), "Test requires CUDA")
+class TestPrefixChunkInfo(CustomTestCase):
+    def setUp(self):
+        # Common test parameters
+        self.num_local_heads = 128
+        self.kv_lora_rank = 512
+        self.qk_rope_head_dim = 64
+        self.device = torch.device("cuda")
+        self.dtype = torch.bfloat16
+        self.extend_len = 64
+        self.max_bs = 4
+        self.max_seq_len = 128
+        # req_to_token_pool
+        self.req_to_token_pool = MockReqToTokenPool(
+            self.max_bs,
+            self.max_seq_len,
+            self.device,
+        )
+        # token_to_kv_pool
+        self.token_to_kv_pool = MLATokenToKVPool(
+            size=self.max_bs * self.max_seq_len,
+            page_size=1,  # only consider page=1 for unit test
+            dtype=self.dtype,
+            kv_lora_rank=self.kv_lora_rank,
+            qk_rope_head_dim=self.qk_rope_head_dim,
+            layer_num=1,  # only consider layer=1 for unit test
+            device=self.device,
+            enable_memory_saver=False,
+        )
+    def test_prefix_chunk_info(self):
+        """Test the standard extend operation."""
+        for test_case in TEST_CASES:
+            print(
+                f"Test case with batch_size={test_case['batch_size']}, prefix_lens={test_case['prefix_lens']}, max_chunk_capacity={test_case['max_chunk_capacity']}"
+            )
+            batch_size = test_case["batch_size"]
+            prefix_lens_cpu = test_case["prefix_lens"]
+            assert len(prefix_lens_cpu) == batch_size
+            prefix_lens = torch.tensor(prefix_lens_cpu, device=self.device)
+            max_chunk_capacity = test_case["max_chunk_capacity"]
+            seq_lens_cpu = [
+                self.extend_len + prefix_lens_cpu[i] for i in range(batch_size)
+            ]
+            seq_lens = torch.tensor(seq_lens_cpu, device=self.device)
+            # Create forward batch
+            # input_ids and out_cache_loc are dummy tensors in this test
+            forward_batch = MockForwardBatch(
+                max_chunk_capacity=max_chunk_capacity,
+                batch_size=batch_size,
+                input_ids=torch.randint(
+                    0, 100, (batch_size, self.extend_len), device=self.device
+                ),
+                out_cache_loc=torch.arange(
+                    self.max_bs * self.max_seq_len - batch_size * self.extend_len,
+                    self.max_bs * self.max_seq_len,
+                    device=self.device,
+                ),
+                seq_lens_sum=sum(seq_lens_cpu),
+                forward_mode=ForwardMode.EXTEND,
+                req_pool_indices=torch.arange(batch_size, device=self.device),
+                seq_lens=seq_lens,
+                seq_lens_cpu=seq_lens_cpu,
+                extend_prefix_lens=prefix_lens,
+                extend_prefix_lens_cpu=prefix_lens_cpu,
+            )
+            forward_batch.req_to_token_pool = self.req_to_token_pool
+            forward_batch.token_to_kv_pool = self.token_to_kv_pool
+            forward_batch.prepare_chunked_prefix_cache_info(self.device)
+            assert forward_batch.get_max_chunk_capacity() == max_chunk_capacity
+            assert forward_batch.prefix_chunk_len == test_case["prefix_chunk_len"]
+            assert forward_batch.num_prefix_chunks == test_case["num_prefix_chunks"]
+            assert torch.allclose(
+                forward_batch.prefix_chunk_starts,
+                test_case["prefix_chunk_starts"].to(self.device),
+            )
+            assert torch.allclose(
+                forward_batch.prefix_chunk_seq_lens,
+                test_case["prefix_chunk_seq_lens"].to(self.device),
+            )
+            check_kv_indices(forward_batch)
+if __name__ == "__main__":
+    unittest.main()

sglang/test/test_block_fp8.py CHANGED Viewed

@@ -7,10 +7,12 @@ import torch
 from sglang.srt.layers.activation import SiluAndMul
 from sglang.srt.layers.moe.fused_moe_triton.fused_moe import fused_moe
 from sglang.srt.layers.quantization.fp8_kernel import (
+    per_tensor_quant_mla_fp8,
     per_token_group_quant_fp8,
     static_quant_fp8,
     w8a8_block_fp8_matmul,
 )
+from sglang.srt.layers.quantization.fp8_utils import input_to_float8
 from sglang.test.test_utils import CustomTestCase
 _is_cuda = torch.cuda.is_available() and torch.version.cuda
@@ -155,6 +157,61 @@ class TestStaticQuantFP8(CustomTestCase):
                 self._static_quant_fp8(*params)
+class TestPerTensorQuantMlaFP8(CustomTestCase):
+    DTYPES = [torch.half, torch.bfloat16, torch.float32]
+    NUM_TOKENS = [7, 83, 2048]
+    D = [512, 4096, 5120, 13824]
+    LAST_D_EXT = [1024, 0]
+    LAST_D = [512]
+    SEEDS = [0]
+    @classmethod
+    def setUpClass(cls):
+        if not torch.cuda.is_available():
+            raise unittest.SkipTest("CUDA is not available")
+        torch.set_default_device("cuda")
+    def _per_tensor_quant_mla_fp8(self, num_tokens, d, last_d_ext, last_d, dtype, seed):
+        torch.manual_seed(seed)
+        x = torch.rand(
+            (num_tokens, d // last_d, last_d + last_d_ext),
+            dtype=dtype,
+        )
+        x_sub, _ = x.split([last_d, last_d_ext], dim=-1)
+        with torch.inference_mode():
+            ref_out, ref_s = input_to_float8(x_sub.transpose(0, 1))
+            out, out_s = per_tensor_quant_mla_fp8(x_sub.transpose(0, 1))
+        self.assertTrue(out.is_contiguous())
+        self.assertTrue(
+            torch.allclose(out.to(torch.float32), ref_out.to(torch.float32), rtol=0.50)
+        )
+        self.assertTrue(
+            torch.allclose(out_s.to(torch.float32), ref_s.to(torch.float32))
+        )
+    def test_per_tensor_quant_mla_fp8(self):
+        for params in itertools.product(
+            self.NUM_TOKENS,
+            self.D,
+            self.LAST_D_EXT,
+            self.LAST_D,
+            self.DTYPES,
+            self.SEEDS,
+        ):
+            with self.subTest(
+                num_tokens=params[0],
+                d=params[1],
+                last_d_ext=params[2],
+                last_d=params[3],
+                dtype=params[4],
+                seed=params[5],
+            ):
+                self._per_tensor_quant_mla_fp8(*params)
 # For test
 def native_w8a8_block_fp8_matmul(A, B, As, Bs, block_size, output_dtype=torch.float16):
     """This function performs matrix multiplication with block-wise quantization using native torch.

sglang/test/test_utils.py CHANGED Viewed

@@ -25,7 +25,12 @@ from sglang.bench_serving import run_benchmark
 from sglang.global_config import global_config
 from sglang.lang.backend.openai import OpenAI
 from sglang.lang.backend.runtime_endpoint import RuntimeEndpoint
-from sglang.srt.utils import get_bool_env_var, kill_process_tree, retry
+from sglang.srt.utils import (
+    get_bool_env_var,
+    is_port_available,
+    kill_process_tree,
+    retry,
+)
 from sglang.test.run_eval import run_eval
 from sglang.utils import get_exception_traceback
@@ -37,11 +42,6 @@ DEFAULT_FP8_MODEL_NAME_FOR_DYNAMIC_QUANT_ACCURACY_TEST = (
 DEFAULT_FP8_MODEL_NAME_FOR_MODELOPT_QUANT_ACCURACY_TEST = (
     "nvidia/Llama-3.1-8B-Instruct-FP8"
 )
-# TODO(yundai424): right now specifying to an older revision since the latest one
-#  carries kv cache quantization which doesn't work yet
-DEFAULT_FP8_MODEL_NAME_FOR_MODELOPT_QUANT_ACCURACY_TEST_REVISION = (
-    "13858565416dbdc0b4e7a4a677fadfbd5b9e5bb9"
-)
 DEFAULT_MODEL_NAME_FOR_TEST = "meta-llama/Llama-3.1-8B-Instruct"
 DEFAULT_SMALL_MODEL_NAME_FOR_TEST = "meta-llama/Llama-3.2-1B-Instruct"
@@ -103,6 +103,17 @@ def call_generate_lightllm(prompt, temperature, max_tokens, stop=None, url=None)
     return pred
+def find_available_port(base_port: int):
+    port = base_port + random.randint(100, 1000)
+    while True:
+        if is_port_available(port):
+            return port
+        if port < 60000:
+            port += 42
+        else:
+            port -= 43
 def call_generate_vllm(prompt, temperature, max_tokens, stop=None, n=1, url=None):
     assert url is not None
@@ -674,8 +685,6 @@ def run_bench_one_batch(model, other_args):
         "python3",
         "-m",
         "sglang.bench_one_batch",
-        "--model-path",
-        model,
         "--batch-size",
         "1",
         "--input",
@@ -684,6 +693,8 @@ def run_bench_one_batch(model, other_args):
         "8",
         *[str(x) for x in other_args],
     ]
+    if model is not None:
+        command += ["--model-path", model]
     process = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
     try:

sglang/version.py CHANGED Viewed

	@@ -1 +1 @@
1	- __version__ = "0.4.5"
1	+ __version__ = "0.4.5.post1"

{sglang-0.4.5.dist-info → sglang-0.4.5.post1.dist-info}/METADATA RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.4
 Name: sglang
-Version: 0.4.5
+Version: 0.4.5.post1
 Summary: SGLang is yet another fast serving framework for large language models and vision language models.
 License:                                  Apache License
                                    Version 2.0, January 2004
@@ -239,20 +239,30 @@ Requires-Dist: python-multipart; extra == "runtime-common"
 Requires-Dist: pyzmq>=25.1.2; extra == "runtime-common"
 Requires-Dist: soundfile==0.13.1; extra == "runtime-common"
 Requires-Dist: torchao>=0.7.0; extra == "runtime-common"
-Requires-Dist: transformers==4.51.0; extra == "runtime-common"
+Requires-Dist: transformers==4.51.1; extra == "runtime-common"
 Requires-Dist: uvicorn; extra == "runtime-common"
 Requires-Dist: uvloop; extra == "runtime-common"
 Requires-Dist: compressed-tensors; extra == "runtime-common"
 Requires-Dist: xgrammar==0.1.17; extra == "runtime-common"
 Provides-Extra: srt
 Requires-Dist: sglang[runtime_common]; extra == "srt"
-Requires-Dist: sgl-kernel==0.0.8; extra == "srt"
+Requires-Dist: sgl-kernel==0.0.9.post1; extra == "srt"
 Requires-Dist: flashinfer_python==0.2.3; extra == "srt"
 Requires-Dist: torch==2.5.1; extra == "srt"
+Requires-Dist: torchvision==0.20.1; extra == "srt"
 Requires-Dist: cuda-python; extra == "srt"
 Requires-Dist: outlines<=0.1.11,>=0.0.44; extra == "srt"
 Requires-Dist: partial_json_parser; extra == "srt"
 Requires-Dist: einops; extra == "srt"
+Provides-Extra: blackwell
+Requires-Dist: sglang[runtime_common]; extra == "blackwell"
+Requires-Dist: sgl-kernel; extra == "blackwell"
+Requires-Dist: torch; extra == "blackwell"
+Requires-Dist: torchvision; extra == "blackwell"
+Requires-Dist: cuda-python; extra == "blackwell"
+Requires-Dist: outlines<=0.1.11,>=0.0.44; extra == "blackwell"
+Requires-Dist: partial_json_parser; extra == "blackwell"
+Requires-Dist: einops; extra == "blackwell"
 Provides-Extra: srt-hip
 Requires-Dist: sglang[runtime_common]; extra == "srt-hip"
 Requires-Dist: torch; extra == "srt-hip"
@@ -391,7 +401,7 @@ Learn more in the release blogs: [v0.2 blog](https://lmsys.org/blog/2024-07-25-s
 ## Adoption and Sponsorship
 The project has been deployed to large-scale production, generating trillions of tokens every day.
-It is supported by the following institutions: AMD, Atlas Cloud, Baseten, Cursor, DataCrunch, Etched, Hyperbolic, Iflytek, Jam & Tea Studios, LinkedIn, LMSYS, Meituan, Nebius, Novita AI, NVIDIA, RunPod, Stanford, UC Berkeley, UCLA, xAI, and 01.AI.
+It is supported by the following institutions: AMD, Atlas Cloud, Baseten, Cursor, DataCrunch, Etched, Hyperbolic, Iflytek, Jam & Tea Studios, LinkedIn, LMSYS, Meituan, Nebius, Novita AI, NVIDIA, Oracle, RunPod, Stanford, UC Berkeley, UCLA, xAI, and 01.AI.
 <img src="https://raw.githubusercontent.com/sgl-project/sgl-learning-materials/main/slides/adoption.png" alt="logo" width="800" margin="10px"></img>

sglang 0.4.5__py3-none-any.whl → 0.4.5.post1__py3-none-any.whl

sglang 0.4.5py3-none-any.whl → 0.4.5.post1py3-none-any.whl