PyPI - sglang - Versions diffs - 0.2.10__py3-none-any.whl → 0.2.12__py3-none-any.whl - Mend

sglang 0.2.10py3-none-any.whl → 0.2.12py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (89) hide show

sglang/__init__.py +8 -0
sglang/api.py +10 -2
sglang/bench_latency.py +151 -40
sglang/bench_serving.py +46 -22
sglang/check_env.py +24 -2
sglang/global_config.py +0 -1
sglang/lang/backend/base_backend.py +3 -1
sglang/lang/backend/openai.py +8 -3
sglang/lang/backend/runtime_endpoint.py +46 -29
sglang/lang/choices.py +164 -0
sglang/lang/compiler.py +2 -2
sglang/lang/interpreter.py +6 -13
sglang/lang/ir.py +14 -5
sglang/srt/constrained/base_tool_cache.py +1 -1
sglang/srt/constrained/fsm_cache.py +12 -2
sglang/srt/layers/activation.py +33 -0
sglang/srt/layers/{token_attention.py → decode_attention.py} +9 -5
sglang/srt/layers/extend_attention.py +6 -1
sglang/srt/layers/layernorm.py +65 -0
sglang/srt/layers/logits_processor.py +6 -1
sglang/srt/layers/pooler.py +50 -0
sglang/srt/layers/{context_flashattention_nopad.py → prefill_attention.py} +5 -0
sglang/srt/layers/radix_attention.py +4 -7
sglang/srt/managers/detokenizer_manager.py +31 -9
sglang/srt/managers/io_struct.py +63 -0
sglang/srt/managers/policy_scheduler.py +173 -25
sglang/srt/managers/schedule_batch.py +174 -380
sglang/srt/managers/tokenizer_manager.py +197 -112
sglang/srt/managers/tp_worker.py +299 -364
sglang/srt/mem_cache/{base_cache.py → base_prefix_cache.py} +9 -4
sglang/srt/mem_cache/chunk_cache.py +43 -20
sglang/srt/mem_cache/memory_pool.py +10 -15
sglang/srt/mem_cache/radix_cache.py +74 -40
sglang/srt/model_executor/cuda_graph_runner.py +27 -12
sglang/srt/model_executor/forward_batch_info.py +319 -0
sglang/srt/model_executor/model_runner.py +30 -47
sglang/srt/models/chatglm.py +1 -1
sglang/srt/models/commandr.py +1 -1
sglang/srt/models/dbrx.py +1 -1
sglang/srt/models/deepseek.py +1 -1
sglang/srt/models/deepseek_v2.py +1 -1
sglang/srt/models/gemma.py +1 -1
sglang/srt/models/gemma2.py +1 -2
sglang/srt/models/gpt_bigcode.py +1 -1
sglang/srt/models/grok.py +1 -1
sglang/srt/models/internlm2.py +3 -8
sglang/srt/models/llama2.py +5 -5
sglang/srt/models/llama_classification.py +1 -1
sglang/srt/models/llama_embedding.py +88 -0
sglang/srt/models/llava.py +1 -2
sglang/srt/models/llavavid.py +1 -2
sglang/srt/models/minicpm.py +1 -1
sglang/srt/models/mixtral.py +1 -1
sglang/srt/models/mixtral_quant.py +1 -1
sglang/srt/models/qwen.py +1 -1
sglang/srt/models/qwen2.py +1 -1
sglang/srt/models/qwen2_moe.py +1 -12
sglang/srt/models/stablelm.py +1 -1
sglang/srt/openai_api/adapter.py +189 -39
sglang/srt/openai_api/protocol.py +43 -1
sglang/srt/sampling/penaltylib/__init__.py +13 -0
sglang/srt/sampling/penaltylib/orchestrator.py +357 -0
sglang/srt/sampling/penaltylib/penalizers/frequency_penalty.py +80 -0
sglang/srt/sampling/penaltylib/penalizers/min_new_tokens.py +105 -0
sglang/srt/sampling/penaltylib/penalizers/presence_penalty.py +79 -0
sglang/srt/sampling/penaltylib/penalizers/repetition_penalty.py +83 -0
sglang/srt/sampling_params.py +31 -4
sglang/srt/server.py +93 -21
sglang/srt/server_args.py +30 -19
sglang/srt/utils.py +31 -13
sglang/test/run_eval.py +10 -1
sglang/test/runners.py +63 -63
sglang/test/simple_eval_humaneval.py +2 -8
sglang/test/simple_eval_mgsm.py +203 -0
sglang/test/srt/sampling/penaltylib/utils.py +337 -0
sglang/test/test_layernorm.py +60 -0
sglang/test/test_programs.py +4 -2
sglang/test/test_utils.py +21 -3
sglang/utils.py +0 -1
sglang/version.py +1 -1
{sglang-0.2.10.dist-info → sglang-0.2.12.dist-info}/METADATA +50 -31
sglang-0.2.12.dist-info/RECORD +112 -0
sglang/srt/layers/linear.py +0 -884
sglang/srt/layers/quantization/__init__.py +0 -64
sglang/srt/layers/quantization/fp8.py +0 -677
sglang-0.2.10.dist-info/RECORD +0 -100
{sglang-0.2.10.dist-info → sglang-0.2.12.dist-info}/LICENSE +0 -0
{sglang-0.2.10.dist-info → sglang-0.2.12.dist-info}/WHEEL +0 -0
{sglang-0.2.10.dist-info → sglang-0.2.12.dist-info}/top_level.txt +0 -0

sglang/test/srt/sampling/penaltylib/utils.py ADDED Viewed

@@ -0,0 +1,337 @@
+import dataclasses
+import enum
+import typing
+import unittest
+import torch
+from sglang.srt.sampling.penaltylib.orchestrator import (
+    BatchedPenalizerOrchestrator,
+    _BatchedPenalizer,
+    _BatchLike,
+)
+@dataclasses.dataclass
+class MockSamplingParams:
+    frequency_penalty: float = 0.0
+    min_new_tokens: int = 0
+    stop_token_ids: typing.List[int] = None
+    presence_penalty: float = 0.0
+    repetition_penalty: float = 1.0
+@dataclasses.dataclass
+class MockTokenizer:
+    eos_token_id: int
+@dataclasses.dataclass
+class MockReq:
+    origin_input_ids: typing.List[int]
+    sampling_params: MockSamplingParams
+    tokenizer: MockTokenizer
+class StepType(enum.Enum):
+    INPUT = "input"
+    OUTPUT = "output"
+@dataclasses.dataclass
+class Step:
+    type: StepType
+    token_ids: typing.List[int]
+    expected_tensors: typing.Dict[str, torch.Tensor]
+    # assume initial logits are all 1
+    expected_logits: torch.Tensor
+@dataclasses.dataclass
+class Subject:
+    sampling_params: MockSamplingParams
+    # first step must be input, which will be converted to Req
+    steps: typing.List[Step]
+    eos_token_id: int = -1
+    def __post_init__(self):
+        if self.steps[0].type != StepType.INPUT:
+            raise ValueError("First step must be input")
+        # each steps should have the same expected_tensors.keys()
+        for i in range(1, len(self.steps)):
+            if self.tensor_keys(i) != self.tensor_keys():
+                raise ValueError(
+                    f"Expected tensors keys must be the same for all steps. Got {self.steps[i].expected_tensors.keys()} for key={i} and {self.steps[0].expected_tensors.keys()}"
+                )
+    def tensor_keys(self, i: int = 0) -> typing.Set[str]:
+        return set(self.steps[i].expected_tensors.keys())
+    def to_req(self) -> MockReq:
+        return MockReq(
+            origin_input_ids=self.steps[0].token_ids,
+            sampling_params=self.sampling_params,
+            tokenizer=MockTokenizer(eos_token_id=self.eos_token_id),
+        )
+@dataclasses.dataclass
+class Case:
+    enabled: bool
+    test_subjects: typing.List[Subject]
+    def __post_init__(self):
+        # each test_subjects.steps should have the same expected_tensors.keys()
+        for i in range(1, len(self.test_subjects)):
+            if self.tensor_keys(i) != self.tensor_keys():
+                raise ValueError(
+                    f"Expected tensors keys must be the same for all test_subjects. Got {self.test_subjects[i].tensor_keys()} for key={i} and {self.test_subjects[0].tensor_keys()}"
+                )
+    def tensor_keys(self, i: int = 0) -> typing.List[str]:
+        return set(self.test_subjects[i].tensor_keys())
+class BaseBatchedPenalizerTest(unittest.TestCase):
+    Penalizer: typing.Type[_BatchedPenalizer]
+    device = "cuda"
+    vocab_size = 5
+    enabled: Subject = None
+    disabled: Subject = None
+    def setUp(self):
+        if self.__class__ == BaseBatchedPenalizerTest:
+            self.skipTest("Base class for penalizer tests")
+        self.create_test_subjects()
+        self.create_test_cases()
+    def tensor(self, data, **kwargs) -> torch.Tensor:
+        """
+        Shortcut to create a tensor with device=self.device.
+        """
+        return torch.tensor(data, **kwargs, device=self.device)
+    def create_test_subjects(self) -> typing.List[Subject]:
+        raise NotImplementedError()
+    def create_test_cases(self):
+        self.test_cases = [
+            Case(enabled=True, test_subjects=[self.enabled]),
+            Case(enabled=False, test_subjects=[self.disabled]),
+            Case(enabled=True, test_subjects=[self.enabled, self.disabled]),
+        ]
+    def _create_penalizer(
+        self, case: Case
+    ) -> typing.Tuple[BatchedPenalizerOrchestrator, _BatchedPenalizer]:
+        orchestrator = BatchedPenalizerOrchestrator(
+            vocab_size=self.vocab_size,
+            batch=_BatchLike(reqs=[subject.to_req() for subject in case.test_subjects]),
+            device=self.device,
+            Penalizers={self.Penalizer},
+        )
+        return orchestrator, orchestrator.penalizers[self.Penalizer]
+    def test_is_required(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                _, penalizer = self._create_penalizer(case)
+                self.assertEqual(case.enabled, penalizer.is_required())
+    def test_prepare(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                orchestrator, penalizer = self._create_penalizer(case)
+                self.assertEqual(case.enabled, penalizer.is_prepared())
+                if case.enabled:
+                    for key, tensor in {
+                        key: torch.cat(
+                            tensors=[
+                                subject.steps[0].expected_tensors[key]
+                                for subject in case.test_subjects
+                            ],
+                        )
+                        for key in case.tensor_keys()
+                    }.items():
+                        torch.testing.assert_close(
+                            actual=getattr(penalizer, key),
+                            expected=tensor,
+                            msg=f"key={key}\nactual={getattr(penalizer, key)}\nexpected={tensor}",
+                        )
+                actual = orchestrator.apply(
+                    torch.ones(
+                        size=(len(case.test_subjects), self.vocab_size),
+                        dtype=torch.float32,
+                        device=self.device,
+                    )
+                )
+                expected = torch.cat(
+                    tensors=[
+                        subject.steps[0].expected_logits
+                        for subject in case.test_subjects
+                    ],
+                )
+                torch.testing.assert_close(
+                    actual=actual,
+                    expected=expected,
+                    msg=f"logits\nactual={actual}\nexpected={expected}",
+                )
+    def test_teardown(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                _, penalizer = self._create_penalizer(case)
+                penalizer.teardown()
+                for key in case.test_subjects[0].steps[0].expected_tensors.keys():
+                    self.assertIsNone(getattr(penalizer, key, None))
+    def test_filter(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                orchestrator, penalizer = self._create_penalizer(case)
+                indices_to_keep = [0]
+                orchestrator.filter(indices_to_keep=indices_to_keep)
+                filtered_subjects = [case.test_subjects[i] for i in indices_to_keep]
+                if penalizer.is_required():
+                    self.assertTrue(penalizer.is_prepared())
+                    for key, tensor in {
+                        key: torch.cat(
+                            tensors=[
+                                subject.steps[0].expected_tensors[key]
+                                for subject in filtered_subjects
+                            ],
+                        )
+                        for key in case.tensor_keys()
+                    }.items():
+                        torch.testing.assert_close(
+                            actual=getattr(penalizer, key),
+                            expected=tensor,
+                            msg=f"key={key}\nactual={getattr(penalizer, key)}\nexpected={tensor}",
+                        )
+                actual_logits = orchestrator.apply(
+                    torch.ones(
+                        size=(len(filtered_subjects), self.vocab_size),
+                        dtype=torch.float32,
+                        device=self.device,
+                    )
+                )
+                filtered_expected_logits = torch.cat(
+                    tensors=[
+                        subject.steps[0].expected_logits
+                        for subject in filtered_subjects
+                    ],
+                )
+                torch.testing.assert_close(
+                    actual=actual_logits,
+                    expected=filtered_expected_logits,
+                    msg=f"logits\nactual={actual_logits}\nexpected={filtered_expected_logits}",
+                )
+    def test_merge_enabled_with_disabled(self):
+        enabled_test_case = self.test_cases[0]
+        disabled_test_case = self.test_cases[1]
+        orchestrator, penalizer = self._create_penalizer(enabled_test_case)
+        theirs, _ = self._create_penalizer(disabled_test_case)
+        orchestrator.merge(theirs)
+        for key, tensor in {
+            key: torch.cat(
+                tensors=[
+                    enabled_test_case.test_subjects[0].steps[0].expected_tensors[key],
+                    disabled_test_case.test_subjects[0].steps[0].expected_tensors[key],
+                ],
+            )
+            for key in enabled_test_case.tensor_keys()
+        }.items():
+            torch.testing.assert_close(
+                actual=getattr(penalizer, key),
+                expected=tensor,
+                msg=f"key={key}\nactual={getattr(penalizer, key)}\nexpected={tensor}",
+            )
+    def test_cumulate_apply_repeat(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                orchestrator, penalizer = self._create_penalizer(case)
+                max_step = max(len(subject.steps) for subject in case.test_subjects)
+                for i in range(1, max_step):
+                    orchestrator.filter(
+                        indices_to_keep=[
+                            j
+                            for j, subject in enumerate(case.test_subjects)
+                            if i < len(subject.steps)
+                        ]
+                    )
+                    filtered_subjects = [
+                        subject
+                        for subject in case.test_subjects
+                        if i < len(subject.steps)
+                    ]
+                    inputs: typing.List[typing.List[int]] = []
+                    outputs: typing.List[typing.List[int]] = []
+                    for subject in filtered_subjects:
+                        step = subject.steps[i]
+                        if step.type == StepType.INPUT:
+                            inputs.append(step.token_ids)
+                            outputs.append([])
+                        else:
+                            inputs.append([])
+                            outputs.append(step.token_ids)
+                    if any(inputs):
+                        orchestrator.cumulate_input_tokens(inputs)
+                    if any(outputs):
+                        orchestrator.cumulate_output_tokens(outputs)
+                    if penalizer.is_required():
+                        self.assertTrue(penalizer.is_prepared())
+                        for key, tensor in {
+                            key: torch.cat(
+                                tensors=[
+                                    subject.steps[i].expected_tensors[key]
+                                    for subject in filtered_subjects
+                                ],
+                            )
+                            for key in case.tensor_keys()
+                        }.items():
+                            torch.testing.assert_close(
+                                actual=getattr(penalizer, key),
+                                expected=tensor,
+                                msg=f"key={key}\nactual={getattr(penalizer, key)}\nexpected={tensor}",
+                            )
+                    actual_logits = orchestrator.apply(
+                        torch.ones(
+                            size=(len(filtered_subjects), self.vocab_size),
+                            dtype=torch.float32,
+                            device=self.device,
+                        )
+                    )
+                    filtered_expected_logits = torch.cat(
+                        tensors=[
+                            subject.steps[i].expected_logits
+                            for subject in filtered_subjects
+                        ],
+                    )
+                    torch.testing.assert_close(
+                        actual=actual_logits,
+                        expected=filtered_expected_logits,
+                        msg=f"logits\nactual={actual_logits}\nexpected={filtered_expected_logits}",
+                    )

sglang/test/test_layernorm.py ADDED Viewed

@@ -0,0 +1,60 @@
+import itertools
+import unittest
+import torch
+from sglang.srt.layers.layernorm import RMSNorm
+class TestRMSNorm(unittest.TestCase):
+    DTYPES = [torch.half, torch.bfloat16]
+    NUM_TOKENS = [7, 83, 4096]
+    HIDDEN_SIZES = [768, 769, 770, 771, 5120, 5124, 5125, 5126, 8192, 8199]
+    ADD_RESIDUAL = [False, True]
+    SEEDS = [0]
+    @classmethod
+    def setUpClass(cls):
+        if not torch.cuda.is_available():
+            raise unittest.SkipTest("CUDA is not available")
+        torch.set_default_device("cuda")
+    def _run_rms_norm_test(self, num_tokens, hidden_size, add_residual, dtype, seed):
+        torch.manual_seed(seed)
+        layer = RMSNorm(hidden_size).to(dtype=dtype)
+        layer.weight.data.normal_(mean=1.0, std=0.1)
+        scale = 1 / (2 * hidden_size)
+        x = torch.randn(num_tokens, hidden_size, dtype=dtype) * scale
+        residual = torch.randn_like(x) * scale if add_residual else None
+        with torch.inference_mode():
+            ref_out = layer.forward_native(x, residual)
+            out = layer(x, residual)
+        if add_residual:
+            self.assertTrue(torch.allclose(out[0], ref_out[0], atol=1e-2, rtol=1e-2))
+            self.assertTrue(torch.allclose(out[1], ref_out[1], atol=1e-2, rtol=1e-2))
+        else:
+            self.assertTrue(torch.allclose(out, ref_out, atol=1e-2, rtol=1e-2))
+    def test_rms_norm(self):
+        for params in itertools.product(
+            self.NUM_TOKENS,
+            self.HIDDEN_SIZES,
+            self.ADD_RESIDUAL,
+            self.DTYPES,
+            self.SEEDS,
+        ):
+            with self.subTest(
+                num_tokens=params[0],
+                hidden_size=params[1],
+                add_residual=params[2],
+                dtype=params[3],
+                seed=params[4],
+            ):
+                self._run_rms_norm_test(*params)
+if __name__ == "__main__":
+    unittest.main(verbosity=2)

sglang/test/test_programs.py CHANGED Viewed

@@ -149,7 +149,7 @@ def test_decode_json():
     assert isinstance(js_obj["population"], int)
-def test_expert_answer():
+def test_expert_answer(check_answer=True):
     @sgl.function
     def expert_answer(s, question):
         s += "Question: " + question + "\n"
@@ -167,7 +167,9 @@ def test_expert_answer():
         )
     ret = expert_answer.run(question="What is the capital of France?", temperature=0.1)
-    assert "paris" in ret.text().lower()
+    if check_answer:
+        assert "paris" in ret.text().lower(), f"Answer: {ret.text()}"
 def test_tool_use():

sglang/test/test_utils.py CHANGED Viewed

@@ -12,13 +12,16 @@ from typing import Callable, List, Optional
 import numpy as np
 import requests
+import torch
+import torch.nn.functional as F
 from sglang.global_config import global_config
 from sglang.lang.backend.openai import OpenAI
 from sglang.lang.backend.runtime_endpoint import RuntimeEndpoint
 from sglang.utils import get_exception_traceback
-MODEL_NAME_FOR_TEST = "meta-llama/Meta-Llama-3.1-8B-Instruct"
+DEFAULT_MODEL_NAME_FOR_TEST = "meta-llama/Meta-Llama-3.1-8B-Instruct"
+DEFAULT_URL_FOR_TEST = "http://127.0.0.1:8157"
 def call_generate_lightllm(prompt, temperature, max_tokens, stop=None, url=None):
@@ -396,6 +399,8 @@ def popen_launch_server(
     timeout: float,
     api_key: Optional[str] = None,
     other_args: tuple = (),
+    env: Optional[dict] = None,
+    return_stdout_stderr: bool = False,
 ):
     _, host, port = base_url.split(":")
     host = host[2:]
@@ -415,7 +420,16 @@ def popen_launch_server(
     if api_key:
         command += ["--api-key", api_key]
-    process = subprocess.Popen(command, stdout=None, stderr=None)
+    if return_stdout_stderr:
+        process = subprocess.Popen(
+            command,
+            stdout=subprocess.PIPE,
+            stderr=subprocess.PIPE,
+            env=env,
+            text=True,
+        )
+    else:
+        process = subprocess.Popen(command, stdout=None, stderr=None, env=env)
     start_time = time.time()
     while time.time() - start_time < timeout:
@@ -482,7 +496,7 @@ def run_unittest_files(files: List[str], timeout_per_file: float):
             p.terminate()
             time.sleep(5)
             print(
-                "\nTimeout after {timeout_per_file} seconds when running {filename}\n"
+                f"\nTimeout after {timeout_per_file} seconds when running {filename}\n"
             )
             return False
@@ -492,3 +506,7 @@ def run_unittest_files(files: List[str], timeout_per_file: float):
         print(f"Fail. Time elapsed: {time.time() - tic:.2f}s")
     return 0 if success else -1
+def get_similarities(vec1, vec2):
+    return F.cosine_similarity(torch.tensor(vec1), torch.tensor(vec2), dim=0)

sglang/utils.py CHANGED Viewed

@@ -6,7 +6,6 @@ import json
 import logging
 import signal
 import sys
-import threading
 import traceback
 import urllib.request
 from concurrent.futures import ThreadPoolExecutor

sglang/version.py CHANGED Viewed

	@@ -1 +1 @@
1	- __version__ = "0.2.10"
1	+ __version__ = "0.2.12"

{sglang-0.2.10.dist-info → sglang-0.2.12.dist-info}/METADATA RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.1
 Name: sglang
-Version: 0.2.10
+Version: 0.2.12
 Summary: SGLang is yet another fast serving framework for large language models and vision language models.
 License: Apache License
                                    Version 2.0, January 2004
@@ -221,6 +221,9 @@ Requires-Dist: sglang[anthropic]; extra == "all"
 Requires-Dist: sglang[litellm]; extra == "all"
 Provides-Extra: anthropic
 Requires-Dist: anthropic>=0.20.0; extra == "anthropic"
+Provides-Extra: dev
+Requires-Dist: sglang[all]; extra == "dev"
+Requires-Dist: sglang[test]; extra == "dev"
 Provides-Extra: litellm
 Requires-Dist: litellm>=1.0.0; extra == "litellm"
 Provides-Extra: openai
@@ -232,7 +235,6 @@ Requires-Dist: fastapi; extra == "srt"
 Requires-Dist: hf-transfer; extra == "srt"
 Requires-Dist: huggingface-hub; extra == "srt"
 Requires-Dist: interegular; extra == "srt"
-Requires-Dist: jsonlines; extra == "srt"
 Requires-Dist: packaging; extra == "srt"
 Requires-Dist: pillow; extra == "srt"
 Requires-Dist: psutil; extra == "srt"
@@ -242,8 +244,12 @@ Requires-Dist: torch; extra == "srt"
 Requires-Dist: uvicorn; extra == "srt"
 Requires-Dist: uvloop; extra == "srt"
 Requires-Dist: zmq; extra == "srt"
-Requires-Dist: vllm==0.5.3.post1; extra == "srt"
+Requires-Dist: vllm==0.5.4; extra == "srt"
 Requires-Dist: outlines>=0.0.44; extra == "srt"
+Provides-Extra: test
+Requires-Dist: jsonlines; extra == "test"
+Requires-Dist: matplotlib; extra == "test"
+Requires-Dist: pandas; extra == "test"
 <div align="center">
 <img src="https://raw.githubusercontent.com/sgl-project/sglang/main/assets/logo.png" alt="logo" width="400"></img>
@@ -296,20 +302,20 @@ pip install --upgrade pip
 pip install "sglang[all]"
 # Install FlashInfer CUDA kernels
-pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3/
+pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.4/
 ```
 ### Method 2: From source
 ```
 # Use the last release branch
-git clone -b v0.2.10 https://github.com/sgl-project/sglang.git
+git clone -b v0.2.12 https://github.com/sgl-project/sglang.git
 cd sglang
 pip install --upgrade pip
 pip install -e "python[all]"
 # Install FlashInfer CUDA kernels
-pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.3/
+pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.4/
 ```
 ### Method 3: Using docker
@@ -383,22 +389,26 @@ response = client.chat.completions.create(
 print(response)
 ```
-It supports streaming, vision, and most features of the Chat/Completions/Models endpoints specified by the [OpenAI API Reference](https://platform.openai.com/docs/api-reference/).
+It supports streaming, vision, and most features of the Chat/Completions/Models/Batch endpoints specified by the [OpenAI API Reference](https://platform.openai.com/docs/api-reference/).
 ### Additional Server Arguments
-- Add `--tp 2` to enable tensor parallelism. If it indicates `peer access is not supported between these two devices`, add `--enable-p2p-check` option.
+- Add `--tp 2` to enable multi-GPU tensor parallelism. If it reports the error "peer access is not supported between these two devices", add `--enable-p2p-check` to the server launch command.
 ```
 python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --port 30000 --tp 2
 ```
-- Add `--dp 2` to enable data parallelism. It can also be used together with tp. Data parallelism is better for throughput if there is enough memory.
+- Add `--dp 2` to enable multi-GPU data parallelism. It can also be used together with tensor parallelism. Data parallelism is better for throughput if there is enough memory.
 ```
 python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --port 30000 --dp 2 --tp 2
 ```
-- If you see out-of-memory errors during serving, please try to reduce the memory usage of the KV cache pool by setting a smaller value of `--mem-fraction-static`. The default value is `0.9`
+- If you see out-of-memory errors during serving, try to reduce the memory usage of the KV cache pool by setting a smaller value of `--mem-fraction-static`. The default value is `0.9`.
 ```
 python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --port 30000 --mem-fraction-static 0.7
 ```
 - See [hyperparameter_tuning.md](docs/en/hyperparameter_tuning.md) on tuning hyperparameters for better performance.
+- If you see out-of-memory errors during prefill for long prompts, try to set a smaller chunked prefill size.
+```
+python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --port 30000 --chunked-prefill-size 4096
+```
 - Add `--nnodes 2` to run tensor parallelism on multiple nodes. If you have two nodes with two GPUs on each node and want to run TP=4, let `sgl-dev-0` be the hostname of the first node and `50000` be an available port.
 ```
 # Node 0
@@ -408,29 +418,13 @@ python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct
 python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --tp 4 --nccl-init sgl-dev-0:50000 --nnodes 2 --node-rank 1
 ```
 - If the model does not have a template in the Hugging Face tokenizer, you can specify a [custom chat template](docs/en/custom_chat_template.md).
-- To enable fp8 quantization, you can add `--quantization fp8` on a fp16 checkpoint or directly load a fp8 checkpoint without specifying any arguments.
 - To enable experimental torch.compile support, you can add `--enable-torch-compile`. It accelerates small models on small batch sizes.
-### Run Llama 3.1 405B
-```bash
-## Run 405B (fp8) on a single node
-python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-405B-Instruct-FP8 --tp 8
-## Run 405B (fp16) on two nodes
-# replace the `172.16.4.52:20000` with your own first node ip address and port, disable CUDA Graph temporarily
-# on the first node
-GLOO_SOCKET_IFNAME=eth0 python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-405B-Instruct --tp 16 --nccl-init-addr 172.16.4.52:20000 --nnodes 2 --node-rank 0 --disable-cuda-graph --mem-frac 0.75
-# on the second
-GLOO_SOCKET_IFNAME=eth0 python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-405B-Instruct --tp 16 --nccl-init-addr 172.16.4.52:20000 --nnodes 2 --node-rank 1 --disable-cuda-graph --mem-frac 0.75
-```
+- To enable fp8 quantization, you can add `--quantization fp8` on a fp16 checkpoint or directly load a fp8 checkpoint without specifying any arguments.
 ### Supported Models
 - Llama / Llama 2 / Llama 3 / Llama 3.1
-- Mistral / Mixtral
+- Mistral / Mixtral / Mistral NeMo
 - Gemma / Gemma 2
 - Qwen / Qwen 2 / Qwen 2 MoE
 - DeepSeek / DeepSeek 2
@@ -448,10 +442,35 @@ GLOO_SOCKET_IFNAME=eth0 python3 -m sglang.launch_server --model-path meta-llama/
 - Grok
 - ChatGLM
 - InternLM 2
-- Mistral NeMo
 Instructions for supporting a new model are [here](https://github.com/sgl-project/sglang/blob/main/docs/en/model_support.md).
+#### Use Models From ModelScope
+To use model from [ModelScope](https://www.modelscope.cn), setting environment variable SGLANG_USE_MODELSCOPE.
+```
+export SGLANG_USE_MODELSCOPE=true
+```
+Launch [Qwen2-7B-Instruct](https://www.modelscope.cn/models/qwen/qwen2-7b-instruct) Server
+```
+SGLANG_USE_MODELSCOPE=true python -m sglang.launch_server --model-path qwen/Qwen2-7B-Instruct --port 30000
+```
+#### Run Llama 3.1 405B
+```bash
+## Run 405B (fp8) on a single node
+python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-405B-Instruct-FP8 --tp 8
+## Run 405B (fp16) on two nodes
+# replace the `172.16.4.52:20000` with your own first node ip address and port, disable CUDA Graph temporarily
+# on the first node
+GLOO_SOCKET_IFNAME=eth0 python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-405B-Instruct --tp 16 --nccl-init-addr 172.16.4.52:20000 --nnodes 2 --node-rank 0 --disable-cuda-graph --mem-frac 0.75
+# on the second
+GLOO_SOCKET_IFNAME=eth0 python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-405B-Instruct --tp 16 --nccl-init-addr 172.16.4.52:20000 --nnodes 2 --node-rank 1 --disable-cuda-graph --mem-frac 0.75
+```
 ### Benchmark Performance
 - Benchmark a single static batch by running the following command without launching a server. The arguments are the same as for `launch_server.py`. Note that this is not a dynamic batching server, so it may run out of memory for a batch size that a real server can handle. A real server truncates the prefill into several batches, while this unit test does not. For accurate large batch testing, consider using `sglang.bench_serving`.
@@ -464,7 +483,7 @@ Instructions for supporting a new model are [here](https://github.com/sgl-projec
   ```
 ## Frontend: Structured Generation Language (SGLang)
-The frontend language can be used with local models or API models.
+The frontend language can be used with local models or API models. It is an alternative to the OpenAI API. You may found it easier to use for complex prompting workflow.
 ### Quick Start
 The example below shows how to use sglang to answer a mulit-turn question.

sglang 0.2.10__py3-none-any.whl → 0.2.12__py3-none-any.whl

sglang 0.2.10py3-none-any.whl → 0.2.12py3-none-any.whl