PyPI - sglang - Versions diffs - 0.2.11__py3-none-any.whl → 0.2.13__py3-none-any.whl - Mend

sglang 0.2.11py3-none-any.whl → 0.2.13py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (85) hide show

sglang/api.py +7 -1
sglang/bench_latency.py +9 -6
sglang/bench_serving.py +46 -22
sglang/global_config.py +1 -1
sglang/lang/backend/runtime_endpoint.py +60 -49
sglang/lang/compiler.py +2 -2
sglang/lang/interpreter.py +4 -2
sglang/lang/ir.py +16 -7
sglang/srt/constrained/base_tool_cache.py +1 -1
sglang/srt/constrained/fsm_cache.py +12 -2
sglang/srt/constrained/jump_forward.py +13 -2
sglang/srt/layers/activation.py +32 -0
sglang/srt/layers/{token_attention.py → decode_attention.py} +9 -5
sglang/srt/layers/extend_attention.py +9 -2
sglang/srt/layers/fused_moe/__init__.py +1 -0
sglang/srt/layers/{fused_moe.py → fused_moe/fused_moe.py} +165 -108
sglang/srt/layers/fused_moe/layer.py +587 -0
sglang/srt/layers/layernorm.py +65 -0
sglang/srt/layers/logits_processor.py +7 -2
sglang/srt/layers/pooler.py +50 -0
sglang/srt/layers/{context_flashattention_nopad.py → prefill_attention.py} +5 -0
sglang/srt/layers/radix_attention.py +40 -16
sglang/srt/managers/detokenizer_manager.py +31 -9
sglang/srt/managers/io_struct.py +63 -0
sglang/srt/managers/policy_scheduler.py +173 -25
sglang/srt/managers/schedule_batch.py +115 -97
sglang/srt/managers/tokenizer_manager.py +194 -112
sglang/srt/managers/tp_worker.py +290 -359
sglang/srt/mem_cache/{base_cache.py → base_prefix_cache.py} +9 -4
sglang/srt/mem_cache/chunk_cache.py +43 -20
sglang/srt/mem_cache/memory_pool.py +2 -2
sglang/srt/mem_cache/radix_cache.py +74 -40
sglang/srt/model_executor/cuda_graph_runner.py +71 -25
sglang/srt/model_executor/forward_batch_info.py +293 -156
sglang/srt/model_executor/model_runner.py +77 -57
sglang/srt/models/chatglm.py +2 -2
sglang/srt/models/commandr.py +1 -1
sglang/srt/models/deepseek.py +2 -2
sglang/srt/models/deepseek_v2.py +7 -6
sglang/srt/models/gemma.py +1 -1
sglang/srt/models/gemma2.py +11 -6
sglang/srt/models/grok.py +50 -396
sglang/srt/models/internlm2.py +2 -7
sglang/srt/models/llama2.py +4 -4
sglang/srt/models/llama_embedding.py +88 -0
sglang/srt/models/minicpm.py +2 -2
sglang/srt/models/mixtral.py +56 -254
sglang/srt/models/mixtral_quant.py +1 -4
sglang/srt/models/qwen.py +2 -2
sglang/srt/models/qwen2.py +2 -2
sglang/srt/models/qwen2_moe.py +2 -13
sglang/srt/models/stablelm.py +1 -1
sglang/srt/openai_api/adapter.py +187 -48
sglang/srt/openai_api/protocol.py +37 -1
sglang/srt/sampling/penaltylib/__init__.py +13 -0
sglang/srt/sampling/penaltylib/orchestrator.py +357 -0
sglang/srt/sampling/penaltylib/penalizers/frequency_penalty.py +80 -0
sglang/srt/sampling/penaltylib/penalizers/min_new_tokens.py +105 -0
sglang/srt/sampling/penaltylib/penalizers/presence_penalty.py +79 -0
sglang/srt/sampling/penaltylib/penalizers/repetition_penalty.py +83 -0
sglang/srt/sampling_params.py +31 -8
sglang/srt/server.py +91 -29
sglang/srt/server_args.py +32 -19
sglang/srt/utils.py +32 -15
sglang/test/run_eval.py +10 -1
sglang/test/runners.py +81 -73
sglang/test/simple_eval_humaneval.py +2 -8
sglang/test/simple_eval_mgsm.py +203 -0
sglang/test/srt/sampling/penaltylib/utils.py +337 -0
sglang/test/test_layernorm.py +60 -0
sglang/test/test_programs.py +36 -7
sglang/test/test_utils.py +24 -2
sglang/utils.py +0 -1
sglang/version.py +1 -1
{sglang-0.2.11.dist-info → sglang-0.2.13.dist-info}/METADATA +33 -16
sglang-0.2.13.dist-info/RECORD +112 -0
{sglang-0.2.11.dist-info → sglang-0.2.13.dist-info}/WHEEL +1 -1
sglang/srt/layers/linear.py +0 -884
sglang/srt/layers/quantization/__init__.py +0 -64
sglang/srt/layers/quantization/fp8.py +0 -677
sglang/srt/model_loader/model_loader.py +0 -292
sglang/srt/model_loader/utils.py +0 -275
sglang-0.2.11.dist-info/RECORD +0 -102
{sglang-0.2.11.dist-info → sglang-0.2.13.dist-info}/LICENSE +0 -0
{sglang-0.2.11.dist-info → sglang-0.2.13.dist-info}/top_level.txt +0 -0

sglang/test/srt/sampling/penaltylib/utils.py ADDED Viewed

@@ -0,0 +1,337 @@
+import dataclasses
+import enum
+import typing
+import unittest
+import torch
+from sglang.srt.sampling.penaltylib.orchestrator import (
+    BatchedPenalizerOrchestrator,
+    _BatchedPenalizer,
+    _BatchLike,
+)
+@dataclasses.dataclass
+class MockSamplingParams:
+    frequency_penalty: float = 0.0
+    min_new_tokens: int = 0
+    stop_token_ids: typing.List[int] = None
+    presence_penalty: float = 0.0
+    repetition_penalty: float = 1.0
+@dataclasses.dataclass
+class MockTokenizer:
+    eos_token_id: int
+@dataclasses.dataclass
+class MockReq:
+    origin_input_ids: typing.List[int]
+    sampling_params: MockSamplingParams
+    tokenizer: MockTokenizer
+class StepType(enum.Enum):
+    INPUT = "input"
+    OUTPUT = "output"
+@dataclasses.dataclass
+class Step:
+    type: StepType
+    token_ids: typing.List[int]
+    expected_tensors: typing.Dict[str, torch.Tensor]
+    # assume initial logits are all 1
+    expected_logits: torch.Tensor
+@dataclasses.dataclass
+class Subject:
+    sampling_params: MockSamplingParams
+    # first step must be input, which will be converted to Req
+    steps: typing.List[Step]
+    eos_token_id: int = -1
+    def __post_init__(self):
+        if self.steps[0].type != StepType.INPUT:
+            raise ValueError("First step must be input")
+        # each steps should have the same expected_tensors.keys()
+        for i in range(1, len(self.steps)):
+            if self.tensor_keys(i) != self.tensor_keys():
+                raise ValueError(
+                    f"Expected tensors keys must be the same for all steps. Got {self.steps[i].expected_tensors.keys()} for key={i} and {self.steps[0].expected_tensors.keys()}"
+                )
+    def tensor_keys(self, i: int = 0) -> typing.Set[str]:
+        return set(self.steps[i].expected_tensors.keys())
+    def to_req(self) -> MockReq:
+        return MockReq(
+            origin_input_ids=self.steps[0].token_ids,
+            sampling_params=self.sampling_params,
+            tokenizer=MockTokenizer(eos_token_id=self.eos_token_id),
+        )
+@dataclasses.dataclass
+class Case:
+    enabled: bool
+    test_subjects: typing.List[Subject]
+    def __post_init__(self):
+        # each test_subjects.steps should have the same expected_tensors.keys()
+        for i in range(1, len(self.test_subjects)):
+            if self.tensor_keys(i) != self.tensor_keys():
+                raise ValueError(
+                    f"Expected tensors keys must be the same for all test_subjects. Got {self.test_subjects[i].tensor_keys()} for key={i} and {self.test_subjects[0].tensor_keys()}"
+                )
+    def tensor_keys(self, i: int = 0) -> typing.List[str]:
+        return set(self.test_subjects[i].tensor_keys())
+class BaseBatchedPenalizerTest(unittest.TestCase):
+    Penalizer: typing.Type[_BatchedPenalizer]
+    device = "cuda"
+    vocab_size = 5
+    enabled: Subject = None
+    disabled: Subject = None
+    def setUp(self):
+        if self.__class__ == BaseBatchedPenalizerTest:
+            self.skipTest("Base class for penalizer tests")
+        self.create_test_subjects()
+        self.create_test_cases()
+    def tensor(self, data, **kwargs) -> torch.Tensor:
+        """
+        Shortcut to create a tensor with device=self.device.
+        """
+        return torch.tensor(data, **kwargs, device=self.device)
+    def create_test_subjects(self) -> typing.List[Subject]:
+        raise NotImplementedError()
+    def create_test_cases(self):
+        self.test_cases = [
+            Case(enabled=True, test_subjects=[self.enabled]),
+            Case(enabled=False, test_subjects=[self.disabled]),
+            Case(enabled=True, test_subjects=[self.enabled, self.disabled]),
+        ]
+    def _create_penalizer(
+        self, case: Case
+    ) -> typing.Tuple[BatchedPenalizerOrchestrator, _BatchedPenalizer]:
+        orchestrator = BatchedPenalizerOrchestrator(
+            vocab_size=self.vocab_size,
+            batch=_BatchLike(reqs=[subject.to_req() for subject in case.test_subjects]),
+            device=self.device,
+            Penalizers={self.Penalizer},
+        )
+        return orchestrator, orchestrator.penalizers[self.Penalizer]
+    def test_is_required(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                _, penalizer = self._create_penalizer(case)
+                self.assertEqual(case.enabled, penalizer.is_required())
+    def test_prepare(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                orchestrator, penalizer = self._create_penalizer(case)
+                self.assertEqual(case.enabled, penalizer.is_prepared())
+                if case.enabled:
+                    for key, tensor in {
+                        key: torch.cat(
+                            tensors=[
+                                subject.steps[0].expected_tensors[key]
+                                for subject in case.test_subjects
+                            ],
+                        )
+                        for key in case.tensor_keys()
+                    }.items():
+                        torch.testing.assert_close(
+                            actual=getattr(penalizer, key),
+                            expected=tensor,
+                            msg=f"key={key}\nactual={getattr(penalizer, key)}\nexpected={tensor}",
+                        )
+                actual = orchestrator.apply(
+                    torch.ones(
+                        size=(len(case.test_subjects), self.vocab_size),
+                        dtype=torch.float32,
+                        device=self.device,
+                    )
+                )
+                expected = torch.cat(
+                    tensors=[
+                        subject.steps[0].expected_logits
+                        for subject in case.test_subjects
+                    ],
+                )
+                torch.testing.assert_close(
+                    actual=actual,
+                    expected=expected,
+                    msg=f"logits\nactual={actual}\nexpected={expected}",
+                )
+    def test_teardown(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                _, penalizer = self._create_penalizer(case)
+                penalizer.teardown()
+                for key in case.test_subjects[0].steps[0].expected_tensors.keys():
+                    self.assertIsNone(getattr(penalizer, key, None))
+    def test_filter(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                orchestrator, penalizer = self._create_penalizer(case)
+                indices_to_keep = [0]
+                orchestrator.filter(indices_to_keep=indices_to_keep)
+                filtered_subjects = [case.test_subjects[i] for i in indices_to_keep]
+                if penalizer.is_required():
+                    self.assertTrue(penalizer.is_prepared())
+                    for key, tensor in {
+                        key: torch.cat(
+                            tensors=[
+                                subject.steps[0].expected_tensors[key]
+                                for subject in filtered_subjects
+                            ],
+                        )
+                        for key in case.tensor_keys()
+                    }.items():
+                        torch.testing.assert_close(
+                            actual=getattr(penalizer, key),
+                            expected=tensor,
+                            msg=f"key={key}\nactual={getattr(penalizer, key)}\nexpected={tensor}",
+                        )
+                actual_logits = orchestrator.apply(
+                    torch.ones(
+                        size=(len(filtered_subjects), self.vocab_size),
+                        dtype=torch.float32,
+                        device=self.device,
+                    )
+                )
+                filtered_expected_logits = torch.cat(
+                    tensors=[
+                        subject.steps[0].expected_logits
+                        for subject in filtered_subjects
+                    ],
+                )
+                torch.testing.assert_close(
+                    actual=actual_logits,
+                    expected=filtered_expected_logits,
+                    msg=f"logits\nactual={actual_logits}\nexpected={filtered_expected_logits}",
+                )
+    def test_merge_enabled_with_disabled(self):
+        enabled_test_case = self.test_cases[0]
+        disabled_test_case = self.test_cases[1]
+        orchestrator, penalizer = self._create_penalizer(enabled_test_case)
+        theirs, _ = self._create_penalizer(disabled_test_case)
+        orchestrator.merge(theirs)
+        for key, tensor in {
+            key: torch.cat(
+                tensors=[
+                    enabled_test_case.test_subjects[0].steps[0].expected_tensors[key],
+                    disabled_test_case.test_subjects[0].steps[0].expected_tensors[key],
+                ],
+            )
+            for key in enabled_test_case.tensor_keys()
+        }.items():
+            torch.testing.assert_close(
+                actual=getattr(penalizer, key),
+                expected=tensor,
+                msg=f"key={key}\nactual={getattr(penalizer, key)}\nexpected={tensor}",
+            )
+    def test_cumulate_apply_repeat(self):
+        for case in self.test_cases:
+            with self.subTest(case=case):
+                orchestrator, penalizer = self._create_penalizer(case)
+                max_step = max(len(subject.steps) for subject in case.test_subjects)
+                for i in range(1, max_step):
+                    orchestrator.filter(
+                        indices_to_keep=[
+                            j
+                            for j, subject in enumerate(case.test_subjects)
+                            if i < len(subject.steps)
+                        ]
+                    )
+                    filtered_subjects = [
+                        subject
+                        for subject in case.test_subjects
+                        if i < len(subject.steps)
+                    ]
+                    inputs: typing.List[typing.List[int]] = []
+                    outputs: typing.List[typing.List[int]] = []
+                    for subject in filtered_subjects:
+                        step = subject.steps[i]
+                        if step.type == StepType.INPUT:
+                            inputs.append(step.token_ids)
+                            outputs.append([])
+                        else:
+                            inputs.append([])
+                            outputs.append(step.token_ids)
+                    if any(inputs):
+                        orchestrator.cumulate_input_tokens(inputs)
+                    if any(outputs):
+                        orchestrator.cumulate_output_tokens(outputs)
+                    if penalizer.is_required():
+                        self.assertTrue(penalizer.is_prepared())
+                        for key, tensor in {
+                            key: torch.cat(
+                                tensors=[
+                                    subject.steps[i].expected_tensors[key]
+                                    for subject in filtered_subjects
+                                ],
+                            )
+                            for key in case.tensor_keys()
+                        }.items():
+                            torch.testing.assert_close(
+                                actual=getattr(penalizer, key),
+                                expected=tensor,
+                                msg=f"key={key}\nactual={getattr(penalizer, key)}\nexpected={tensor}",
+                            )
+                    actual_logits = orchestrator.apply(
+                        torch.ones(
+                            size=(len(filtered_subjects), self.vocab_size),
+                            dtype=torch.float32,
+                            device=self.device,
+                        )
+                    )
+                    filtered_expected_logits = torch.cat(
+                        tensors=[
+                            subject.steps[i].expected_logits
+                            for subject in filtered_subjects
+                        ],
+                    )
+                    torch.testing.assert_close(
+                        actual=actual_logits,
+                        expected=filtered_expected_logits,
+                        msg=f"logits\nactual={actual_logits}\nexpected={filtered_expected_logits}",
+                    )

sglang/test/test_layernorm.py ADDED Viewed

@@ -0,0 +1,60 @@
+import itertools
+import unittest
+import torch
+from sglang.srt.layers.layernorm import RMSNorm
+class TestRMSNorm(unittest.TestCase):
+    DTYPES = [torch.half, torch.bfloat16]
+    NUM_TOKENS = [7, 83, 4096]
+    HIDDEN_SIZES = [768, 769, 770, 771, 5120, 5124, 5125, 5126, 8192, 8199]
+    ADD_RESIDUAL = [False, True]
+    SEEDS = [0]
+    @classmethod
+    def setUpClass(cls):
+        if not torch.cuda.is_available():
+            raise unittest.SkipTest("CUDA is not available")
+        torch.set_default_device("cuda")
+    def _run_rms_norm_test(self, num_tokens, hidden_size, add_residual, dtype, seed):
+        torch.manual_seed(seed)
+        layer = RMSNorm(hidden_size).to(dtype=dtype)
+        layer.weight.data.normal_(mean=1.0, std=0.1)
+        scale = 1 / (2 * hidden_size)
+        x = torch.randn(num_tokens, hidden_size, dtype=dtype) * scale
+        residual = torch.randn_like(x) * scale if add_residual else None
+        with torch.inference_mode():
+            ref_out = layer.forward_native(x, residual)
+            out = layer(x, residual)
+        if add_residual:
+            self.assertTrue(torch.allclose(out[0], ref_out[0], atol=1e-2, rtol=1e-2))
+            self.assertTrue(torch.allclose(out[1], ref_out[1], atol=1e-2, rtol=1e-2))
+        else:
+            self.assertTrue(torch.allclose(out, ref_out, atol=1e-2, rtol=1e-2))
+    def test_rms_norm(self):
+        for params in itertools.product(
+            self.NUM_TOKENS,
+            self.HIDDEN_SIZES,
+            self.ADD_RESIDUAL,
+            self.DTYPES,
+            self.SEEDS,
+        ):
+            with self.subTest(
+                num_tokens=params[0],
+                hidden_size=params[1],
+                add_residual=params[2],
+                dtype=params[3],
+                seed=params[4],
+            ):
+                self._run_rms_norm_test(*params)
+if __name__ == "__main__":
+    unittest.main(verbosity=2)

sglang/test/test_programs.py CHANGED Viewed

@@ -103,16 +103,19 @@ def test_decode_int():
 def test_decode_json_regex():
     @sgl.function
     def decode_json(s):
-        from sglang.lang.ir import REGEX_FLOAT, REGEX_INT, REGEX_STRING
+        from sglang.lang.ir import REGEX_FLOAT, REGEX_INT, REGEX_STR
         s += "Generate a JSON object to describe the basic city information of Paris.\n"
+        s += "Here are the JSON object:\n"
+        # NOTE: we recommend using dtype gen or whole regex string to control the output
         with s.var_scope("json_output"):
             s += "{\n"
-            s += '  "name": ' + sgl.gen(regex=REGEX_STRING + ",") + "\n"
-            s += '  "population": ' + sgl.gen(regex=REGEX_INT + ",") + "\n"
-            s += '  "area": ' + sgl.gen(regex=REGEX_INT + ",") + "\n"
-            s += '  "latitude": ' + sgl.gen(regex=REGEX_FLOAT) + "\n"
+            s += '  "name": ' + sgl.gen(regex=REGEX_STR) + ",\n"
+            s += '  "population": ' + sgl.gen(regex=REGEX_INT, stop=[" ", "\n"]) + ",\n"
+            s += '  "area": ' + sgl.gen(regex=REGEX_INT, stop=[" ", "\n"]) + ",\n"
+            s += '  "latitude": ' + sgl.gen(regex=REGEX_FLOAT, stop=[" ", "\n"]) + "\n"
             s += "}"
     ret = decode_json.run(temperature=0.0)
@@ -149,7 +152,7 @@ def test_decode_json():
     assert isinstance(js_obj["population"], int)
-def test_expert_answer():
+def test_expert_answer(check_answer=True):
     @sgl.function
     def expert_answer(s, question):
         s += "Question: " + question + "\n"
@@ -167,7 +170,9 @@ def test_expert_answer():
         )
     ret = expert_answer.run(question="What is the capital of France?", temperature=0.1)
-    assert "paris" in ret.text().lower()
+    if check_answer:
+        assert "paris" in ret.text().lower(), f"Answer: {ret.text()}"
 def test_tool_use():
@@ -357,6 +362,30 @@ def test_regex():
     assert re.match(regex, answer)
+def test_dtype_gen():
+    @sgl.function
+    def dtype_gen(s):
+        s += "Q: What is the full name of DNS?\n"
+        s += "A: The full nams is " + sgl.gen("str_res", dtype=str, stop="\n") + "\n"
+        s += "Q: Which year was DNS invented?\n"
+        s += "A: " + sgl.gen("int_res", dtype=int) + "\n"
+        s += "Q: What is the value of pi?\n"
+        s += "A: " + sgl.gen("float_res", dtype=float) + "\n"
+        s += "Q: Is the sky blue?\n"
+        s += "A: " + sgl.gen("bool_res", dtype=bool) + "\n"
+    state = dtype_gen.run()
+    try:
+        state["int_res"] = int(state["int_res"])
+        state["float_res"] = float(state["float_res"])
+        state["bool_res"] = bool(state["bool_res"])
+        # assert state["str_res"].startswith('"') and state["str_res"].endswith('"')
+    except ValueError:
+        print(state)
+        raise
 def test_completion_speculative():
     @sgl.function(num_api_spec_tokens=64)
     def gen_character_spec(s):

sglang/test/test_utils.py CHANGED Viewed

@@ -12,6 +12,8 @@ from typing import Callable, List, Optional
 import numpy as np
 import requests
+import torch
+import torch.nn.functional as F
 from sglang.global_config import global_config
 from sglang.lang.backend.openai import OpenAI
@@ -19,6 +21,11 @@ from sglang.lang.backend.runtime_endpoint import RuntimeEndpoint
 from sglang.utils import get_exception_traceback
 DEFAULT_MODEL_NAME_FOR_TEST = "meta-llama/Meta-Llama-3.1-8B-Instruct"
+DEFAULT_MOE_MODEL_NAME_FOR_TEST = "mistralai/Mixtral-8x7B-Instruct-v0.1"
+DEFAULT_URL_FOR_MOE_TEST = "http://127.0.0.1:6157"
+DEFAULT_URL_FOR_ACCURACY_TEST = "http://127.0.0.1:7157"
+DEFAULT_URL_FOR_UNIT_TEST = "http://127.0.0.1:8157"
+DEFAULT_URL_FOR_E2E_TEST = "http://127.0.0.1:9157"
 def call_generate_lightllm(prompt, temperature, max_tokens, stop=None, url=None):
@@ -396,6 +403,8 @@ def popen_launch_server(
     timeout: float,
     api_key: Optional[str] = None,
     other_args: tuple = (),
+    env: Optional[dict] = None,
+    return_stdout_stderr: bool = False,
 ):
     _, host, port = base_url.split(":")
     host = host[2:]
@@ -415,7 +424,16 @@ def popen_launch_server(
     if api_key:
         command += ["--api-key", api_key]
-    process = subprocess.Popen(command, stdout=None, stderr=None)
+    if return_stdout_stderr:
+        process = subprocess.Popen(
+            command,
+            stdout=subprocess.PIPE,
+            stderr=subprocess.PIPE,
+            env=env,
+            text=True,
+        )
+    else:
+        process = subprocess.Popen(command, stdout=None, stderr=None, env=env)
     start_time = time.time()
     while time.time() - start_time < timeout:
@@ -482,7 +500,7 @@ def run_unittest_files(files: List[str], timeout_per_file: float):
             p.terminate()
             time.sleep(5)
             print(
-                "\nTimeout after {timeout_per_file} seconds when running {filename}\n"
+                f"\nTimeout after {timeout_per_file} seconds when running {filename}\n"
             )
             return False
@@ -492,3 +510,7 @@ def run_unittest_files(files: List[str], timeout_per_file: float):
         print(f"Fail. Time elapsed: {time.time() - tic:.2f}s")
     return 0 if success else -1
+def get_similarities(vec1, vec2):
+    return F.cosine_similarity(torch.tensor(vec1), torch.tensor(vec2), dim=0)

sglang/utils.py CHANGED Viewed

@@ -6,7 +6,6 @@ import json
 import logging
 import signal
 import sys
-import threading
 import traceback
 import urllib.request
 from concurrent.futures import ThreadPoolExecutor

sglang/version.py CHANGED Viewed

	@@ -1 +1 @@
1	- __version__ = "0.2.11"
1	+ __version__ = "0.2.13"

{sglang-0.2.11.dist-info → sglang-0.2.13.dist-info}/METADATA RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.1
 Name: sglang
-Version: 0.2.11
+Version: 0.2.13
 Summary: SGLang is yet another fast serving framework for large language models and vision language models.
 License: Apache License
                                    Version 2.0, January 2004
@@ -308,7 +308,7 @@ pip install flashinfer -i https://flashinfer.ai/whl/cu121/torch2.4/
 ### Method 2: From source
 ```
 # Use the last release branch
-git clone -b v0.2.11 https://github.com/sgl-project/sglang.git
+git clone -b v0.2.13 https://github.com/sgl-project/sglang.git
 cd sglang
 pip install --upgrade pip
@@ -329,11 +329,19 @@ docker run --gpus all \
     --env "HF_TOKEN=<secret>" \
     --ipc=host \
     lmsysorg/sglang:latest \
-    python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --host 0.0.0.0 --port 30000
+    python3 -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-8B-Instruct --host 0.0.0.0 --port 30000
 ```
+### Method 4: Using docker compose
+> This method is recommended if you plan to serve it as a service.
+> A better approach is to use the [k8s-sglang-service.yaml](./docker/k8s-sglang-service.yaml).
+1. Copy the [compose.yml](./docker/compose.yaml) to your local machine
+2. Execute the command `docker compose up -d` in your terminal.
 ### Common Notes
-- If you cannot install FlashInfer, check out its [installation](https://docs.flashinfer.ai/installation.html#) page. If you still cannot install it, you can use the slower Triton kernels by adding `--disable-flashinfer` when launching the server.
+- [FlashInfer](https://github.com/flashinfer-ai/flashinfer) is currently one of the dependencies that must be installed for SGLang. If you are using NVIDIA GPU devices below sm80, such as T4, you can't use SGLang for the time being. We expect to resolve this issue soon, so please stay tuned. If you encounter any FlashInfer-related issues on sm80+ devices (e.g., A100, L40S, H100), consider using Triton's kernel by `--disable-flashinfer --disable-flashinfer-sampling` and raise a issue.
 - If you only need to use the OpenAI backend, you can avoid installing other dependencies by using `pip install "sglang[openai]"`.
 ## Backend: SGLang Runtime (SRT)
@@ -392,23 +400,23 @@ print(response)
 It supports streaming, vision, and most features of the Chat/Completions/Models/Batch endpoints specified by the [OpenAI API Reference](https://platform.openai.com/docs/api-reference/).
 ### Additional Server Arguments
-- Add `--tp 2` to enable tensor parallelism. If it indicates `peer access is not supported between these two devices`, add `--enable-p2p-check` option.
+- Add `--tp 2` to enable multi-GPU tensor parallelism. If it reports the error "peer access is not supported between these two devices", add `--enable-p2p-check` to the server launch command.
 ```
 python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --port 30000 --tp 2
 ```
-- Add `--dp 2` to enable data parallelism. It can also be used together with tp. Data parallelism is better for throughput if there is enough memory.
+- Add `--dp 2` to enable multi-GPU data parallelism. It can also be used together with tensor parallelism. Data parallelism is better for throughput if there is enough memory.
 ```
 python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --port 30000 --dp 2 --tp 2
 ```
-- If you see out-of-memory errors during serving, please try to reduce the memory usage of the KV cache pool by setting a smaller value of `--mem-fraction-static`. The default value is `0.9`.
+- If you see out-of-memory errors during serving, try to reduce the memory usage of the KV cache pool by setting a smaller value of `--mem-fraction-static`. The default value is `0.9`.
 ```
 python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --port 30000 --mem-fraction-static 0.7
 ```
-- If you see out-of-memory errors during prefill for long prompts on a model that supports long context, consider using chunked prefill.
+- See [hyperparameter_tuning.md](docs/en/hyperparameter_tuning.md) on tuning hyperparameters for better performance.
+- If you see out-of-memory errors during prefill for long prompts, try to set a smaller chunked prefill size.
 ```
-python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3.1-8B-Instruct --port 30000 --chunked-prefill-size 8192
+python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --port 30000 --chunked-prefill-size 4096
 ```
-- See [hyperparameter_tuning.md](docs/en/hyperparameter_tuning.md) on tuning hyperparameters for better performance.
 - Add `--nnodes 2` to run tensor parallelism on multiple nodes. If you have two nodes with two GPUs on each node and want to run TP=4, let `sgl-dev-0` be the hostname of the first node and `50000` be an available port.
 ```
 # Node 0
@@ -418,13 +426,13 @@ python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct
 python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct --tp 4 --nccl-init sgl-dev-0:50000 --nnodes 2 --node-rank 1
 ```
 - If the model does not have a template in the Hugging Face tokenizer, you can specify a [custom chat template](docs/en/custom_chat_template.md).
-- To enable fp8 quantization, you can add `--quantization fp8` on a fp16 checkpoint or directly load a fp8 checkpoint without specifying any arguments.
 - To enable experimental torch.compile support, you can add `--enable-torch-compile`. It accelerates small models on small batch sizes.
+- To enable fp8 quantization, you can add `--quantization fp8` on a fp16 checkpoint or directly load a fp8 checkpoint without specifying any arguments.
 ### Supported Models
 - Llama / Llama 2 / Llama 3 / Llama 3.1
-- Mistral / Mixtral
+- Mistral / Mixtral / Mistral NeMo
 - Gemma / Gemma 2
 - Qwen / Qwen 2 / Qwen 2 MoE
 - DeepSeek / DeepSeek 2
@@ -442,11 +450,20 @@ python -m sglang.launch_server --model-path meta-llama/Meta-Llama-3-8B-Instruct
 - Grok
 - ChatGLM
 - InternLM 2
-- Mistral NeMo
 Instructions for supporting a new model are [here](https://github.com/sgl-project/sglang/blob/main/docs/en/model_support.md).
-### Run Llama 3.1 405B
+#### Use Models From ModelScope
+To use model from [ModelScope](https://www.modelscope.cn), setting environment variable SGLANG_USE_MODELSCOPE.
+```
+export SGLANG_USE_MODELSCOPE=true
+```
+Launch [Qwen2-7B-Instruct](https://www.modelscope.cn/models/qwen/qwen2-7b-instruct) Server
+```
+SGLANG_USE_MODELSCOPE=true python -m sglang.launch_server --model-path qwen/Qwen2-7B-Instruct --port 30000
+```
+#### Run Llama 3.1 405B
 ```bash
 ## Run 405B (fp8) on a single node
@@ -474,7 +491,7 @@ GLOO_SOCKET_IFNAME=eth0 python3 -m sglang.launch_server --model-path meta-llama/
   ```
 ## Frontend: Structured Generation Language (SGLang)
-The frontend language can be used with local models or API models.
+The frontend language can be used with local models or API models. It is an alternative to the OpenAI API. You may found it easier to use for complex prompting workflow.
 ### Quick Start
 The example below shows how to use sglang to answer a mulit-turn question.

sglang 0.2.11__py3-none-any.whl → 0.2.13__py3-none-any.whl

sglang 0.2.11py3-none-any.whl → 0.2.13py3-none-any.whl