PyPI - sglang - Versions diffs - 0.4.1.post3__py3-none-any.whl → 0.4.1.post4__py3-none-any.whl - Mend

sglang 0.4.1.post3py3-none-any.whl → 0.4.1.post4py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (63) hide show

sglang/srt/managers/schedule_batch.py CHANGED Viewed

@@ -1,3 +1,5 @@
+from __future__ import annotations
 # Copyright 2023-2024 SGLang Team
 # Licensed under the Apache License, Version 2.0 (the "License");
 # you may not use this file except in compliance with the License.
@@ -29,7 +31,7 @@ ScheduleBatch -> ModelWorkerBatch -> ForwardBatch
 import dataclasses
 import logging
-from typing import List, Optional, Set, Tuple, Union
+from typing import TYPE_CHECKING, List, Optional, Set, Tuple, Union
 import numpy as np
 import torch
@@ -47,6 +49,10 @@ from sglang.srt.sampling.sampling_batch_info import SamplingBatchInfo
 from sglang.srt.sampling.sampling_params import SamplingParams
 from sglang.srt.server_args import ServerArgs
+if TYPE_CHECKING:
+    from sglang.srt.speculative.spec_info import SpecInfo, SpeculativeAlgorithm
 INIT_INCREMENTAL_DETOKENIZATION_OFFSET = 5
 # Put some global args for easy access
@@ -565,9 +571,13 @@ class ScheduleBatch:
     # Has grammar
     has_grammar: bool = False
-    # device
+    # Device
     device: str = "cuda"
+    # Speculative decoding
+    spec_algorithm: SpeculativeAlgorithm = None
+    spec_info: Optional[SpecInfo] = None
     @classmethod
     def init_new(
         cls,
@@ -577,6 +587,7 @@ class ScheduleBatch:
         tree_cache: BasePrefixCache,
         model_config: ModelConfig,
         enable_overlap: bool,
+        spec_algorithm: SpeculativeAlgorithm,
     ):
         return cls(
             reqs=reqs,
@@ -589,6 +600,7 @@ class ScheduleBatch:
             has_stream=any(req.stream for req in reqs),
             has_grammar=any(req.grammar for req in reqs),
             device=req_to_token_pool.device,
+            spec_algorithm=spec_algorithm,
         )
     def batch_size(self):
@@ -998,6 +1010,8 @@ class ScheduleBatch:
     def prepare_for_decode(self):
         self.forward_mode = ForwardMode.DECODE
+        if self.spec_algorithm.is_eagle():
+            return
         self.input_ids = self.output_ids
         self.output_ids = None
@@ -1103,6 +1117,9 @@ class ScheduleBatch:
         self.has_stream |= other.has_stream
         self.has_grammar |= other.has_grammar
+        if self.spec_info:
+            self.spec_info.merge_batch(other.spec_info)
     def get_model_worker_batch(self):
         if self.forward_mode.is_decode() or self.forward_mode.is_idle():
             extend_seq_lens = extend_prefix_lens = extend_logprob_start_lens = None
@@ -1144,6 +1161,8 @@ class ScheduleBatch:
             lora_paths=[req.lora_path for req in self.reqs],
             sampling_info=self.sampling_info,
             input_embeds=self.input_embeds,
+            spec_algorithm=self.spec_algorithm,
+            spec_info=self.spec_info,
         )
     def copy(self):
@@ -1155,6 +1174,7 @@ class ScheduleBatch:
             out_cache_loc=self.out_cache_loc,
             return_logprob=self.return_logprob,
             decoding_reqs=self.decoding_reqs,
+            spec_algorithm=self.spec_algorithm,
         )
     def __str__(self):
@@ -1214,6 +1234,10 @@ class ModelWorkerBatch:
     # The input Embeds
     input_embeds: Optional[torch.tensor] = None
+    # Speculative decoding
+    spec_algorithm: SpeculativeAlgorithm = None
+    spec_info: Optional[SpecInfo] = None
 @triton.jit
 def write_req_to_token_pool_triton(

sglang/srt/managers/schedule_policy.py CHANGED Viewed

@@ -18,7 +18,7 @@ import random
 from collections import defaultdict
 from contextlib import contextmanager
 from enum import Enum, auto
-from typing import Dict, List, Optional
+from typing import Dict, List, Optional, Set, Union
 import torch
@@ -50,13 +50,26 @@ IN_BATCH_PREFIX_CACHING_DEPRIORITIZE_THRESHOLD = int(
 )
+class CacheAwarePolicy(Enum):
+    """Scheduling policies that are aware of the tree cache."""
+    LPM = "lpm"  # longest prefix match
+    DFS_WEIGHT = "dfs-weight"  # depth-first search weighting
+class CacheAgnosticPolicy(Enum):
+    """Scheduling policies that are not aware of the tree cache."""
+    FCFS = "fcfs"  # first come first serve
+    LOF = "lof"  # longest output first
+    RANDOM = "random"
 class SchedulePolicy:
-    def __init__(self, policy: str, tree_cache: BasePrefixCache):
-        if tree_cache.disable and policy in ["lpm", "dfs-weight"]:
-            # LPM and DFS-weight is meaningless when the tree cache is disabled.
-            policy = "fcfs"
+    Policy = Union[CacheAwarePolicy, CacheAgnosticPolicy]
-        self.policy = policy
+    def __init__(self, policy: str, tree_cache: BasePrefixCache):
+        self.policy = self._validate_and_adjust_policy(policy, tree_cache)
         self.tree_cache = tree_cache
         # It is used to find the matching prefix for in-batch prefix caching.
@@ -64,110 +77,166 @@ class SchedulePolicy:
             req_to_token_pool=None, token_to_kv_pool=None, disable=False
         )
-    def calc_priority(self, waiting_queue: List[Req]):
-        if len(waiting_queue) > 128 and self.policy == "lpm":
-            # Turn off the expensive prefix matching and sorting when the #queue is large.
-            policy = "fcfs"
-        else:
-            policy = self.policy
+    def calc_priority(self, waiting_queue: List[Req]) -> bool:
+        policy = self._determine_active_policy(waiting_queue)
-        # Compute matched prefix length
         prefix_computed = False
-        if policy == "lpm" or policy == "dfs-weight":
-            # rid to deprioritize in the current run for in-batch prefix caching.
-            temporary_deprioritized = set()
-            self.waiting_queue_radix_tree.reset()
-            for r in waiting_queue:
-                prefix_ids = r.adjust_max_prefix_ids()
-                # NOTE: the prefix_indices must always be aligned with last_node
-                r.prefix_indices, r.last_node = self.tree_cache.match_prefix(
-                    rid=r.rid, key=prefix_ids
+        if isinstance(policy, CacheAwarePolicy):
+            prefix_computed = True
+            temporary_deprioritized = self._compute_prefix_matches(
+                waiting_queue, policy
+            )
+            if policy == CacheAwarePolicy.LPM:
+                SchedulePolicy._sort_by_longest_prefix(
+                    waiting_queue, temporary_deprioritized
                 )
+            elif policy == CacheAwarePolicy.DFS_WEIGHT:
+                SchedulePolicy._sort_by_dfs_weight(waiting_queue, self.tree_cache)
+            else:
+                raise ValueError(f"Unknown CacheAware Policy: {policy=}")
+        else:
+            if policy == CacheAgnosticPolicy.FCFS:
+                pass
+            elif policy == CacheAgnosticPolicy.LOF:
+                SchedulePolicy._sort_by_longest_output(waiting_queue)
+            elif policy == CacheAgnosticPolicy.RANDOM:
+                SchedulePolicy._sort_randomly(waiting_queue)
+            else:
+                raise ValueError(f"Unknown CacheAgnostic Policy: {policy=}")
-                # NOTE(sang): This logic is for in-batch prefix caching;
-                # If there are more than 1 request that have small matching prefix from
-                # existing cache, but all those requests share the same prefix, we prefer
-                # to schedule only one of them so that we can increase the cache hit rate.
-                # We prefer to set IN_BATCH_PREFIX_CACHING_CHECK_THRESHOLD > 0 because too small
-                # threshold means we cannot use in-batch prefix caching for short prefixes.
-                # It is kind of common when the engine is long running (e.g., imagine the prefix "the").
-                if len(r.prefix_indices) <= IN_BATCH_PREFIX_CACHING_CHECK_THRESHOLD:
-                    in_batch_matching_prefixes, _ = (
-                        self.waiting_queue_radix_tree.match_prefix(
-                            rid=r.rid, key=prefix_ids
-                        )
-                    )
-                    if (
-                        len(in_batch_matching_prefixes)
-                        >= IN_BATCH_PREFIX_CACHING_DEPRIORITIZE_THRESHOLD
-                    ):
-                        temporary_deprioritized.add(r.rid)
-                    else:
-                        # Insert with a dummy key
-                        self.waiting_queue_radix_tree.insert(
-                            prefix_ids, torch.empty(len(prefix_ids), dtype=torch.bool)
-                        )
+        return prefix_computed
-            prefix_computed = True
+    def _determine_active_policy(self, waiting_queue: List[Req]) -> Policy:
+        if len(waiting_queue) > 128 and self.policy == CacheAwarePolicy.LPM:
+            # Turn off the expensive prefix matching and sorting when the #queue is large.
+            return CacheAgnosticPolicy.FCFS
+        return self.policy
+    def _validate_and_adjust_policy(
+        self, policy: str, tree_cache: BasePrefixCache
+    ) -> Policy:
+        """
+        Validates the policy and adjusts it if necessary based on tree cache settings.
+        """
+        try:
+            policy_enum = CacheAwarePolicy(policy)
+            if tree_cache.disable:
+                # If tree_cache is disabled, using CacheAgnosticPolicy policy
+                return CacheAgnosticPolicy.FCFS
+            return policy_enum
+        except ValueError:
+            try:
+                return CacheAgnosticPolicy(policy)
+            except ValueError:
+                raise ValueError(f"Unknown schedule_policy: {policy=}")
+    def _compute_prefix_matches(
+        self, waiting_queue: List[Req], policy: CacheAwarePolicy
+    ) -> Set[int]:
+        """
+        Computes and caches the matching prefixes for requests in the waiting queue,
+            and handles in-batch prefix caching logic.
+        """
+        temporary_deprioritized: Set[int] = set()
+        self.waiting_queue_radix_tree.reset()
+        for r in waiting_queue:
+            prefix_ids = r.adjust_max_prefix_ids()
+            # NOTE: the prefix_indices must always be aligned with last_node
+            r.prefix_indices, r.last_node = self.tree_cache.match_prefix(
+                rid=r.rid, key=prefix_ids
+            )
-        if policy == "lpm":
-            # Longest Prefix Match
-            waiting_queue.sort(
-                key=lambda r: (
-                    -len(r.prefix_indices)
-                    if r.rid not in temporary_deprioritized
-                    else float("inf")
+            # NOTE(sang): This logic is for in-batch prefix caching;
+            # If there are more than 1 request that have small matching prefix from
+            # existing cache, but all those requests share the same prefix, we prefer
+            # to schedule only one of them so that we can increase the cache hit rate.
+            # We prefer to set IN_BATCH_PREFIX_CACHING_CHECK_THRESHOLD > 0 because too small
+            # threshold means we cannot use in-batch prefix caching for short prefixes.
+            # It is kind of common when the engine is long running (e.g., imagine the prefix "the").
+            if len(r.prefix_indices) <= IN_BATCH_PREFIX_CACHING_CHECK_THRESHOLD:
+                in_batch_matching_prefixes, _ = (
+                    self.waiting_queue_radix_tree.match_prefix(
+                        rid=r.rid, key=prefix_ids
+                    )
                 )
+                if (
+                    len(in_batch_matching_prefixes)
+                    >= IN_BATCH_PREFIX_CACHING_DEPRIORITIZE_THRESHOLD
+                ):
+                    temporary_deprioritized.add(r.rid)
+                else:
+                    # Insert with a dummy key
+                    self.waiting_queue_radix_tree.insert(
+                        prefix_ids, torch.empty(len(prefix_ids), dtype=torch.bool)
+                    )
+        return temporary_deprioritized
+    @staticmethod
+    def _sort_by_longest_prefix(
+        waiting_queue: List[Req], temporary_deprioritized: Set[int]
+    ) -> None:
+        """Sorts the waiting queue based on the longest prefix match."""
+        waiting_queue.sort(
+            key=lambda r: (
+                -len(r.prefix_indices)
+                if r.rid not in temporary_deprioritized
+                else float("inf")
             )
-        elif policy == "fcfs":
-            # first come first serve
-            pass
-        elif policy == "lof":
-            # longest output first
-            waiting_queue.sort(key=lambda x: -x.sampling_params.max_new_tokens)
-        elif policy == "random":
-            random.shuffle(waiting_queue)
-        elif policy == "dfs-weight":
-            # Experimental policy based on custom weights
-            last_node_to_reqs = defaultdict(list)
-            for req in waiting_queue:
-                last_node_to_reqs[req.last_node].append(req)
-            node_to_weight = defaultdict(int)
-            for node in last_node_to_reqs:
-                node_to_weight[node] = len(last_node_to_reqs[node])
-            self.calc_weight(self.tree_cache.root_node, node_to_weight)
-            waiting_queue.clear()
-            self.get_dfs_priority(
-                self.tree_cache.root_node,
-                node_to_weight,
-                last_node_to_reqs,
-                waiting_queue,
-            )
-        else:
-            raise ValueError(f"Unknown schedule_policy: {policy=}")
+        )
-        return prefix_computed
+    @staticmethod
+    def _sort_by_dfs_weight(
+        waiting_queue: List[Req], tree_cache: BasePrefixCache
+    ) -> None:
+        """Sorts the waiting queue based on a depth-first search weighting."""
+        last_node_to_reqs = defaultdict(list)
+        for req in waiting_queue:
+            last_node_to_reqs[req.last_node].append(req)
+        node_to_weight = defaultdict(int)
+        for node in last_node_to_reqs:
+            node_to_weight[node] = len(last_node_to_reqs[node])
+        SchedulePolicy._calc_weight(tree_cache.root_node, node_to_weight)
+        waiting_queue.clear()
+        SchedulePolicy._get_dfs_priority(
+            tree_cache.root_node,
+            node_to_weight,
+            last_node_to_reqs,
+            waiting_queue,
+        )
+    @staticmethod
+    def _sort_by_longest_output(waiting_queue: List[Req]) -> None:
+        """Sorts the waiting queue based on the longest output (max_new_tokens)."""
+        waiting_queue.sort(key=lambda x: -x.sampling_params.max_new_tokens)
-    def calc_weight(self, cur_node: TreeNode, node_to_weight: Dict):
+    @staticmethod
+    def _sort_randomly(waiting_queue: List[Req]) -> None:
+        """Shuffles the waiting queue randomly."""
+        random.shuffle(waiting_queue)
+    @staticmethod
+    def _calc_weight(cur_node: TreeNode, node_to_weight: Dict[TreeNode, int]) -> None:
         for child in cur_node.children.values():
-            self.calc_weight(child, node_to_weight)
+            SchedulePolicy._calc_weight(child, node_to_weight)
             node_to_weight[cur_node] += node_to_weight[child]
-    def get_dfs_priority(
-        self,
+    @staticmethod
+    def _get_dfs_priority(
         cur_node: TreeNode,
         node_to_priority: Dict[TreeNode, int],
         last_node_to_reqs: Dict[TreeNode, List[Req]],
         q: List,
-    ):
+    ) -> None:
         childs = [child for child in cur_node.children.values()]
         childs.sort(key=lambda x: -node_to_priority[x])
         for child in childs:
-            self.get_dfs_priority(child, node_to_priority, last_node_to_reqs, q)
+            SchedulePolicy._get_dfs_priority(
+                child, node_to_priority, last_node_to_reqs, q
+            )
         q.extend(last_node_to_reqs[cur_node])

sglang/srt/managers/scheduler.py CHANGED Viewed

@@ -76,6 +76,7 @@ from sglang.srt.mem_cache.radix_cache import RadixCache
 from sglang.srt.metrics.collector import SchedulerMetricsCollector, SchedulerStats
 from sglang.srt.model_executor.forward_batch_info import ForwardMode
 from sglang.srt.server_args import PortArgs, ServerArgs
+from sglang.srt.speculative.spec_info import SpeculativeAlgorithm
 from sglang.srt.utils import (
     broadcast_pyobj,
     configure_logger,
@@ -116,6 +117,14 @@ class Scheduler:
         self.enable_overlap = not server_args.disable_overlap_schedule
         self.skip_tokenizer_init = server_args.skip_tokenizer_init
         self.enable_metrics = server_args.enable_metrics
+        self.spec_algorithm = SpeculativeAlgorithm.from_string(
+            server_args.speculative_algorithm
+        )
+        self.decode_mem_cache_buf_multiplier = (
+            self.server_args.speculative_num_draft_tokens
+            if not self.spec_algorithm.is_none()
+            else 1
+        )
         # Init inter-process communication
         context = zmq.Context(2)
@@ -199,6 +208,21 @@ class Scheduler:
             nccl_port=port_args.nccl_port,
         )
+        # Launch worker for speculative decoding if need
+        if self.spec_algorithm.is_eagle():
+            from sglang.srt.speculative.eagle_worker import EAGLEWorker
+            self.draft_worker = EAGLEWorker(
+                gpu_id=gpu_id,
+                tp_rank=tp_rank,
+                server_args=server_args,
+                nccl_port=port_args.nccl_port,
+                target_worker=self.tp_worker,
+                dp_rank=dp_rank,
+            )
+        else:
+            self.draft_worker = None
         # Get token and memory info from the model worker
         (
             self.max_total_num_tokens,
@@ -855,6 +879,7 @@ class Scheduler:
             self.tree_cache,
             self.model_config,
             self.enable_overlap,
+            self.spec_algorithm,
         )
         new_batch.prepare_for_extend()
@@ -888,11 +913,15 @@ class Scheduler:
             return None
         # Check if decode out of memory
-        if not batch.check_decode_mem() or (test_retract and batch.batch_size() > 10):
+        if not batch.check_decode_mem(self.decode_mem_cache_buf_multiplier) or (
+            test_retract and batch.batch_size() > 10
+        ):
             old_ratio = self.new_token_ratio
             retracted_reqs, new_token_ratio = batch.retract_decode()
             self.new_token_ratio = new_token_ratio
+            if self.draft_worker:
+                self.draft_worker.finish_request(retracted_reqs)
             logger.info(
                 "Decode out of memory happened. "
@@ -926,11 +955,17 @@ class Scheduler:
         self.forward_ct += 1
         if self.is_generation:
-            model_worker_batch = batch.get_model_worker_batch()
             if batch.forward_mode.is_decode() or batch.extend_num_tokens != 0:
-                logits_output, next_token_ids = self.tp_worker.forward_batch_generation(
-                    model_worker_batch
-                )
+                if self.spec_algorithm.is_none():
+                    model_worker_batch = batch.get_model_worker_batch()
+                    logits_output, next_token_ids = (
+                        self.tp_worker.forward_batch_generation(model_worker_batch)
+                    )
+                else:
+                    logits_output, next_token_ids, model_worker_batch, spec_info = (
+                        self.draft_worker.forward_batch_speculative_generation(batch)
+                    )
+                    batch.spec_info = spec_info
             elif batch.forward_mode.is_idle():
                 model_worker_batch = batch.get_model_worker_batch()
                 self.tp_worker.forward_batch_idle(model_worker_batch)
@@ -974,12 +1009,10 @@ class Scheduler:
                 logits_output, next_token_ids = self.tp_worker.resolve_batch_result(bid)
             else:
                 # Move next_token_ids and logprobs to cpu
+                next_token_ids = next_token_ids.tolist()
                 if batch.return_logprob:
                     logits_output.next_token_logprobs = (
-                        logits_output.next_token_logprobs[
-                            torch.arange(len(next_token_ids), device=self.device),
-                            next_token_ids,
-                        ].tolist()
+                        logits_output.next_token_logprobs.tolist()
                     )
                     logits_output.input_token_logprobs = (
                         logits_output.input_token_logprobs.tolist()
@@ -987,7 +1020,6 @@ class Scheduler:
                     logits_output.normalized_prompt_logprobs = (
                         logits_output.normalized_prompt_logprobs.tolist()
                     )
-                next_token_ids = next_token_ids.tolist()
             # Check finish conditions
             logprob_pt = 0
@@ -1064,13 +1096,9 @@ class Scheduler:
             logits_output, next_token_ids = self.tp_worker.resolve_batch_result(bid)
             next_token_logprobs = logits_output.next_token_logprobs
         else:
-            # Move next_token_ids and logprobs to cpu
-            if batch.return_logprob:
-                next_token_logprobs = logits_output.next_token_logprobs[
-                    torch.arange(len(next_token_ids), device=self.device),
-                    next_token_ids,
-                ].tolist()
             next_token_ids = next_token_ids.tolist()
+            if batch.return_logprob:
+                next_token_logprobs = logits_output.next_token_logprobs.tolist()
         self.token_to_kv_pool.free_group_begin()
@@ -1084,7 +1112,10 @@ class Scheduler:
                 self.token_to_kv_pool.free(batch.out_cache_loc[i : i + 1])
                 continue
-            req.output_ids.append(next_token_id)
+            if batch.spec_algorithm.is_none():
+                # speculative worker will solve the output_ids in speculative decoding
+                req.output_ids.append(next_token_id)
             req.check_finished()
             if req.finished():
@@ -1095,10 +1126,10 @@ class Scheduler:
                 req.output_token_logprobs_idx.append(next_token_id)
                 if req.top_logprobs_num > 0:
                     req.output_top_logprobs_val.append(
-                        logits_output.output_top_logprobs_val[i]
+                        logits_output.next_token_top_logprobs_val[i]
                     )
                     req.output_top_logprobs_idx.append(
-                        logits_output.output_top_logprobs_idx[i]
+                        logits_output.next_token_top_logprobs_idx[i]
                     )
             if req.grammar is not None:
@@ -1200,8 +1231,9 @@ class Scheduler:
                 req.output_top_logprobs_idx.extend(
                     output.input_top_logprobs_idx[i][-req.last_update_decode_tokens :]
                 )
-            req.output_top_logprobs_val.append(output.output_top_logprobs_val[i])
-            req.output_top_logprobs_idx.append(output.output_top_logprobs_idx[i])
+            req.output_top_logprobs_val.append(output.next_token_top_logprobs_val[i])
+            req.output_top_logprobs_idx.append(output.next_token_top_logprobs_idx[i])
         return num_input_logprobs
@@ -1258,6 +1290,9 @@ class Scheduler:
                     # If not stream, we still want to output some tokens to get the benefit of incremental decoding.
                     or (not req.stream and len(req.output_ids) % 50 == 0)
                 ):
+                    if self.draft_worker and req.finished():
+                        self.draft_worker.finish_request(req)
                     rids.append(req.rid)
                     finished_reasons.append(
                         req.finished_reason.to_json() if req.finished_reason else None
@@ -1329,11 +1364,11 @@ class Scheduler:
             embeddings = []
             prompt_tokens = []
             for req in reqs:
-                assert req.finished()
-                rids.append(req.rid)
-                finished_reasons.append(req.finished_reason.to_json())
-                embeddings.append(req.embedding)
-                prompt_tokens.append(len(req.origin_input_ids))
+                if req.finished():
+                    rids.append(req.rid)
+                    finished_reasons.append(req.finished_reason.to_json())
+                    embeddings.append(req.embedding)
+                    prompt_tokens.append(len(req.origin_input_ids))
             self.send_to_detokenizer.send_pyobj(
                 BatchEmbeddingOut(rids, finished_reasons, embeddings, prompt_tokens)
             )
@@ -1389,6 +1424,7 @@ class Scheduler:
             self.tree_cache,
             self.model_config,
             self.enable_overlap,
+            self.spec_algorithm,
         )
         idle_batch.prepare_for_idle()
         return idle_batch

sglang 0.4.1.post3__py3-none-any.whl → 0.4.1.post4__py3-none-any.whl

sglang 0.4.1.post3py3-none-any.whl → 0.4.1.post4py3-none-any.whl