[spec decoding] implement proposing tree drafts

TheEpicDolphin · TheEpicDolphin · commit 0e691d5eda15 · 2025-07-15T16:21:54.000-07:00
Signed-off-by: Giancarlo Delfin &lt;gdelfin@meta.com&gt;
diff --git a/pyproject.toml b/pyproject.toml
@@ -61,7 +61,7 @@ ignore_patterns = [
 
 [tool.ruff]
 # Allow lines to be as long as 80.
-line-length = 80
+line-length = 90
 
 [tool.ruff.lint.per-file-ignores]
 "vllm/third_party/**" = ["ALL"]
diff --git a/vllm/attention/layer.py b/vllm/attention/layer.py
@@ -52,7 +52,6 @@ def __init__(
         prefix: str = "",
         attn_type: str = AttentionType.DECODER,
         kv_sharing_target_layer_name: Optional[str] = None,
-        is_draft: bool = False,
         **extra_impl_args,
     ) -> None:
         """
@@ -136,8 +135,7 @@ def __init__(
                                         block_size,
                                         is_attention_free,
                                         blocksparse_params is not None,
-                                        use_mla=use_mla,
-                                        is_draft=is_draft)
+                                        use_mla=use_mla)
         impl_cls = attn_backend.get_impl_cls()
         self.impl = impl_cls(num_heads, head_size, scale, num_kv_heads,
                              alibi_slopes, sliding_window, kv_cache_dtype,
diff --git a/vllm/attention/selector.py b/vllm/attention/selector.py
@@ -27,22 +27,23 @@ def backend_name_to_enum(backend_name: str) -> Optional[_Backend]:
             loaded.
     """
     assert backend_name is not None
-    return _Backend[
-        backend_name] if backend_name in _Backend.__members__ else None
+    return _Backend[backend_name] if backend_name in _Backend.__members__ else \
+          None
 
 
 def get_env_variable_attn_backend() -> Optional[_Backend]:
-    """
+    '''
     Get the backend override specified by the vLLM attention
     backend environment variable, if one is specified.
 
     Returns:
 
     * _Backend enum value if an override is specified
     * None otherwise
-    """
+    '''
     backend_name = os.environ.get(STR_BACKEND_ENV_VAR)
-    return None if backend_name is None else backend_name_to_enum(backend_name)
+    return (None
+            if backend_name is None else backend_name_to_enum(backend_name))
 
 
 # Global state allows a particular choice of backend
@@ -56,7 +57,7 @@ def get_env_variable_attn_backend() -> Optional[_Backend]:
 
 
 def global_force_attn_backend(attn_backend: Optional[_Backend]) -> None:
-    """
+    '''
     Force all attention operations to use a specified backend.
 
     Passing `None` for the argument re-enables automatic
@@ -65,16 +66,16 @@ def global_force_attn_backend(attn_backend: Optional[_Backend]) -> None:
     Arguments:
 
     * attn_backend: backend selection (None to revert to auto)
-    """
+    '''
     global forced_attn_backend
     forced_attn_backend = attn_backend
 
 
 def get_global_forced_attn_backend() -> Optional[_Backend]:
-    """
+    '''
     Get the currently-forced choice of attention backend,
     or None if auto-selection is currently enabled.
-    """
+    '''
     return forced_attn_backend
 
 
@@ -86,7 +87,6 @@ def get_attn_backend(
     is_attention_free: bool,
     is_blocksparse: bool = False,
     use_mla: bool = False,
-    is_draft: bool = False,
 ) -> Type[AttentionBackend]:
     """Selects which attention backend to use and lazily imports it."""
     # Accessing envs.* behind an @lru_cache decorator can cause the wrong
@@ -102,7 +102,6 @@ def get_attn_backend(
         is_blocksparse=is_blocksparse,
         use_v1=envs.VLLM_USE_V1,
         use_mla=use_mla,
-        is_draft=is_draft,
     )
 
 
@@ -116,28 +115,18 @@ def _cached_get_attn_backend(
     is_blocksparse: bool = False,
     use_v1: bool = False,
     use_mla: bool = False,
-    is_draft: bool = False,
 ) -> Type[AttentionBackend]:
-    # Draft model backend is currently forced to FlashAttentionBackend for
-    # consistency with EagleProposer using FlashAttentionMetadata.
-    if use_v1 and is_draft:
-        from vllm.v1.attention.backends.flash_attn import FlashAttentionBackend
-
-        return FlashAttentionBackend
-
     if is_blocksparse:
         logger.info("Using BlocksparseFlashAttention backend.")
         from vllm.attention.backends.blocksparse_attn import (
             BlocksparseFlashAttentionBackend)
-
         return BlocksparseFlashAttentionBackend
 
     # If there are no attention layers (e.g. we are running Mamba),
     # use the placeholder NO_ATTENTION
     if is_attention_free:
         from vllm.attention.backends.placeholder_attn import (
             PlaceholderAttentionBackend)
-
         return PlaceholderAttentionBackend
 
     # Check whether a particular choice of backend was
@@ -146,8 +135,8 @@ def _cached_get_attn_backend(
     # THIS SELECTION OVERRIDES THE VLLM_ATTENTION_BACKEND
     # ENVIRONMENT VARIABLE.
     selected_backend = None
-    backend_by_global_setting: Optional[
-        _Backend] = get_global_forced_attn_backend()
+    backend_by_global_setting: Optional[_Backend] = (
+        get_global_forced_attn_backend())
     if backend_by_global_setting is not None:
         selected_backend = backend_by_global_setting
     else:
@@ -168,8 +157,8 @@ def _cached_get_attn_backend(
 
 @contextmanager
 def global_force_attn_backend_context_manager(
-    attn_backend: _Backend, ) -> Generator[None, None, None]:
-    """
+        attn_backend: _Backend) -> Generator[None, None, None]:
+    '''
     Globally force a vLLM attention backend override within a
     context manager, reverting the global attention backend
     override to its prior state upon exiting the context
@@ -182,7 +171,7 @@ def global_force_attn_backend_context_manager(
     Returns:
 
     * Generator
-    """
+    '''
 
     # Save the current state of the global backend override (if any)
     original_value = get_global_forced_attn_backend()
diff --git a/vllm/config.py b/vllm/config.py
@@ -2724,6 +2724,13 @@ def __post_init__(self):
                             f"num_speculative_tokens:{self.num_speculative_tokens}"
                             f" must be divisible by {n_predict=}")
 
+                if self.speculative_token_tree is None:
+                    # Generate chain of tokens.
+                    self.speculative_token_tree = str([[
+                        (i + 1) * (0, )
+                        for i in range(self.num_speculative_tokens)
+                    ]])
+
                 self.draft_tensor_parallel_size = \
                     SpeculativeConfig._verify_and_get_draft_tp(
                         self.target_parallel_config,
diff --git a/vllm/engine/arg_utils.py b/vllm/engine/arg_utils.py
@@ -1427,7 +1427,6 @@ def _is_v1_supported_oracle(self, model_config: ModelConfig) -> bool:
                                    recommend_to_remove=False)
                 return False
 
-        # No XFormers so far.
         V1_BACKENDS = [
             "FLASH_ATTN_VLLM_V1",
             "FLASH_ATTN",
diff --git a/vllm/model_executor/models/llama.py b/vllm/model_executor/models/llama.py
@@ -113,7 +113,6 @@ def __init__(
         cache_config: Optional[CacheConfig] = None,
         prefix: str = "",
         attn_type: str = AttentionType.DECODER,
-        is_draft: bool = False,
     ) -> None:
         super().__init__()
         layer_idx = extract_layer_index(prefix)
@@ -191,7 +190,6 @@ def __init__(
             per_layer_sliding_window=sliding_window,
             attn_type=attn_type,
             prefix=f"{prefix}.attn",
-            is_draft=is_draft,
         )
 
     def forward(
@@ -233,7 +231,6 @@ def __init__(
         cache_config: Optional[CacheConfig] = None,
         quant_config: Optional[QuantizationConfig] = None,
         prefix: str = "",
-        is_draft: bool = False,
     ) -> None:
         super().__init__()
         self.hidden_size = config.hidden_size
@@ -278,7 +275,6 @@ def __init__(
             cache_config=cache_config,
             prefix=f"{prefix}.self_attn",
             attn_type=attn_type,
-            is_draft=is_draft,
         )
         self.mlp = LlamaMLP(
             hidden_size=self.hidden_size,
diff --git a/vllm/model_executor/models/llama_eagle.py b/vllm/model_executor/models/llama_eagle.py
@@ -31,7 +31,7 @@ def __init__(
         disable_input_layernorm: bool,
         prefix: str = "",
     ) -> None:
-        super().__init__(config, prefix=prefix, is_draft=True)
+        super().__init__(config, prefix=prefix)
 
         # Skip the input_layernorm
         # https://github.com/SafeAILab/EAGLE/blob/35c78f6cdc19a73e05cf5c330b4c358dad970c6a/eagle/model/cnets.py#L427
diff --git a/vllm/v1/attention/backends/flash_attn.py b/vllm/v1/attention/backends/flash_attn.py
@@ -134,7 +134,7 @@ def _get_sliding_window_configs(
     sliding_window_configs: set[Optional[tuple[int, int]]] = set()
     layers = get_layers_from_vllm_config(vllm_config, Attention)
     for layer in layers.values():
-        assert hasattr(layer.impl, "sliding_window")
+        assert isinstance(layer.impl, FlashAttentionImpl)
         sliding_window_configs.add(layer.impl.sliding_window)
     return sliding_window_configs
 
diff --git a/vllm/v1/attention/backends/tree_attn.py b/vllm/v1/attention/backends/tree_attn.py
diff --git a/vllm/v1/attention/backends/utils.py b/vllm/v1/attention/backends/utils.py
diff --git a/vllm/v1/spec_decode/eagle.py b/vllm/v1/spec_decode/eagle.py