[spec decoding] add tree attention backend selection

TheEpicDolphin · TheEpicDolphin · commit 3ff7ebeb983b · 2025-07-02T16:00:33.000-07:00
Signed-off-by: Giancarlo Delfin &lt;gdelfin@meta.com&gt;
diff --git a/vllm/attention/layer.py b/vllm/attention/layer.py
@@ -52,6 +52,7 @@ def __init__(
         prefix: str = "",
         attn_type: str = AttentionType.DECODER,
         kv_sharing_target_layer_name: Optional[str] = None,
+        is_draft: bool = False,
         **extra_impl_args,
     ) -> None:
         """
@@ -135,7 +136,8 @@ def __init__(
                                         block_size,
                                         is_attention_free,
                                         blocksparse_params is not None,
-                                        use_mla=use_mla)
+                                        use_mla=use_mla,
+                                        is_draft=is_draft)
         impl_cls = attn_backend.get_impl_cls()
         self.impl = impl_cls(num_heads, head_size, scale, num_kv_heads,
                              alibi_slopes, sliding_window, kv_cache_dtype,
diff --git a/vllm/attention/selector.py b/vllm/attention/selector.py
@@ -27,23 +27,22 @@ def backend_name_to_enum(backend_name: str) -> Optional[_Backend]:
             loaded.
     """
     assert backend_name is not None
-    return _Backend[backend_name] if backend_name in _Backend.__members__ else \
-          None
+    return _Backend[
+        backend_name] if backend_name in _Backend.__members__ else None
 
 
 def get_env_variable_attn_backend() -> Optional[_Backend]:
-    '''
+    """
     Get the backend override specified by the vLLM attention
     backend environment variable, if one is specified.
 
     Returns:
 
     * _Backend enum value if an override is specified
     * None otherwise
-    '''
+    """
     backend_name = os.environ.get(STR_BACKEND_ENV_VAR)
-    return (None
-            if backend_name is None else backend_name_to_enum(backend_name))
+    return None if backend_name is None else backend_name_to_enum(backend_name)
 
 
 # Global state allows a particular choice of backend
@@ -57,7 +56,7 @@ def get_env_variable_attn_backend() -> Optional[_Backend]:
 
 
 def global_force_attn_backend(attn_backend: Optional[_Backend]) -> None:
-    '''
+    """
     Force all attention operations to use a specified backend.
 
     Passing `None` for the argument re-enables automatic
@@ -66,16 +65,16 @@ def global_force_attn_backend(attn_backend: Optional[_Backend]) -> None:
     Arguments:
 
     * attn_backend: backend selection (None to revert to auto)
-    '''
+    """
     global forced_attn_backend
     forced_attn_backend = attn_backend
 
 
 def get_global_forced_attn_backend() -> Optional[_Backend]:
-    '''
+    """
     Get the currently-forced choice of attention backend,
     or None if auto-selection is currently enabled.
-    '''
+    """
     return forced_attn_backend
 
 
@@ -87,6 +86,7 @@ def get_attn_backend(
     is_attention_free: bool,
     is_blocksparse: bool = False,
     use_mla: bool = False,
+    is_draft: bool = False,
 ) -> Type[AttentionBackend]:
     """Selects which attention backend to use and lazily imports it."""
     # Accessing envs.* behind an @lru_cache decorator can cause the wrong
@@ -102,6 +102,7 @@ def get_attn_backend(
         is_blocksparse=is_blocksparse,
         use_v1=envs.VLLM_USE_V1,
         use_mla=use_mla,
+        is_draft=is_draft,
     )
 
 
@@ -115,18 +116,28 @@ def _cached_get_attn_backend(
     is_blocksparse: bool = False,
     use_v1: bool = False,
     use_mla: bool = False,
+    is_draft: bool = False,
 ) -> Type[AttentionBackend]:
+    # Draft model backend is currently forced to FlashAttentionBackend for
+    # consistency with EagleProposer using FlashAttentionMetadata.
+    if use_v1 and is_draft:
+        from vllm.v1.attention.backends.flash_attn import FlashAttentionBackend
+
+        return FlashAttentionBackend
+
     if is_blocksparse:
         logger.info("Using BlocksparseFlashAttention backend.")
         from vllm.attention.backends.blocksparse_attn import (
             BlocksparseFlashAttentionBackend)
+
         return BlocksparseFlashAttentionBackend
 
     # If there are no attention layers (e.g. we are running Mamba),
     # use the placeholder NO_ATTENTION
     if is_attention_free:
         from vllm.attention.backends.placeholder_attn import (
             PlaceholderAttentionBackend)
+
         return PlaceholderAttentionBackend
 
     # Check whether a particular choice of backend was
@@ -135,8 +146,8 @@ def _cached_get_attn_backend(
     # THIS SELECTION OVERRIDES THE VLLM_ATTENTION_BACKEND
     # ENVIRONMENT VARIABLE.
     selected_backend = None
-    backend_by_global_setting: Optional[_Backend] = (
-        get_global_forced_attn_backend())
+    backend_by_global_setting: Optional[
+        _Backend] = get_global_forced_attn_backend()
     if backend_by_global_setting is not None:
         selected_backend = backend_by_global_setting
     else:
@@ -157,8 +168,8 @@ def _cached_get_attn_backend(
 
 @contextmanager
 def global_force_attn_backend_context_manager(
-        attn_backend: _Backend) -> Generator[None, None, None]:
-    '''
+    attn_backend: _Backend, ) -> Generator[None, None, None]:
+    """
     Globally force a vLLM attention backend override within a
     context manager, reverting the global attention backend
     override to its prior state upon exiting the context
@@ -171,7 +182,7 @@ def global_force_attn_backend_context_manager(
     Returns:
 
     * Generator
-    '''
+    """
 
     # Save the current state of the global backend override (if any)
     original_value = get_global_forced_attn_backend()
diff --git a/vllm/engine/arg_utils.py b/vllm/engine/arg_utils.py
@@ -1414,6 +1414,7 @@ def _is_v1_supported_oracle(self, model_config: ModelConfig) -> bool:
             "ROCM_AITER_MLA",
             "TORCH_SDPA_VLLM_V1",
             "FLEX_ATTENTION",
+            "TREE_ATTN",
         ]
         if (envs.is_set("VLLM_ATTENTION_BACKEND")
                 and envs.VLLM_ATTENTION_BACKEND not in V1_BACKENDS):
diff --git a/vllm/model_executor/models/llama.py b/vllm/model_executor/models/llama.py
@@ -113,6 +113,7 @@ def __init__(
         cache_config: Optional[CacheConfig] = None,
         prefix: str = "",
         attn_type: str = AttentionType.DECODER,
+        is_draft: bool = False,
     ) -> None:
         super().__init__()
         layer_idx = extract_layer_index(prefix)
@@ -190,6 +191,7 @@ def __init__(
             per_layer_sliding_window=sliding_window,
             attn_type=attn_type,
             prefix=f"{prefix}.attn",
+            is_draft=is_draft,
         )
 
     def forward(
@@ -231,6 +233,7 @@ def __init__(
         cache_config: Optional[CacheConfig] = None,
         quant_config: Optional[QuantizationConfig] = None,
         prefix: str = "",
+        is_draft: bool = False,
     ) -> None:
         super().__init__()
         self.hidden_size = config.hidden_size
@@ -275,6 +278,7 @@ def __init__(
             cache_config=cache_config,
             prefix=f"{prefix}.self_attn",
             attn_type=attn_type,
+            is_draft=is_draft,
         )
         self.mlp = LlamaMLP(
             hidden_size=self.hidden_size,
diff --git a/vllm/model_executor/models/llama_eagle.py b/vllm/model_executor/models/llama_eagle.py
@@ -31,7 +31,7 @@ def __init__(
         disable_input_layernorm: bool,
         prefix: str = "",
     ) -> None:
-        super().__init__(config, prefix=prefix)
+        super().__init__(config, prefix=prefix, is_draft=True)
 
         # Skip the input_layernorm
         # https://github.com/SafeAILab/EAGLE/blob/35c78f6cdc19a73e05cf5c330b4c358dad970c6a/eagle/model/cnets.py#L427
diff --git a/vllm/platforms/cuda.py b/vllm/platforms/cuda.py
@@ -248,6 +248,10 @@ def get_attn_backend_cls(cls, selected_backend, head_size, dtype,
                 logger.info_once("Using Flash Attention backend on V1 engine.")
                 return ("vllm.v1.attention.backends."
                         "flash_attn.FlashAttentionBackend")
+            elif selected_backend == _Backend.TREE_ATTN:
+                logger.info_once("Using Tree Attention backend on V1 engine.")
+                return ("vllm.v1.attention.backends."
+                        "tree_attn.TreeAttentionBackend")
 
             # Default backends for V1 engine
             # Prefer FlashInfer for Blackwell GPUs if installed
diff --git a/vllm/platforms/interface.py b/vllm/platforms/interface.py
@@ -61,6 +61,7 @@ class _Backend(enum.Enum):
     DUAL_CHUNK_FLASH_ATTN = enum.auto()
     NO_ATTENTION = enum.auto()
     FLEX_ATTENTION = enum.auto()
+    TREE_ATTN = enum.auto()
 
 
 class PlatformEnum(enum.Enum):
diff --git a/vllm/v1/attention/backends/flash_attn.py b/vllm/v1/attention/backends/flash_attn.py
@@ -132,7 +132,7 @@ def _get_sliding_window_configs(
     sliding_window_configs: set[Optional[tuple[int, int]]] = set()
     layers = get_layers_from_vllm_config(vllm_config, Attention)
     for layer in layers.values():
-        assert isinstance(layer.impl, FlashAttentionImpl)
+        assert hasattr(layer.impl, "sliding_window")
         sliding_window_configs.add(layer.impl.sliding_window)
     return sliding_window_configs
 
diff --git a/vllm/v1/attention/backends/tree_attn.py b/vllm/v1/attention/backends/tree_attn.py
@@ -380,6 +380,7 @@ def __init__(
             None,  # Skip KV reshape and cache. This class handles it.
             use_irope=use_irope,
         )
+        self.sliding_window = self.prefill_attention_impl.sliding_window
 
     def forward(
         self,

Original file line number	Diff line number	Diff line change
`@@ -1414,6 +1414,7 @@ def _is_v1_supported_oracle(self, model_config: ModelConfig) -> bool:`
`1414`	`1414`	`"ROCM_AITER_MLA",`
`1415`	`1415`	`"TORCH_SDPA_VLLM_V1",`
`1416`	`1416`	`"FLEX_ATTENTION",`
	`1417`	`+ "TREE_ATTN",`
`1417`	`1418`	`]`
`1418`	`1419`	`if (envs.is_set("VLLM_ATTENTION_BACKEND")`
`1419`	`1420`	`and envs.VLLM_ATTENTION_BACKEND not in V1_BACKENDS):`
Original file line number	Diff line number	Diff line change
`@@ -380,6 +380,7 @@ def __init__(`
`380`	`380`	`None, # Skip KV reshape and cache. This class handles it.`
`381`	`381`	`use_irope=use_irope,`
`382`	`382`	`)`
	`383`	`+ self.sliding_window = self.prefill_attention_impl.sliding_window`
`383`	`384`
`384`	`385`	`def forward(`
`385`	`386`	`self,`