[spec decoding] add tree attention backend selection

TheEpicDolphin · TheEpicDolphin · commit 5a37c7882e31 · 2025-07-02T14:44:59.000-07:00
Signed-off-by: Giancarlo Delfin &lt;gdelfin@meta.com&gt;
diff --git a/vllm/attention/layer.py b/vllm/attention/layer.py
@@ -52,6 +52,7 @@ def __init__(
         prefix: str = "",
         attn_type: str = AttentionType.DECODER,
         kv_sharing_target_layer_name: Optional[str] = None,
+        is_draft: bool = False,
         **extra_impl_args,
     ) -> None:
         """
@@ -135,7 +136,8 @@ def __init__(
                                         block_size,
                                         is_attention_free,
                                         blocksparse_params is not None,
-                                        use_mla=use_mla)
+                                        use_mla=use_mla,
+                                        is_draft=is_draft)
         impl_cls = attn_backend.get_impl_cls()
         self.impl = impl_cls(num_heads, head_size, scale, num_kv_heads,
                              alibi_slopes, sliding_window, kv_cache_dtype,
diff --git a/vllm/attention/selector.py b/vllm/attention/selector.py
@@ -87,6 +87,7 @@ def get_attn_backend(
     is_attention_free: bool,
     is_blocksparse: bool = False,
     use_mla: bool = False,
+    is_draft: bool = False,
 ) -> Type[AttentionBackend]:
     """Selects which attention backend to use and lazily imports it."""
     # Accessing envs.* behind an @lru_cache decorator can cause the wrong
@@ -102,6 +103,7 @@ def get_attn_backend(
         is_blocksparse=is_blocksparse,
         use_v1=envs.VLLM_USE_V1,
         use_mla=use_mla,
+        is_draft=is_draft,
     )
 
 
@@ -115,7 +117,15 @@ def _cached_get_attn_backend(
     is_blocksparse: bool = False,
     use_v1: bool = False,
     use_mla: bool = False,
+    is_draft: bool = False,
 ) -> Type[AttentionBackend]:
+    # TODO(gdelfin): Allow selection of draft model backend for EAGLE. Currently,
+    # it is forced to FlashAttentionBackend to be consistent with EagleProposer.
+    if use_v1 and is_draft:
+        from vllm.v1.attention.backends.flash_attn import FlashAttentionBackend
+
+        return FlashAttentionBackend
+
     if is_blocksparse:
         logger.info("Using BlocksparseFlashAttention backend.")
         from vllm.attention.backends.blocksparse_attn import (
diff --git a/vllm/engine/arg_utils.py b/vllm/engine/arg_utils.py
@@ -1414,6 +1414,7 @@ def _is_v1_supported_oracle(self, model_config: ModelConfig) -> bool:
             "ROCM_AITER_MLA",
             "TORCH_SDPA_VLLM_V1",
             "FLEX_ATTENTION",
+            "TREE_ATTN",
         ]
         if (envs.is_set("VLLM_ATTENTION_BACKEND")
                 and envs.VLLM_ATTENTION_BACKEND not in V1_BACKENDS):
diff --git a/vllm/model_executor/models/llama.py b/vllm/model_executor/models/llama.py
@@ -113,6 +113,7 @@ def __init__(
         cache_config: Optional[CacheConfig] = None,
         prefix: str = "",
         attn_type: str = AttentionType.DECODER,
+        is_draft: bool = False,
     ) -> None:
         super().__init__()
         layer_idx = extract_layer_index(prefix)
@@ -190,6 +191,7 @@ def __init__(
             per_layer_sliding_window=sliding_window,
             attn_type=attn_type,
             prefix=f"{prefix}.attn",
+            is_draft=is_draft,
         )
 
     def forward(
@@ -231,6 +233,7 @@ def __init__(
         cache_config: Optional[CacheConfig] = None,
         quant_config: Optional[QuantizationConfig] = None,
         prefix: str = "",
+        is_draft: bool = False,
     ) -> None:
         super().__init__()
         self.hidden_size = config.hidden_size
@@ -275,6 +278,7 @@ def __init__(
             cache_config=cache_config,
             prefix=f"{prefix}.self_attn",
             attn_type=attn_type,
+            is_draft=is_draft,
         )
         self.mlp = LlamaMLP(
             hidden_size=self.hidden_size,
diff --git a/vllm/model_executor/models/llama_eagle.py b/vllm/model_executor/models/llama_eagle.py
@@ -31,7 +31,7 @@ def __init__(
         disable_input_layernorm: bool,
         prefix: str = "",
     ) -> None:
-        super().__init__(config, prefix=prefix)
+        super().__init__(config, prefix=prefix, is_draft=True)
 
         # Skip the input_layernorm
         # https://github.com/SafeAILab/EAGLE/blob/35c78f6cdc19a73e05cf5c330b4c358dad970c6a/eagle/model/cnets.py#L427
diff --git a/vllm/platforms/cuda.py b/vllm/platforms/cuda.py
@@ -248,6 +248,10 @@ def get_attn_backend_cls(cls, selected_backend, head_size, dtype,
                 logger.info_once("Using Flash Attention backend on V1 engine.")
                 return ("vllm.v1.attention.backends."
                         "flash_attn.FlashAttentionBackend")
+            elif selected_backend == _Backend.TREE_ATTN:
+                logger.info_once("Using Tree Attention backend on V1 engine.")
+                return ("vllm.v1.attention.backends."
+                        "tree_attn.TreeAttentionBackend")
 
             # Default backends for V1 engine
             # Prefer FlashInfer for Blackwell GPUs if installed
diff --git a/vllm/platforms/interface.py b/vllm/platforms/interface.py
@@ -61,6 +61,7 @@ class _Backend(enum.Enum):
     DUAL_CHUNK_FLASH_ATTN = enum.auto()
     NO_ATTENTION = enum.auto()
     FLEX_ATTENTION = enum.auto()
+    TREE_ATTN = enum.auto()
 
 
 class PlatformEnum(enum.Enum):
diff --git a/vllm/v1/attention/backends/flash_attn.py b/vllm/v1/attention/backends/flash_attn.py
@@ -132,7 +132,7 @@ def _get_sliding_window_configs(
     sliding_window_configs: set[Optional[tuple[int, int]]] = set()
     layers = get_layers_from_vllm_config(vllm_config, Attention)
     for layer in layers.values():
-        assert isinstance(layer.impl, FlashAttentionImpl)
+        assert hasattr(layer.impl, "sliding_window")
         sliding_window_configs.add(layer.impl.sliding_window)
     return sliding_window_configs
 
diff --git a/vllm/v1/attention/backends/tree_attn.py b/vllm/v1/attention/backends/tree_attn.py
@@ -380,6 +380,7 @@ def __init__(
             None,  # Skip KV reshape and cache. This class handles it.
             use_irope=use_irope,
         )
+        self.sliding_window = self.prefill_attention_impl.sliding_window
 
     def forward(
         self,

Original file line number	Diff line number	Diff line change
`@@ -1414,6 +1414,7 @@ def _is_v1_supported_oracle(self, model_config: ModelConfig) -> bool:`
`1414`	`1414`	`"ROCM_AITER_MLA",`
`1415`	`1415`	`"TORCH_SDPA_VLLM_V1",`
`1416`	`1416`	`"FLEX_ATTENTION",`
	`1417`	`+ "TREE_ATTN",`
`1417`	`1418`	`]`
`1418`	`1419`	`if (envs.is_set("VLLM_ATTENTION_BACKEND")`
`1419`	`1420`	`and envs.VLLM_ATTENTION_BACKEND not in V1_BACKENDS):`
Original file line number	Diff line number	Diff line change
`@@ -380,6 +380,7 @@ def __init__(`
`380`	`380`	`None, # Skip KV reshape and cache. This class handles it.`
`381`	`381`	`use_irope=use_irope,`
`382`	`382`	`)`
	`383`	`+ self.sliding_window = self.prefill_attention_impl.sliding_window`
`383`	`384`
`384`	`385`	`def forward(`
`385`	`386`	`self,`