[full_graph] Fix query_start_loc padding (#19321)

yinghai · web-flow · commit 770e5dcdb8c2 · 2025-06-09T21:32:56.000+08:00
Signed-off-by: Yinghai Lu &lt;yinghai@thinkingmachines.ai&gt;
diff --git a/vllm/v1/worker/gpu_model_runner.py b/vllm/v1/worker/gpu_model_runner.py
@@ -655,7 +655,10 @@ def _prepare_inputs(
 
         # Fill unused with -1. Needed for reshape_and_cache
         self.seq_lens[num_reqs:].fill_(0)
-        self.query_start_loc[num_reqs + 1:].fill_(-1)
+        # Note: pad query_start_loc to be non-decreasing, as kernels
+        # like FlashAttention requires that
+        self.query_start_loc[num_reqs + 1:].fill_(
+            self.query_start_loc_cpu[num_reqs].item())
 
         query_start_loc = self.query_start_loc[:num_reqs + 1]
         seq_lens = self.seq_lens[:num_reqs]