[Bugfix] Fix cuda graph sizes when running with speculative decoding (#30330)

PatrykSaffer · Patryk999 · web-flow · commit 4c2e10ea19b9 · 2025-12-10T00:47:07.000Z
Signed-off-by: Patryk Saffer &lt;patryk.saffer99@gmail.com&gt;
Signed-off-by: PatrykSaffer &lt;patryk.saffer@mistral.ai&gt;
Co-authored-by: Patryk Saffer &lt;patryk.saffer99@gmail.com&gt;
diff --git a/vllm/config/vllm.py b/vllm/config/vllm.py
@@ -1047,8 +1047,14 @@ def _set_cudagraph_sizes(self):
                 self.compilation_config.max_cudagraph_capture_size
             )
             if max_cudagraph_capture_size is None:
+                decode_query_len = 1
+                if (
+                    self.speculative_config
+                    and self.speculative_config.num_speculative_tokens
+                ):
+                    decode_query_len += self.speculative_config.num_speculative_tokens
                 max_cudagraph_capture_size = min(
-                    self.scheduler_config.max_num_seqs * 2, 512
+                    self.scheduler_config.max_num_seqs * decode_query_len * 2, 512
                 )
             max_num_tokens = self.scheduler_config.max_num_batched_tokens
             max_cudagraph_capture_size = min(max_num_tokens, max_cudagraph_capture_size)