[ROCm] [V1] [SpecDec] Enable Speculative Decoding on ROCm V1 Engine (#21496)

Signed-off-by: tjtanaa <tunjian.tan@embeddedllm.com>
2025-08-07 19:13:17 -07:00
parent acf8aeb79e
commit 1ee5ead5f8
6 changed files with 128 additions and 41 deletions
--- a/tests/v1/spec_decode/test_max_len.py
+++ b/tests/v1/spec_decode/test_max_len.py
@@ -4,7 +4,9 @@

 import pytest

+from tests.utils import get_attn_backend_list_based_on_platform
 from vllm import LLM, SamplingParams
+from vllm.platforms import current_platform

 _PROMPTS = [
    "1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1",
@@ -14,36 +16,40 @@ _PROMPTS = [


@pytest.mark.parametrize("num_speculative_tokens", [1, 3, 10])
-def test_ngram_max_len(
-    monkeypatch: pytest.MonkeyPatch,
-    num_speculative_tokens: int,
-):
-    with monkeypatch.context() as m:
-        m.setenv("VLLM_USE_V1", "1")
-
-        llm = LLM(
-            model="facebook/opt-125m",
-            max_model_len=100,
-            enforce_eager=True,  # For faster initialization.
-            speculative_config={
-                "method": "ngram",
-                "prompt_lookup_max": 5,
-                "prompt_lookup_min": 3,
-                "num_speculative_tokens": num_speculative_tokens,
-            },
-        )
-        sampling_params = SamplingParams(max_tokens=100, ignore_eos=True)
-        llm.generate(_PROMPTS, sampling_params)
+def test_ngram_max_len(num_speculative_tokens: int):
+    llm = LLM(
+        model="facebook/opt-125m",
+        max_model_len=100,
+        enforce_eager=True,  # For faster initialization.
+        speculative_config={
+            "method": "ngram",
+            "prompt_lookup_max": 5,
+            "prompt_lookup_min": 3,
+            "num_speculative_tokens": num_speculative_tokens,
+        },
+    )
+    sampling_params = SamplingParams(max_tokens=100, ignore_eos=True)
+    llm.generate(_PROMPTS, sampling_params)


@pytest.mark.parametrize("num_speculative_tokens", [1, 3, 10])
-def test_eagle_max_len(
-    monkeypatch: pytest.MonkeyPatch,
-    num_speculative_tokens: int,
-):
+@pytest.mark.parametrize("attn_backend",
+                         get_attn_backend_list_based_on_platform())
+def test_eagle_max_len(monkeypatch: pytest.MonkeyPatch,
+                       num_speculative_tokens: int, attn_backend: str):
    with monkeypatch.context() as m:
        m.setenv("VLLM_USE_V1", "1")

+        m.setenv("VLLM_ATTENTION_BACKEND", attn_backend)
+
+        if (attn_backend == "TRITON_ATTN_VLLM_V1"
+                and not current_platform.is_rocm()):
+            pytest.skip("TRITON_ATTN_VLLM_V1 does not support "
+                        "multi-token eagle spec decode on current platform")
+
+        if attn_backend == "FLASH_ATTN_VLLM_V1" and current_platform.is_rocm():
+            m.setenv("VLLM_ROCM_USE_AITER", "1")
+
        llm = LLM(
            model="meta-llama/Meta-Llama-3-8B-Instruct",
            enforce_eager=True,  # For faster initialization.