[Spec-Decode] Support piecewise cudagraphs for Eagle head (#25109)

Signed-off-by: Lucas Wilkinson <lwilkins@redhat.com>
Signed-off-by: Lucas Wilkinson <LucasWilkinson@users.noreply.github.com>
Co-authored-by: Benjamin Chislett <chislett.ben@gmail.com>
This commit is contained in:
Lucas Wilkinson
2025-10-10 01:20:31 -04:00
committed by GitHub
parent da4455609d
commit 29255cfc3b
6 changed files with 84 additions and 16 deletions

View File

@@ -7,6 +7,7 @@ import torch
import torch.nn as nn
from transformers import PretrainedConfig
from vllm.compilation.decorators import support_torch_compile
from vllm.config import VllmConfig
from vllm.model_executor.layers.fused_moe import FusedMoE
from vllm.model_executor.layers.layernorm import RMSNorm
@@ -162,6 +163,7 @@ class DeepSeekMultiTokenPredictor(nn.Module):
return logits
@support_torch_compile
class DeepSeekMTP(nn.Module, SupportsPP):
def __init__(self, *, vllm_config: VllmConfig, prefix: str = ""):
super().__init__()