[Kernel] moe wna16 marlin kernel (#14447)

Signed-off-by: Jinzhen Lin <linjinzhen@hotmail.com> Co-authored-by: Michael Goin <michael@neuralmagic.com> Co-authored-by: mgoin <mgoin64@gmail.com>
2025-04-15 11:05:22 +08:00
parent 6b40996ae8
commit d06ba4ed3f
16 changed files with 3477 additions and 329 deletions
--- a/vllm/model_executor/layers/fused_moe/layer.py
+++ b/vllm/model_executor/layers/fused_moe/layer.py
@@ -472,6 +472,7 @@ class FusedMoE(torch.nn.Module):
        self.global_num_experts = num_experts

        assert intermediate_size % self.tp_size == 0
+        self.hidden_size = hidden_size
        self.intermediate_size_per_partition = intermediate_size // self.tp_size
        self.reduce_results = reduce_results
        self.renormalize = renormalize