[Kernel] W8A16 Int8 inside FusedMoE (#7415)

2024-08-16 20:06:51 +03:00
parent e837b624f2
commit 7fc23be81c
15 changed files with 412 additions and 136 deletions
--- a/vllm/model_executor/layers/quantization/fp8.py
+++ b/vllm/model_executor/layers/quantization/fp8.py
@@ -488,7 +488,7 @@ class Fp8MoEMethod(FusedMoEMethodBase):
                             topk_weights=topk_weights,
                             topk_ids=topk_ids,
                             inplace=True,
-                             use_fp8=True,
+                             use_fp8_w8a8=True,
                             w1_scale=layer.w13_weight_scale,
                             w2_scale=layer.w2_weight_scale,
                             a1_scale=layer.w13_input_scale,