[Bugfix] fix use_atomic_add support of marlin kernel when using v1 engine (#15946)

Signed-off-by: Jinzhen Lin <linjinzhen@hotmail.com>
2025-04-06 11:04:22 +08:00
parent 13affc432d
commit 2fa66ef713
2 changed files with 6 additions and 2 deletions
--- a/vllm/model_executor/layers/quantization/utils/marlin_utils.py
+++ b/vllm/model_executor/layers/quantization/utils/marlin_utils.py
@@ -305,7 +305,7 @@ def should_use_atomic_add_reduce(m: int, n: int, k: int, device: torch.device,

    # the performance of atomicAdd is better than global reduce
    # only when m*n is small and k is large
-    return max(m, 64) * n < 64 * 2048 and k >= 2048
+    return n < 2048 and k >= 2048


 def apply_gptq_marlin_linear(