fix: pass slot_expert_ids directly to L2 instead of rebuilding from topk_ids

The L2 function was rebuilding slot_expert_ids by scanning topk_ids with a
local_mask. This produced mismatched slot_k (all-expert mask) vs slot_token
(rank-local mask), causing vectorized_gather_kernel index out of bounds.

Now slot_expert_local is passed directly from the outer routing logic, matching
the same slot ordering as L1.
This commit is contained in:
2026-05-15 09:42:13 +00:00
parent f29b96de09
commit c7db2242ee

View File

@@ -164,15 +164,14 @@ def nvfp4_mega_moe_l2(
x_sf, # (num_slots, sf_k_groups) float8_e4m3fn
l2_weights, # (E_per_rank, INTER//2, HIDDEN) int8, column-major for CUTLASS
l2_scales, # (E_per_rank, sf_k_groups, HIDDEN) float8_e4m3fn, column-major — or prepacked
topk_ids, # (num_tokens, NUM_TOPK) int32 — local expert IDs (for slot mapping)
slot_expert_ids, # (num_slots,) int32 — per-slot local expert IDs (from L1 routing)
slot_token, # (num_slots,) int64 — token index per slot (from L1)
alpha=1.0, # fp32 scalar from stage_activation global scale
sfb_prepacked=False, # True if l2_scales is prepacked CUTLASS layout
):
"""L2 GEMM: down_proj — slot-based, no routing weights.
Reuses the same slot mapping from L1 (same slot_token indices).
topk_ids is passed to rebuild the slot→expert mapping.
Reuses the same slot mapping from L1 (same slot_token and slot_expert_ids).
"""
K_half = x_fp4.shape[1]
K = K_half * 2
@@ -187,11 +186,7 @@ def nvfp4_mega_moe_l2(
else:
w_sf_fp8 = l2_scales # already prepacked
# Build 1D per-slot expert IDs (same slot ordering as L1)
num_experts = l2_weights.shape[0]
local_mask = (topk_ids >= 0) & (topk_ids < num_experts)
_, slot_k = local_mask.nonzero(as_tuple=True)
slot_expert_ids = topk_ids[slot_token, slot_k] # (num_slots,)
# slot_expert_ids passed directly from L1 routing — no rebuild needed
slot_out, _ = cutlass_grouped_nvfp4_gemm(
x_fp4, x_sf_fp8,
@@ -405,7 +400,7 @@ def nvfp4_mega_moe_full(
# Step 5: L2 GEMM — slot-based, no routing weights, prepacked SFB
l2_slots = nvfp4_mega_moe_l2(
l1_fp4, l1_sf_out, l2_w, l2_sf_prepacked,
topk_ids_local, slot_token,
slot_expert_local, slot_token,
alpha=l2_alpha,
sfb_prepacked=True,
) # (num_slots, HIDDEN) bfloat16