diff --git a/cutedsl/bridge.py b/cutedsl/bridge.py index 2f607070..06464bc2 100644 --- a/cutedsl/bridge.py +++ b/cutedsl/bridge.py @@ -112,7 +112,58 @@ def quantize_to_nvfp4(x_bf16, block_size=SF_VEC_SIZE): return x_fp4, block_scale, global_scale -def quantize_weight_to_nvfp4(w_bf16, block_size=SF_VEC_SIZE): +def quantize_activation_nvfp4(x_bf16, global_scale, block_size=SF_VEC_SIZE): + """Quantize BF16 activation tensor to NVFP4 (cudagraph-safe). + + Unlike quantize_to_nvfp4(), this takes a pre-computed global_scale + instead of computing it via .max() (which forces CPU-GPU sync). + All operations are pure GPU with no CPU-GPU syncs. + + Args: + x_bf16: (..., D) BF16 tensor + global_scale: float32 scalar (pre-computed, NOT from .max()) + block_size: NVFP4 block size + + Returns: + x_fp4: (..., D//2) float4_e2m1fn_x2 + x_sf: (..., D//16) float8_e4m3fn + """ + x_f32 = x_bf16.float() + x_norm = x_f32 / global_scale + + last_dim = x_norm.shape[-1] + n_blocks = ceil_div(last_dim, block_size) + + if last_dim % block_size != 0: + pad_size = n_blocks * block_size - last_dim + x_norm = torch.nn.functional.pad(x_norm, (0, pad_size)) + + x_reshaped = x_norm.reshape(*x_norm.shape[:-1], n_blocks, block_size) + block_amax = x_reshaped.abs().amax(dim=-1).clamp(min=1e-8) + block_scale = (block_amax / 6.0).to(torch.float8_e4m3fn) + + block_sf_expanded = block_scale.float().unsqueeze(-1) + x_scaled = x_reshaped / block_sf_expanded.clamp(min=1e-8) + signs = torch.sign(x_scaled) + abs_scaled = x_scaled.abs().clamp(max=6.0) + + half_steps = (abs_scaled * 2.0).round().clamp(0, 12).to(torch.int8) + step_to_idx = _get_step_to_idx_lut(x_bf16.device) + indices = step_to_idx[half_steps.long()] + + nibbles = torch.where(signs < 0, indices + 8, indices).to(torch.uint8) + even = nibbles[..., ::2] + odd = nibbles[..., 1::2] + packed = (odd << 4) | even + + packed_shape = list(x_bf16.shape) + packed_shape[-1] = last_dim // 2 + x_fp4 = packed.view(torch.float4_e2m1fn_x2).reshape(packed_shape) + + sf_shape = list(x_bf16.shape[:-1]) + [n_blocks] + block_scale = block_scale.reshape(sf_shape) + + return x_fp4, block_scaledef quantize_weight_to_nvfp4(w_bf16, block_size=SF_VEC_SIZE): """Quantize BF16 weight matrix to NVFP4. The weight is (K, N) where K is the input dim (packed dimension).