From b314fde9b7ac5b5888c7de7157b509b48cea7c28 Mon Sep 17 00:00:00 2001 From: biondizzle Date: Thu, 4 Jun 2026 00:30:21 +0000 Subject: [PATCH] Fix gsa copy_ cudaErrorInvalidValue: replace view-based copy_ with scalar assignment MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The pattern causes cudaErrorInvalidValue when gsa_gpu is a non-contiguous expanded view (e.g., shape (9,) from quantize_nvfp4_gpu_fused during prefill with M>1). Root cause: copy_() from an expanded/reshaped view can fail when the source tensor has non-standard strides. The expand() operation creates a view with stride-0 dimensions that copy_() may not handle correctly on all CUDA versions. Fix: Replace all gsa copy_ patterns with scalar assignment: self._gsa_buf[0] = gsa_gpu[0] # scalar GPU→GPU, graph-capturable This is simpler, avoids view issues, and is CUDA-graph-compatible. Applied to: shared_expert.py, moe.py, linear.py, grouped_linear.py --- dsv4/layers/grouped_linear.py | 6 ++++-- dsv4/layers/linear.py | 9 +++------ dsv4/layers/moe.py | 6 +++--- dsv4/layers/shared_expert.py | 12 ++++++------ 4 files changed, 16 insertions(+), 17 deletions(-) diff --git a/dsv4/layers/grouped_linear.py b/dsv4/layers/grouped_linear.py index 30c8a555..182308a5 100644 --- a/dsv4/layers/grouped_linear.py +++ b/dsv4/layers/grouped_linear.py @@ -338,10 +338,12 @@ class Nvfp4GroupedLinear: # gsa_gpu is (G*T,) — all rows share same amax (from max over full tensor) # For the GEMM's global_scale_a, fill all group slots with the same gsa value # Use GPU-only copy: no .item(), no CPU sync - self._gsa_buf[:1].copy_(gsa_gpu[:1]) # GPU→GPU scalar copy, no sync + self._gsa_buf[0] = gsa_gpu[0] # scalar GPU→GPU, no sync, graph-capturable # Broadcast to all groups (all get same gsa) + # Use scalar broadcast assignment instead of copy_ from expanded view + # (expanded views can cause cudaErrorInvalidValue in copy_) if self.n_local_groups > 1: - self._gsa_buf[1:].copy_(self._gsa_buf[:1].expand(self.n_local_groups - 1)) + self._gsa_buf[1:] = self._gsa_buf[0] # scalar broadcast, graph-capturable else: self._gsa_buf.fill_(self._activation_global_scale) x_fp4_flat, x_sf_flat = quantize_activation_nvfp4( diff --git a/dsv4/layers/linear.py b/dsv4/layers/linear.py index 32b53473..82f12aaa 100644 --- a/dsv4/layers/linear.py +++ b/dsv4/layers/linear.py @@ -206,7 +206,7 @@ class Nvfp4Linear: if getattr(self, '_use_runtime_gsa', False): from dsv4.ops.quantize import quantize_nvfp4_gpu_fused x_fp4, x_sf, gsa_gpu = quantize_nvfp4_gpu_fused(hidden_states) - self._gsa_buf.copy_(gsa_gpu[:1].reshape(1)) # GPU → GPU, no sync + self._gsa_buf[0] = gsa_gpu[0] # scalar GPU→GPU, no sync, graph-capturable else: # P2 FIX: No per-call fill_(). The _gsa_buf already has the correct # value — set either during initialization (via _ensure_buffer_size) @@ -284,13 +284,10 @@ class Nvfp4Linear: # For M=1 decode: per-row gsa is already scalar, no reduction needed. # For M>1 prefill: reduce per-row gsa to a single scalar (max). if quant.gsa.shape[0] == 1: - gsa = quant.gsa[:1].reshape(1) # Already scalar + self._gsa_buf[0] = quant.gsa[0] # scalar GPU→GPU, graph-capturable else: # Reduce per-row gsa to scalar (max) for GEMM compatibility. - # Per-row gsa is mathematically more precise, but the GEMM only - # supports a single global scale per expert. - gsa = quant.gsa.max().reshape(1) - self._gsa_buf.copy_(gsa) + self._gsa_buf[0] = quant.gsa.max() # GPU max, scalar assign, graph-capturable # Run GEMM out = run_nvfp4_grouped_gemm( diff --git a/dsv4/layers/moe.py b/dsv4/layers/moe.py index 1744e767..0dc0e89e 100644 --- a/dsv4/layers/moe.py +++ b/dsv4/layers/moe.py @@ -630,7 +630,7 @@ class Nvfp4MoE: if getattr(self, '_use_runtime_gsa', False): from dsv4.ops.quantize import quantize_nvfp4_gpu_fused slot_x_fp4, slot_x_sf, gsa_l1_gpu = quantize_nvfp4_gpu_fused(slot_hidden) - self._l1_gsa_buf.copy_(gsa_l1_gpu[:1].reshape(1)) # GPU → GPU, no sync + self._l1_gsa_buf[0] = gsa_l1_gpu[0] # scalar GPU→GPU, no sync, graph-capturable else: slot_x_fp4, slot_x_sf = quantize_nvfp4_gpu( slot_hidden, self._l1_activation_global_scale @@ -666,7 +666,7 @@ class Nvfp4MoE: from dsv4.ops.quantize import deinterleave_amax_quantize_nvfp4_fused slot_l2_x_fp4, slot_l2_x_sf, gsa_l2_gpu = deinterleave_amax_quantize_nvfp4_fused( l1_out_real, self.intermediate_size) - self._l2_gsa_buf.copy_(gsa_l2_gpu[:1].reshape(1)) # GPU → GPU, no sync + self._l2_gsa_buf[0] = gsa_l2_gpu[0] # scalar GPU→GPU, no sync, graph-capturable else: slot_l2_x_fp4, slot_l2_x_sf = deinterleave_quantize_nvfp4_cuda( l1_out_real, self.intermediate_size, self._l2_activation_global_scale @@ -694,7 +694,7 @@ class Nvfp4MoE: if not self._fused_swiglu and getattr(self, '_use_runtime_gsa', False): from dsv4.ops.quantize import quantize_nvfp4_gpu_fused slot_l2_x_fp4, slot_l2_x_sf, gsa_l2_gpu = quantize_nvfp4_gpu_fused(activated) - self._l2_gsa_buf.copy_(gsa_l2_gpu[:1].reshape(1)) # GPU → GPU, no sync + self._l2_gsa_buf[0] = gsa_l2_gpu[0] # scalar GPU→GPU, no sync, graph-capturable elif not self._fused_swiglu: slot_l2_x_fp4, slot_l2_x_sf = quantize_nvfp4_gpu( activated, self._l2_activation_global_scale diff --git a/dsv4/layers/shared_expert.py b/dsv4/layers/shared_expert.py index e07e921a..a8b5344c 100644 --- a/dsv4/layers/shared_expert.py +++ b/dsv4/layers/shared_expert.py @@ -268,7 +268,7 @@ class Nvfp4SharedExpert: if getattr(self, '_use_runtime_gsa', False): from dsv4.ops.quantize import quantize_nvfp4_gpu_fused x_fp4, x_sf, gsa_l1_gpu = quantize_nvfp4_gpu_fused(x_bf16) - self._l1_gsa_buf.copy_(gsa_l1_gpu[:1].reshape(1)) # GPU → GPU + self._l1_gsa_buf[0] = gsa_l1_gpu[0] # scalar GPU→GPU, no sync, graph-capturable else: from dsv4.ops.quantize import quantize_activation_nvfp4 x_fp4, x_sf = quantize_activation_nvfp4(x_bf16, self._l1_activation_global_scale) @@ -316,7 +316,7 @@ class Nvfp4SharedExpert: if getattr(self, '_use_runtime_gsa', False): from dsv4.ops.quantize import quantize_nvfp4_gpu_fused x_fp4, x_sf, gsa_l1_gpu = quantize_nvfp4_gpu_fused(hidden_states) - self._l1_gsa_buf.copy_(gsa_l1_gpu[:1].reshape(1)) # GPU → GPU, no sync + self._l1_gsa_buf[0] = gsa_l1_gpu[0] # scalar GPU→GPU, no sync, graph-capturable else: x_fp4, x_sf = quantize_activation_nvfp4( hidden_states, self._l1_activation_global_scale @@ -366,10 +366,10 @@ class Nvfp4SharedExpert: if not intermediate.is_contiguous(): intermediate = intermediate.contiguous() x_fp4, x_sf, gsa_l2_gpu = quantize_nvfp4_gpu_fused(intermediate) - # DEBUG: verify no CUDA errors from quantize kernel - torch.cuda.synchronize() # DEBUG: catch async errors - print(f" SE L2 gsa: gsa_gpu shape={tuple(gsa_l2_gpu.shape)} dtype={gsa_l2_gpu.dtype} dev={gsa_l2_gpu.device} _l2_gsa_buf shape={tuple(self._l2_gsa_buf.shape)} dev={self._l2_gsa_buf.device}", flush=True) - self._l2_gsa_buf.copy_(gsa_l2_gpu[:1].reshape(1)) # GPU → GPU, no sync + # Copy first element of gsa (scalar for single-expert) to pre-allocated buffer. + # Using scalar assignment avoids copy_() from view which caused cudaErrorInvalidValue + # on non-contiguous gsa_gpu slices (gsa_gpu[:1].reshape(1) — view of expanded tensor). + self._l2_gsa_buf[0] = gsa_l2_gpu[0] # scalar GPU → GPU, no sync, graph-capturable else: x_fp4, x_sf = quantize_activation_nvfp4( intermediate, self._l2_activation_global_scale