Fix o_a_proj weight loading: add BF16 fallback for grouped linear
This commit is contained in:
@@ -102,6 +102,10 @@ def main():
|
||||
wo_a.load_nvfp4_weight(oa_w.to(dev), oa_ws.to(dev),
|
||||
oa_ws2.to(dev) if oa_ws2 is not None else None,
|
||||
oa_isc.to(dev) if oa_isc is not None else None)
|
||||
else:
|
||||
oa_bf = all_w.get(f"{pfx}.o_a_proj.weight")
|
||||
if oa_bf is not None:
|
||||
wo_a.set_bf16_weight(oa_bf.bfloat16().to(dev))
|
||||
pl['o_a'] = wo_a; wo_a._use_runtime_gsa = True
|
||||
pl['o_b'] = make_nvfp4_linear(o_groups * o_rank, H, dev, all_w, pfx, 'o_b_proj')
|
||||
prod_lins[li] = pl
|
||||
|
||||
Reference in New Issue
Block a user