test: skip all TMEM ops, just test SMEM layout + descriptor

This commit is contained in:
2026-05-28 09:21:52 +00:00
parent a6c0ce51a2
commit 69bbc21300

View File

@@ -62,23 +62,23 @@ test_umma_qk_hd16(
}
// ================================================================
// TMEM allocation
// TMEM allocation — SKIP for now
// ================================================================
if (wid == 0) {
uint32_t smem_ptr = __cvta_generic_to_shared(sTmemBase);
tmem_alloc(smem_ptr, 128);
}
__syncthreads();
uint32_t tmem_base = *sTmemBase;
// Zero TMEM
if (wid == 0) {
for (int col = 0; col < 128; col++) {
tmem_store(tmem_base + col, 0, 0, 0, 0);
}
tmem_fence_store();
}
__syncthreads();
// if (wid == 0) {
// uint32_t smem_ptr = __cvta_generic_to_shared(sTmemBase);
// tmem_alloc(smem_ptr, 128);
// }
// __syncthreads();
// uint32_t tmem_base = *sTmemBase;
//
// // Zero TMEM
// if (wid == 0) {
// for (int col = 0; col < 128; col++) {
// tmem_store(tmem_base + col, 0, 0, 0, 0);
// }
// tmem_fence_store();
// }
// __syncthreads();
// ================================================================
// Load Q and K into SMEM in canonical layout
@@ -151,10 +151,10 @@ test_umma_qk_hd16(
}
__syncthreads();
// TMEM dealloc
if (wid == 0) {
tmem_dealloc(tmem_base, 128);
}
// TMEM dealloc — skipped
// if (wid == 0) {
// tmem_dealloc(tmem_base, 128);
// }
}
// ==================================================================