diff --git a/tests/unit/test_tma_debug.cu b/tests/unit/test_tma_debug.cu new file mode 100644 index 00000000..fb32c62e --- /dev/null +++ b/tests/unit/test_tma_debug.cu @@ -0,0 +1,58 @@ +#include +#include +#include + +typedef unsigned short bf16_t; + +int main() { + printf("=== CUtensorMap parameter debug ===\n"); + bf16_t* d_data; + cudaMalloc(&d_data, 128*16*2); + + // Test 1: BF16 2D (16, 128) with byte strides + { + uint64_t gdim[] = {16, 128}; + uint64_t gstr[] = {2, 32}; + uint32_t tdim[] = {16, 128}; + uint32_t tstr[] = {1, 16}; + CUtensorMap tma; + CUresult r = cuTensorMapEncodeTiled(&tma, CU_TENSOR_MAP_DATA_TYPE_BFLOAT16, 2, d_data, gdim, gstr, tdim, tstr, CU_TENSOR_MAP_INTERLEAVE_NONE, CU_TENSOR_MAP_SWIZZLE_NONE, CU_TENSOR_MAP_L2_PROMOTION_NONE, CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE); + printf("BF16 (16,128) gstr=[2,32]: result=%d\n", r); + } + + // Test 2: UINT16 2D + { + uint64_t gdim[] = {16, 128}; + uint64_t gstr[] = {2, 32}; + uint32_t tdim[] = {16, 128}; + uint32_t tstr[] = {1, 16}; + CUtensorMap tma; + CUresult r = cuTensorMapEncodeTiled(&tma, CU_TENSOR_MAP_DATA_TYPE_UINT16, 2, d_data, gdim, gstr, tdim, tstr, CU_TENSOR_MAP_INTERLEAVE_NONE, CU_TENSOR_MAP_SWIZZLE_NONE, CU_TENSOR_MAP_L2_PROMOTION_NONE, CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE); + printf("UINT16 (16,128) gstr=[2,32]: result=%d\n", r); + } + + // Test 3: Smaller tile + { + uint64_t gdim[] = {16, 128}; + uint64_t gstr[] = {2, 32}; + uint32_t tdim[] = {16, 8}; // 8 rows per tile + uint32_t tstr[] = {1, 16}; + CUtensorMap tma; + CUresult r = cuTensorMapEncodeTiled(&tma, CU_TENSOR_MAP_DATA_TYPE_BFLOAT16, 2, d_data, gdim, gstr, tdim, tstr, CU_TENSOR_MAP_INTERLEAVE_NONE, CU_TENSOR_MAP_SWIZZLE_NONE, CU_TENSOR_MAP_L2_PROMOTION_NONE, CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE); + printf("BF16 (16,8) tile: result=%d\n", r); + } + + // Test 4: UINT8 1D + { + uint64_t gdim[] = {4096}; + uint64_t gstr[] = {2}; + uint32_t tdim[] = {256}; + uint32_t tstr[] = {1}; + CUtensorMap tma; + CUresult r = cuTensorMapEncodeTiled(&tma, CU_TENSOR_MAP_DATA_TYPE_UINT8, 1, d_data, gdim, gstr, tdim, tstr, CU_TENSOR_MAP_INTERLEAVE_NONE, CU_TENSOR_MAP_SWIZZLE_NONE, CU_TENSOR_MAP_L2_PROMOTION_NONE, CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE); + printf("UINT8 1D: result=%d\n", r); + } + + cudaFree(d_data); + return 0; +}