update 3.8 v2 (#2112)
* update 3.8 v2 * update 3.8 --------- Co-authored-by: yuzhai <yuzhai@nvidia.com>
This commit is contained in:
@@ -29,7 +29,7 @@
|
||||
#
|
||||
|
||||
#
|
||||
|
||||
if (CUTLASS_NVCC_ARCHS MATCHES 100a)
|
||||
add_custom_target(
|
||||
cutlass_test_unit_gemm_device_sm100_tensorop
|
||||
DEPENDS
|
||||
@@ -38,7 +38,7 @@ add_custom_target(
|
||||
cutlass_test_unit_gemm_device_tensorop_sm100_s8xs8
|
||||
)
|
||||
|
||||
cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
cutlass_test_unit_gemm_device_add_executable(
|
||||
cutlass_test_unit_gemm_device_tensorop_sm100_f16xf16
|
||||
|
||||
BATCH_SOURCES ON
|
||||
@@ -48,7 +48,7 @@ cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
f16_f16_f16_f16_fusion.cu
|
||||
)
|
||||
|
||||
cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
cutlass_test_unit_gemm_device_add_executable(
|
||||
cutlass_test_unit_gemm_device_tensorop_sm100_f8xf8
|
||||
|
||||
BATCH_SOURCES ON
|
||||
@@ -58,7 +58,7 @@ cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
f8_f8_f16_f8_fusion.cu
|
||||
)
|
||||
|
||||
cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
cutlass_test_unit_gemm_device_add_executable(
|
||||
cutlass_test_unit_gemm_device_tensorop_sm100_s8xs8
|
||||
|
||||
BATCH_SOURCES ON
|
||||
@@ -67,5 +67,6 @@ cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
s8_s8_void_s32.cu
|
||||
s8_s8_s32_s32_fusion.cu
|
||||
)
|
||||
endif()
|
||||
|
||||
add_subdirectory(narrow_precision)
|
||||
|
||||
@@ -88,8 +88,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = alpha * acc + beta * C + per-row bias
|
||||
@@ -108,7 +106,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -173,8 +171,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = alpha * acc + beta * C + per-row bias
|
||||
@@ -193,7 +189,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -264,8 +260,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = alpha * acc + beta * C + per-row bias
|
||||
@@ -290,7 +284,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -355,8 +349,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = alpha * acc + beta * C + per-row bias
|
||||
@@ -380,7 +372,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -451,8 +443,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// dY = alpha * acc + beta * C
|
||||
@@ -476,7 +466,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -541,8 +531,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// dY = alpha * acc + beta * C
|
||||
@@ -566,7 +554,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_f16t_f16t_tensor_op_f32, 128x128x64_1x2x1_1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
@@ -82,8 +82,6 @@ TEST(SM100Only_Device_Gemm_f16n_f16t_void_f32n_tensor_op_f32, 64x64x64_4x1x1_1sm
|
||||
using MmaTileShape_MNK = Shape<_64,_64,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_64>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -94,7 +92,7 @@ TEST(SM100Only_Device_Gemm_f16n_f16t_void_f32n_tensor_op_f32, 64x64x64_4x1x1_1sm
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -159,8 +157,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_void_f32t_tensor_op_f32, 64x128x64_1x4x1_1s
|
||||
using MmaTileShape_MNK = Shape<_64,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_64>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -168,7 +164,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_void_f32t_tensor_op_f32, 64x128x64_1x4x1_1s
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -232,8 +228,6 @@ TEST(SM100Only_Device_Gemm_f16n_f16n_void_f32t_tensor_op_f32, 128x64x64_1x8x1_st
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_8,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_64>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -244,7 +238,7 @@ TEST(SM100Only_Device_Gemm_f16n_f16n_void_f32t_tensor_op_f32, 128x64x64_1x8x1_st
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -309,8 +303,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16t_void_f32n_tensor_op_f32, 128x128x64_2x8x1_1
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_8,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -318,7 +310,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16t_void_f32n_tensor_op_f32, 128x128x64_2x8x1_1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -383,8 +375,6 @@ TEST(SM100Only_Device_Gemm_f16n_f16t_void_f32n_tensor_op_f32, 128x64x64_2x4x1_2s
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_64>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -395,7 +385,7 @@ TEST(SM100Only_Device_Gemm_f16n_f16t_void_f32n_tensor_op_f32, 128x64x64_2x4x1_2s
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -461,8 +451,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_void_f32n_tensor_op_f32, 128x128x64_16x1x1_
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_16,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_64>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -470,7 +458,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16n_void_f32n_tensor_op_f32, 128x128x64_16x1x1_
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -534,8 +522,6 @@ TEST(SM100Only_Device_Gemm_f16n_f16n_void_f32n_tensor_op_f32, 256x64x64_4x1x1) {
|
||||
using MmaTileShape_MNK = Shape<_256,_64,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_64>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -543,7 +529,7 @@ TEST(SM100Only_Device_Gemm_f16n_f16n_void_f32n_tensor_op_f32, 256x64x64_4x1x1) {
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -607,8 +593,6 @@ TEST(SM100Only_Device_Gemm_f16t_f16t_void_f32n_tensor_op_f32, 256x256x64_2x1x1)
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_64>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -616,7 +600,7 @@ TEST(SM100Only_Device_Gemm_f16t_f16t_void_f32n_tensor_op_f32, 256x256x64_2x1x1)
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
@@ -88,8 +88,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_f16t_e4m3t_tensor_op_f32, 128x128x128_1x2
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = alpha * scale_a * scale_b * acc + beta * scale_c * C + per-row bias
|
||||
@@ -108,7 +106,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_f16t_e4m3t_tensor_op_f32, 128x128x128_1x2
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -173,8 +171,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_f16t_f32t_tensor_op_f32, 128x128x128_1x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = alpha * scale_a * scale_b * acc + beta * scale_c * C + per-row bias
|
||||
@@ -194,7 +190,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_f16t_f32t_tensor_op_f32, 128x128x128_1x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -265,8 +261,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_f16t_e4m3t_tensor_op_f32, 128x128x128_1x2
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = alpha * scale_a * scale_b * acc + beta * scale_c * C + per-row bias
|
||||
@@ -294,7 +288,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_f16t_e4m3t_tensor_op_f32, 128x128x128_1x2
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -359,8 +353,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_f16t_f32t_tensor_op_f32, 128x128x128_1x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_64>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_64>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = alpha * scale_a * scale_b * acc + beta * scale_c * C + per-row bias
|
||||
@@ -388,7 +380,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_f16t_f32t_tensor_op_f32, 128x128x128_1x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
@@ -82,8 +82,6 @@ TEST(SM100Only_Device_Gemm_e4m3n_e4m3t_void_f32n_tensor_op_f32, 64x64x128_4x1x1_
|
||||
using MmaTileShape_MNK = Shape<_64,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -91,7 +89,7 @@ TEST(SM100Only_Device_Gemm_e4m3n_e4m3t_void_f32n_tensor_op_f32, 64x64x128_4x1x1_
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -155,8 +153,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e5m2n_void_f32t_tensor_op_f32, 64x128x128_1x4x1
|
||||
using MmaTileShape_MNK = Shape<_64,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -167,7 +163,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e5m2n_void_f32t_tensor_op_f32, 64x128x128_1x4x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -232,8 +228,6 @@ TEST(SM100Only_Device_Gemm_e5m2n_e4m3n_void_f32t_tensor_op_f32, 128x64x128_1x8x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_8,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -241,7 +235,7 @@ TEST(SM100Only_Device_Gemm_e5m2n_e4m3n_void_f32t_tensor_op_f32, 128x64x128_1x8x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -305,8 +299,6 @@ TEST(SM100Only_Device_Gemm_e5m2t_e5m2t_void_f32n_tensor_op_f32, 128x128x128_2x8x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_8,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -317,7 +309,7 @@ TEST(SM100Only_Device_Gemm_e5m2t_e5m2t_void_f32n_tensor_op_f32, 128x128x128_2x8x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -383,8 +375,6 @@ TEST(SM100Only_Device_Gemm_e5m2n_e4m3t_void_f32n_tensor_op_f32, 128x64x128_2x4x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -392,7 +382,7 @@ TEST(SM100Only_Device_Gemm_e5m2n_e4m3t_void_f32n_tensor_op_f32, 128x64x128_2x4x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -457,8 +447,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_void_f32n_tensor_op_f32, 128x128x128_16x1
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_16,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -469,7 +457,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3n_void_f32n_tensor_op_f32, 128x128x128_16x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -534,8 +522,6 @@ TEST(SM100Only_Device_Gemm_e4m3n_e4m3n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_256,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -543,7 +529,7 @@ TEST(SM100Only_Device_Gemm_e4m3n_e4m3n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -607,8 +593,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3t_void_f32n_tensor_op_f32, 256x256x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -619,7 +603,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e4m3t_void_f32n_tensor_op_f32, 256x256x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
@@ -29,7 +29,7 @@
|
||||
#
|
||||
|
||||
#
|
||||
|
||||
if (CUTLASS_NVCC_ARCHS MATCHES 100a)
|
||||
add_custom_target(
|
||||
cutlass_test_unit_gemm_device_sm100_tensorop_narrow_precision
|
||||
DEPENDS
|
||||
@@ -38,7 +38,7 @@ add_custom_target(
|
||||
cutlass_test_unit_gemm_device_tensorop_sm100_f8xf6f4
|
||||
)
|
||||
|
||||
cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
cutlass_test_unit_gemm_device_add_executable(
|
||||
cutlass_test_unit_gemm_device_tensorop_sm100_f6f4xf6f4
|
||||
|
||||
BATCH_SOURCES ON
|
||||
@@ -50,7 +50,7 @@ cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
f6f4_f6f4_void_f32_tt_layout.cu
|
||||
)
|
||||
|
||||
cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
cutlass_test_unit_gemm_device_add_executable(
|
||||
cutlass_test_unit_gemm_device_tensorop_sm100_f6f4xf8
|
||||
|
||||
BATCH_SOURCES ON
|
||||
@@ -60,7 +60,7 @@ cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
f6f4_f8_void_f32_nt_layout.cu
|
||||
)
|
||||
|
||||
cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
cutlass_test_unit_gemm_device_add_executable(
|
||||
cutlass_test_unit_gemm_device_tensorop_sm100_f8xf6f4
|
||||
|
||||
BATCH_SOURCES ON
|
||||
@@ -69,3 +69,4 @@ cutlass_test_unit_gemm_device_add_executable_split_file(
|
||||
f8_f6f4_void_f32_tn_layout.cu
|
||||
f8_f6f4_void_f32_nt_layout.cu
|
||||
)
|
||||
endif()
|
||||
|
||||
+8
-24
@@ -112,8 +112,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m3n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -124,7 +122,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m3n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -189,8 +187,6 @@ TEST(SM100Only_Device_Gemm_e3m2n_e2m1n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -198,7 +194,7 @@ TEST(SM100Only_Device_Gemm_e3m2n_e2m1n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -262,8 +258,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m1n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_128,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -274,7 +268,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m1n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -339,8 +333,6 @@ TEST(SM100Only_Device_Gemm_e2m3n_e3m2n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -348,7 +340,7 @@ TEST(SM100Only_Device_Gemm_e2m3n_e3m2n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -412,8 +404,6 @@ TEST(SM100Only_Device_Gemm_e3m2n_e3m2n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_256,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -424,7 +414,7 @@ TEST(SM100Only_Device_Gemm_e3m2n_e3m2n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -489,8 +479,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m1n_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_256,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -498,7 +486,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m1n_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -562,8 +550,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m3n_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_256,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -574,7 +560,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m3n_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -639,8 +625,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m1n_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -648,7 +632,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m1n_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
+3
-9
@@ -112,8 +112,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m3t_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -121,7 +119,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m3t_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -185,8 +183,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m1t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -197,7 +193,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e2m1t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -262,8 +258,6 @@ TEST(SM100Only_Device_Gemm_e3m2n_e2m1t_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -271,7 +265,7 @@ TEST(SM100Only_Device_Gemm_e3m2n_e2m1t_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
+16
-48
@@ -112,8 +112,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 64x64x128_4x1x1_
|
||||
using MmaTileShape_MNK = Shape<_64,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -121,7 +119,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 64x64x128_4x1x1_
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -185,8 +183,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e2m3n_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
using MmaTileShape_MNK = Shape<_64,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -197,7 +193,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e2m3n_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -262,8 +258,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 64x192x128_2x4x1
|
||||
using MmaTileShape_MNK = Shape<_64,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_192,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -271,7 +265,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 64x192x128_2x4x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -335,8 +329,6 @@ TEST(SM100Only_Device_Gemm_e3m2t_e3m2n_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
using MmaTileShape_MNK = Shape<_64,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -344,7 +336,7 @@ TEST(SM100Only_Device_Gemm_e3m2t_e3m2n_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -408,8 +400,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -420,7 +410,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -485,8 +475,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -494,7 +482,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -558,8 +546,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e3m2n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_128,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -570,7 +556,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e3m2n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -635,8 +621,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -644,7 +628,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -708,8 +692,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -720,7 +702,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -785,8 +767,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -794,7 +774,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -858,8 +838,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e2m1n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_128,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -870,7 +848,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e2m1n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -935,8 +913,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -944,7 +920,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1008,8 +984,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_256,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -1020,7 +994,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1085,8 +1059,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e3m2n_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_256,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -1094,7 +1066,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e3m2n_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1158,8 +1130,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_256,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -1170,7 +1140,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1n_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1235,8 +1205,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e3m2n_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -1244,7 +1212,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e3m2n_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
+6
-18
@@ -111,8 +111,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1t_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
using MmaTileShape_MNK = Shape<_64,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -123,7 +121,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1t_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -188,8 +186,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e2m3t_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
using MmaTileShape_MNK = Shape<_64,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -197,7 +193,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e2m3t_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -261,8 +257,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1t_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -273,7 +267,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1t_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -338,8 +332,6 @@ TEST(SM100Only_Device_Gemm_e3m2t_e3m2t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -347,7 +339,7 @@ TEST(SM100Only_Device_Gemm_e3m2t_e3m2t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -411,8 +403,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -423,7 +413,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m1t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -488,8 +478,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m3t_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -497,7 +485,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e2m3t_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
+8
-24
@@ -111,8 +111,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e4m3t_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -123,7 +121,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e4m3t_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -188,8 +186,6 @@ TEST(SM100Only_Device_Gemm_e2m3n_e5m2t_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -197,7 +193,7 @@ TEST(SM100Only_Device_Gemm_e2m3n_e5m2t_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -261,8 +257,6 @@ TEST(SM100Only_Device_Gemm_e2m3n_e4m3t_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_128,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -273,7 +267,7 @@ TEST(SM100Only_Device_Gemm_e2m3n_e4m3t_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -338,8 +332,6 @@ TEST(SM100Only_Device_Gemm_e3m2n_e5m2t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -347,7 +339,7 @@ TEST(SM100Only_Device_Gemm_e3m2n_e5m2t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -411,8 +403,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e4m3t_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_256,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -423,7 +413,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e4m3t_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -488,8 +478,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e5m2t_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_256,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -497,7 +485,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e5m2t_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -561,8 +549,6 @@ TEST(SM100Only_Device_Gemm_e2m1n_e4m3t_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_256,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -573,7 +559,7 @@ TEST(SM100Only_Device_Gemm_e2m1n_e4m3t_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -638,8 +624,6 @@ TEST(SM100Only_Device_Gemm_e2m3n_e4m3t_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -647,7 +631,7 @@ TEST(SM100Only_Device_Gemm_e2m3n_e4m3t_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
+16
-48
@@ -112,8 +112,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e4m3n_void_f32n_tensor_op_f32, 64x64x128_4x1x1_
|
||||
using MmaTileShape_MNK = Shape<_64,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -121,7 +119,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e4m3n_void_f32n_tensor_op_f32, 64x64x128_4x1x1_
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -185,8 +183,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e5m2n_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
using MmaTileShape_MNK = Shape<_64,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -197,7 +193,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e5m2n_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -263,8 +259,6 @@ TEST(SM100Only_Device_Gemm_e3m2t_e4m3n_void_f32n_tensor_op_f32, 64x192x128_2x4x1
|
||||
using MmaTileShape_MNK = Shape<_64,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_192,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -272,7 +266,7 @@ TEST(SM100Only_Device_Gemm_e3m2t_e4m3n_void_f32n_tensor_op_f32, 64x192x128_2x4x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -336,8 +330,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
using MmaTileShape_MNK = Shape<_64,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -345,7 +337,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -409,8 +401,6 @@ TEST(SM100Only_Device_Gemm_e3m2t_e4m3n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -421,7 +411,7 @@ TEST(SM100Only_Device_Gemm_e3m2t_e4m3n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -486,8 +476,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e5m2n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -495,7 +483,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e5m2n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -559,8 +547,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e4m3n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_128,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -571,7 +557,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e4m3n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -636,8 +622,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -645,7 +629,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -709,8 +693,6 @@ TEST(SM100Only_Device_Gemm_e2m3t_e4m3n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -721,7 +703,7 @@ TEST(SM100Only_Device_Gemm_e2m3t_e4m3n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -786,8 +768,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -795,7 +775,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -859,8 +839,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_128,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_192,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -868,7 +846,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -932,8 +910,6 @@ TEST(SM100Only_Device_Gemm_e3m2t_e5m2n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -941,7 +917,7 @@ TEST(SM100Only_Device_Gemm_e3m2t_e5m2n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1005,8 +981,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e4m3n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_256,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -1017,7 +991,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e4m3n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1082,8 +1056,6 @@ TEST(SM100Only_Device_Gemm_e3m2t_e5m2n_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_256,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -1091,7 +1063,7 @@ TEST(SM100Only_Device_Gemm_e3m2t_e5m2n_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1155,8 +1127,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e4m3n_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_256,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -1167,7 +1137,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e4m3n_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1232,8 +1202,6 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -1241,7 +1209,7 @@ TEST(SM100Only_Device_Gemm_e2m1t_e5m2n_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
+6
-18
@@ -111,8 +111,6 @@ TEST(SM100Only_Device_Gemm_e4m3n_e2m3t_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
using MmaTileShape_MNK = Shape<_64,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -123,7 +121,7 @@ TEST(SM100Only_Device_Gemm_e4m3n_e2m3t_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -188,8 +186,6 @@ TEST(SM100Only_Device_Gemm_e5m2n_e3m2t_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
using MmaTileShape_MNK = Shape<_64,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -197,7 +193,7 @@ TEST(SM100Only_Device_Gemm_e5m2n_e3m2t_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -261,8 +257,6 @@ TEST(SM100Only_Device_Gemm_e4m3n_e2m1t_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -273,7 +267,7 @@ TEST(SM100Only_Device_Gemm_e4m3n_e2m1t_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -338,8 +332,6 @@ TEST(SM100Only_Device_Gemm_e5m2n_e2m3t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -347,7 +339,7 @@ TEST(SM100Only_Device_Gemm_e5m2n_e2m3t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -411,8 +403,6 @@ TEST(SM100Only_Device_Gemm_e4m3n_e3m2t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -423,7 +413,7 @@ TEST(SM100Only_Device_Gemm_e4m3n_e3m2t_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -488,8 +478,6 @@ TEST(SM100Only_Device_Gemm_e5m2n_e2m1t_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -497,7 +485,7 @@ TEST(SM100Only_Device_Gemm_e5m2n_e2m1t_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
+16
-48
@@ -112,8 +112,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m1n_void_f32n_tensor_op_f32, 64x64x128_4x1x1_
|
||||
using MmaTileShape_MNK = Shape<_64,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -124,7 +122,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m1n_void_f32n_tensor_op_f32, 64x64x128_4x1x1_
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -189,8 +187,6 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m3n_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
using MmaTileShape_MNK = Shape<_64,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -198,7 +194,7 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m3n_void_f32n_tensor_op_f32, 64x128x128_2x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -262,8 +258,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m1n_void_f32n_tensor_op_f32, 64x192x128_2x4x1
|
||||
using MmaTileShape_MNK = Shape<_64,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_192,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -271,7 +265,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m1n_void_f32n_tensor_op_f32, 64x192x128_2x4x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -335,8 +329,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e3m2n_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
using MmaTileShape_MNK = Shape<_64,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -347,7 +339,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e3m2n_void_f32n_tensor_op_f32, 64x256x128_2x2x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -412,8 +404,6 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m1n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -421,7 +411,7 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m1n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -485,8 +475,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m3n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -497,7 +485,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m3n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -562,8 +550,6 @@ TEST(SM100Only_Device_Gemm_e5m2t_e3m2n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_128,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -571,7 +557,7 @@ TEST(SM100Only_Device_Gemm_e5m2t_e3m2n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -635,8 +621,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m3n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -647,7 +631,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m3n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -712,8 +696,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e3m2n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -721,7 +703,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e3m2n_void_f32n_tensor_op_f32, 128x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -785,8 +767,6 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m3n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -797,7 +777,7 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m3n_void_f32n_tensor_op_f32, 128x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -862,8 +842,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m1n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_128,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_192,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -871,7 +849,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m1n_void_f32n_tensor_op_f32, 128x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -935,8 +913,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e3m2n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_128,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_256,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -947,7 +923,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e3m2n_void_f32n_tensor_op_f32, 128x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1012,8 +988,6 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m1n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
using MmaTileShape_MNK = Shape<_256,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -1021,7 +995,7 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m1n_void_f32n_tensor_op_f32, 256x64x128_4x1x1
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1085,8 +1059,6 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m1n_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
using MmaTileShape_MNK = Shape<_256,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -1094,7 +1066,7 @@ TEST(SM100Only_Device_Gemm_e5m2t_e2m1n_void_f32n_tensor_op_f32, 256x128x128_2x1x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1158,8 +1130,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m3n_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
using MmaTileShape_MNK = Shape<_256,_192,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_192,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -1167,7 +1137,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m3n_void_f32n_tensor_op_f32, 256x192x128_2x4x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -1231,8 +1201,6 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m1n_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -1243,7 +1211,7 @@ TEST(SM100Only_Device_Gemm_e4m3t_e2m1n_void_f32n_tensor_op_f32, 256x256x128_2x2x
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
@@ -82,8 +82,6 @@ TEST(SM100Only_Device_Gemm_s8t_s8n_s32t_s32t_tensor_op_f32, 128x128x128_1x2x1_1s
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = per-row alpha * acc + per-row beta * C + per-row bias
|
||||
@@ -101,7 +99,7 @@ TEST(SM100Only_Device_Gemm_s8t_s8n_s32t_s32t_tensor_op_f32, 128x128x128_1x2x1_1s
|
||||
//
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -166,8 +164,6 @@ TEST(SM100Only_Device_Gemm_s8t_s8n_s32t_s32t_tensor_op_f32, 128x128x128_1x2x1_1s
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_2,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
// Epilogue fusion operation
|
||||
// Z = per-col alpha * acc + per-col beta * C + per-col bias
|
||||
@@ -185,7 +181,7 @@ TEST(SM100Only_Device_Gemm_s8t_s8n_s32t_s32t_tensor_op_f32, 128x128x128_1x2x1_1s
|
||||
//
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
@@ -82,8 +82,6 @@ TEST(SM100Only_Device_Gemm_s8n_s8t_void_s32n_tensor_op_f32, 64x64x128_4x1x1_1sm_
|
||||
using MmaTileShape_MNK = Shape<_64,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -94,7 +92,7 @@ TEST(SM100Only_Device_Gemm_s8n_s8t_void_s32n_tensor_op_f32, 64x64x128_4x1x1_1sm_
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -159,8 +157,6 @@ TEST(SM100Only_Device_Gemm_s8t_s8n_void_s32t_tensor_op_f32, 64x128x128_1x4x1_1sm
|
||||
using MmaTileShape_MNK = Shape<_64,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -168,7 +164,7 @@ TEST(SM100Only_Device_Gemm_s8t_s8n_void_s32t_tensor_op_f32, 64x128x128_1x4x1_1sm
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -232,8 +228,6 @@ TEST(SM100Only_Device_Gemm_s8n_s8n_void_s32t_tensor_op_f32, 128x64x128_1x8x1_str
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_1,_8,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -244,7 +238,7 @@ TEST(SM100Only_Device_Gemm_s8n_s8n_void_s32t_tensor_op_f32, 128x64x128_1x8x1_str
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -309,8 +303,6 @@ TEST(SM100Only_Device_Gemm_s8t_s8t_void_s32n_tensor_op_f32, 128x128x128_2x8x1_1s
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_8,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -318,7 +310,7 @@ TEST(SM100Only_Device_Gemm_s8t_s8t_void_s32n_tensor_op_f32, 128x128x128_2x8x1_1s
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -383,8 +375,6 @@ TEST(SM100Only_Device_Gemm_s8n_s8t_void_s32n_tensor_op_f32, 128x64x128_2x4x1_2sm
|
||||
using MmaTileShape_MNK = Shape<_128,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_4,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -395,7 +385,7 @@ TEST(SM100Only_Device_Gemm_s8n_s8t_void_s32n_tensor_op_f32, 128x64x128_2x4x1_2sm
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -461,8 +451,6 @@ TEST(SM100Only_Device_Gemm_s8t_s8n_void_s32n_tensor_op_f32, 128x128x128_16x1x1_2
|
||||
using MmaTileShape_MNK = Shape<_128,_128,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_16,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_64,_128,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -470,7 +458,7 @@ TEST(SM100Only_Device_Gemm_s8t_s8n_void_s32n_tensor_op_f32, 128x128x128_16x1x1_2
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -534,8 +522,6 @@ TEST(SM100Only_Device_Gemm_s8n_s8n_void_s32n_tensor_op_f32, 256x64x128_4x1x1_str
|
||||
using MmaTileShape_MNK = Shape<_256,_64,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_4,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_64,_128>;
|
||||
|
||||
// Tile Scheduler
|
||||
using TileScheduler = cutlass::gemm::StreamKScheduler;
|
||||
@@ -546,7 +532,7 @@ TEST(SM100Only_Device_Gemm_s8n_s8n_void_s32n_tensor_op_f32, 256x64x128_4x1x1_str
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
@@ -611,8 +597,6 @@ TEST(SM100Only_Device_Gemm_s8t_s8t_void_s32n_tensor_op_f32, 256x256x128_2x1x1) {
|
||||
using MmaTileShape_MNK = Shape<_256,_256,_128>;
|
||||
// Cluster size for multicast
|
||||
using ClusterShape_MNK = Shape<_2,_1,_1>;
|
||||
// Collective Epilogue takes the output tile shape for 1 CTA
|
||||
using PerSmTileShape_MNK = Shape<_128,_256,_128>;
|
||||
|
||||
//
|
||||
// Construct CollectiveEpilogue
|
||||
@@ -620,7 +604,7 @@ TEST(SM100Only_Device_Gemm_s8t_s8t_void_s32n_tensor_op_f32, 256x256x128_2x1x1) {
|
||||
|
||||
using CollectiveEpilogue = typename cutlass::epilogue::collective::CollectiveBuilder<
|
||||
cutlass::arch::Sm100, cutlass::arch::OpClassTensorOp, // Arch and Tensorop spec
|
||||
PerSmTileShape_MNK, ClusterShape_MNK, // Epilogue tile shape, and cluster shape
|
||||
MmaTileShape_MNK, ClusterShape_MNK, // Mma instruction tile shape, cluster shape
|
||||
cutlass::epilogue::collective::EpilogueTileAuto, // Epilogue subtile shape. Auto will find a suitable tile shape
|
||||
ElementAccumulator, ElementCompute, // Mma instr's accumulator type and compute precision for epilogue
|
||||
ElementC, GmemLayoutC, AlignC, // C tensor description
|
||||
|
||||
Reference in New Issue
Block a user