CUTLASS 2.6 (#298)

CUTLASS 2.6
This commit is contained in:
Manish Gupta
2021-07-23 00:40:53 -04:00
committed by GitHub
parent 6c29fe20ba
commit e5d51840e8
308 changed files with 32408 additions and 4722 deletions
+53
View File
@@ -29,6 +29,7 @@
#include "../../common/cutlass_unit_test.h"
#include "cutlass/complex.h"
#include "cutlass/quaternion.h"
#include "cutlass/gemm/gemm.h"
#include "cutlass/gemm/warp/mma_simt.h"
@@ -593,3 +594,55 @@ TEST(SM50_warp_gemm_complex_f64_col_row_row, 32x16x1_1x1x1) {
test::gemm::warp::Testbed<Mma, cutlass::gemm::GemmShape<128, 128, 8>>().run();
}
/////////////////////////////////////////////////////////////////////////////////////////////////
TEST(SM50_warp_gemm_quaternion_f32_col_row_col, 16x8x8_1x1x1) {
using Policy = cutlass::gemm::warp::MmaSimtPolicy<
cutlass::MatrixShape<8, 4>,
cutlass::layout::ColumnMajorInterleaved<2>,
cutlass::gemm::GemmShape<1, 1, 1>
>;
using quaternion_f32_t = cutlass::Quaternion<float>;
using Mma = cutlass::gemm::warp::MmaSimt<
cutlass::gemm::GemmShape<16, 8, 8>,
quaternion_f32_t,
cutlass::layout::ColumnMajor,
quaternion_f32_t,
cutlass::layout::RowMajor,
quaternion_f32_t,
cutlass::layout::ColumnMajor,
Policy
>;
test::gemm::warp::Testbed<Mma, cutlass::gemm::GemmShape<128, 128, 8>>().run();
}
/////////////////////////////////////////////////////////////////////////////////////////////////
TEST(SM50_warp_gemm_quaternion_f32_col_row_row, 16x8x8_1x1x1) {
using Policy = cutlass::gemm::warp::MmaSimtPolicy<
cutlass::MatrixShape<8, 4>,
cutlass::layout::ColumnMajorInterleaved<2>,
cutlass::gemm::GemmShape<1, 1, 1>
>;
using quaternion_f32_t = cutlass::Quaternion<float>;
using Mma = cutlass::gemm::warp::MmaSimt<
cutlass::gemm::GemmShape<16, 8, 8>,
quaternion_f32_t,
cutlass::layout::ColumnMajor,
quaternion_f32_t,
cutlass::layout::RowMajor,
quaternion_f32_t,
cutlass::layout::RowMajor,
Policy
>;
test::gemm::warp::Testbed<Mma, cutlass::gemm::GemmShape<128, 128, 8>>().run();
}
/////////////////////////////////////////////////////////////////////////////////////////////////
+1
View File
@@ -1856,3 +1856,4 @@ TEST(SM80_warp_gemm_tensor_op_canonical_tf32_col_row, 32x32x8_64x32x8_8x8x4) {
#endif // if defined(CUTLASS_ARCH_MMA_SM80_SUPPORTED)
+38 -31
View File
@@ -33,6 +33,7 @@
#include "cutlass/numeric_types.h"
#include "cutlass/subbyte_reference.h"
#include "cutlass/platform/platform.h"
#include "cutlass/arch/arch.h"
#include "cutlass/util/host_tensor.h"
#include "cutlass/util/tensor_view_io.h"
@@ -100,9 +101,9 @@ __global__ void kernel(
typename Mma::LayoutB layout_B = Mma::LayoutB::packed({ThreadblockShape::kK, ThreadblockShape::kN});
typename Mma::LayoutC layout_C = Mma::LayoutC::packed({Mma::Shape::kM, Mma::Shape::kN});
typename Mma::IteratorA iter_A({smem_buffer_A.data(), layout_A}, cutlass::LaneId());
typename Mma::IteratorA iter_A({smem_buffer_A.data(), layout_A}, cutlass::arch::LaneId());
typename Mma::IteratorB iter_B({smem_buffer_B.data(), layout_B}, cutlass::LaneId());
typename Mma::IteratorB iter_B({smem_buffer_B.data(), layout_B}, cutlass::arch::LaneId());
FragmentA frag_A;
FragmentB frag_B;
@@ -129,7 +130,7 @@ __global__ void kernel(
}
}
typename Mma::IteratorC iter_C({output_C, layout_C}, cutlass::LaneId());
typename Mma::IteratorC iter_C({output_C, layout_C}, cutlass::arch::LaneId());
iter_C.store(accum);
}
@@ -142,7 +143,7 @@ template <
typename Mma_,
/// Size of threadblock-scoped shape used to store SMEM
typename ThreadblockShape_,
/// The innter product operation performed by GEMM
/// The inner product operation performed by GEMM
typename Operator_ = cutlass::arch::OpMultiplyAdd
>
struct Testbed {
@@ -205,8 +206,10 @@ struct Testbed {
}
uint64_t seed = 7;
cutlass::reference::host::TensorFillRandomUniform(
tensor_A.host_view(), seed, scope_max, scope_min, 0);
cutlass::reference::host::BlockFillRandomUniform(tensor_A.host_data(),
tensor_A.capacity(), seed, scope_max, scope_min, 0);
} else if (init_A == cutlass::Distribution::Sequential) {
cutlass::reference::host::BlockFillSequential(tensor_A.host_data(),
tensor_A.capacity());
@@ -230,8 +233,10 @@ struct Testbed {
}
uint64_t seed = 7;
cutlass::reference::host::TensorFillRandomUniform(
tensor_B.host_view(), seed + 16, scope_max, scope_min, 0);
cutlass::reference::host::BlockFillRandomUniform(tensor_B.host_data(),
tensor_B.capacity(), seed, scope_max, scope_min, 0);
} else if (init_B == cutlass::Distribution::Sequential) {
cutlass::reference::host::BlockFillSequential(tensor_B.host_data(),
tensor_B.capacity());
@@ -313,23 +318,25 @@ struct Testbed {
cutlass::TensorView<ElementA, cutlass::layout::ColumnMajor> tensor_A_physical(
tensor_A.host_data(),
tensor_A.stride(),
tensor_A.stride()[0],
tensor_A.extent());
cutlass::TensorView<ElementB, cutlass::layout::RowMajor> tensor_B_physical(
tensor_B.host_data(),
tensor_B.stride(),
tensor_B.stride()[0],
tensor_B.extent());
std::cout <<"cutlass::sizeof_bits<ElementA>::value = "<<cutlass::sizeof_bits<ElementA>::value<<"\n";
std::cout
<< "A:\n" << tensor_A.host_view() << "\n\n"
<< "A(physical - stride: " << tensor_A.stride() << ", extent: " << tensor_A.extent() << "):\n" << tensor_A_physical << "\n\n";
<< "A(physical - stride: " << tensor_A.stride()[0]
<< ", extent: " << tensor_A.extent() << "):\n" << tensor_A_physical << "\n\n";
std::cout <<"cutlass::sizeof_bits<ElementB>::value = "<<cutlass::sizeof_bits<ElementB>::value<<"\n";
std::cout
<< "B:\n" << tensor_B.host_view() << "\n\n"
<< "B(physical - stride: " << tensor_B.stride() << ", extent: " << tensor_B.extent() << "):\n" << tensor_B_physical << "\n\n";
<< "B(physical - stride: " << tensor_B.stride()[0]
<< ", extent: " << tensor_B.extent() << "):\n" << tensor_B_physical << "\n\n";
std::cout
<< "C:\n" << tensor_C.host_view() << "\n\n"
@@ -493,23 +500,23 @@ struct TestbedComplex {
cutlass::TensorView<ElementA, cutlass::layout::ColumnMajor> tensor_A_physical(
tensor_A.host_data(),
tensor_A.stride(),
tensor_A.stride()[0],
tensor_A.extent());
cutlass::TensorView<ElementB, cutlass::layout::RowMajor> tensor_B_physical(
tensor_B.host_data(),
tensor_B.stride(),
tensor_B.stride()[0],
tensor_B.extent());
std::cout <<"cutlass::sizeof_bits<ElementA>::value = "<<cutlass::sizeof_bits<ElementA>::value<<"\n";
std::cout
<< "A:\n" << tensor_A.host_view() << "\n\n"
<< "A(physical - stride: " << tensor_A.stride() << ", extent: " << tensor_A.extent() << "):\n" << tensor_A_physical << "\n\n";
<< "A(physical - stride: " << tensor_A.stride()[0] << ", extent: " << tensor_A.extent() << "):\n" << tensor_A_physical << "\n\n";
std::cout <<"cutlass::sizeof_bits<ElementB>::value = "<<cutlass::sizeof_bits<ElementB>::value<<"\n";
std::cout
<< "B:\n" << tensor_B.host_view() << "\n\n"
<< "B(physical - stride: " << tensor_B.stride() << ", extent: " << tensor_B.extent() <<"):\n" << tensor_B_physical << "\n\n";
<< "B(physical - stride: " << tensor_B.stride()[0] << ", extent: " << tensor_B.extent() <<"):\n" << tensor_B_physical << "\n\n";
std::cout
<< "C:\n" << tensor_C.host_view() << "\n\n"
@@ -574,9 +581,9 @@ __global__ void kernel_transform(
typename Mma::LayoutB layout_B = Mma::LayoutB::packed({ThreadblockShape::kK, ThreadblockShape::kN});
typename Mma::LayoutC layout_C = Mma::LayoutC::packed({Mma::Shape::kM, Mma::Shape::kN});
typename Mma::IteratorA iter_A({smem_buffer_A.data(), layout_A}, cutlass::LaneId());
typename Mma::IteratorA iter_A({smem_buffer_A.data(), layout_A}, cutlass::arch::LaneId());
typename Mma::IteratorB iter_B({smem_buffer_B.data(), layout_B}, cutlass::LaneId());
typename Mma::IteratorB iter_B({smem_buffer_B.data(), layout_B}, cutlass::arch::LaneId());
FragmentA loaded_frag_A;
FragmentB loaded_frag_B;
@@ -608,7 +615,7 @@ __global__ void kernel_transform(
}
}
typename Mma::IteratorC iter_C({output_C, layout_C}, cutlass::LaneId());
typename Mma::IteratorC iter_C({output_C, layout_C}, cutlass::arch::LaneId());
iter_C.store(accum);
}
@@ -790,23 +797,23 @@ struct TransformTestbed {
cutlass::TensorView<ElementA, cutlass::layout::ColumnMajor> tensor_A_physical(
tensor_A.host_data(),
tensor_A.stride(),
tensor_A.stride()[0],
tensor_A.extent());
cutlass::TensorView<ElementB, cutlass::layout::RowMajor> tensor_B_physical(
tensor_B.host_data(),
tensor_B.stride(),
tensor_B.stride()[0],
tensor_B.extent());
std::cout <<"cutlass::sizeof_bits<ElementA>::value = "<<cutlass::sizeof_bits<ElementA>::value<<"\n";
std::cout
<< "A:\n" << tensor_A.host_view() << "\n\n"
<< "A(physical - stride: " << tensor_A.stride() << ", extent: " << tensor_A.extent() << "):\n" << tensor_A_physical << "\n\n";
<< "A(physical - stride: " << tensor_A.stride()[0] << ", extent: " << tensor_A.extent() << "):\n" << tensor_A_physical << "\n\n";
std::cout <<"cutlass::sizeof_bits<ElementB>::value = "<<cutlass::sizeof_bits<ElementB>::value<<"\n";
std::cout
<< "B:\n" << tensor_B.host_view() << "\n\n"
<< "B(physical - stride: " << tensor_B.stride() << ", extent: " << tensor_B.extent() << "):\n" << tensor_B_physical << "\n\n";
<< "B(physical - stride: " << tensor_B.stride()[0] << ", extent: " << tensor_B.extent() << "):\n" << tensor_B_physical << "\n\n";
std::cout
<< "C:\n" << tensor_C.host_view() << "\n\n"
@@ -970,23 +977,23 @@ struct TransformedTestbedComplex {
cutlass::TensorView<ElementA, cutlass::layout::ColumnMajor> tensor_A_physical(
tensor_A.host_data(),
tensor_A.stride(),
tensor_A.stride()[0],
tensor_A.extent());
cutlass::TensorView<ElementB, cutlass::layout::RowMajor> tensor_B_physical(
tensor_B.host_data(),
tensor_B.stride(),
tensor_B.stride()[0],
tensor_B.extent());
std::cout <<"cutlass::sizeof_bits<ElementA>::value = "<<cutlass::sizeof_bits<ElementA>::value<<"\n";
std::cout
<< "A:\n" << tensor_A.host_view() << "\n\n"
<< "A(physical - stride: " << tensor_A.stride() << ", extent: " << tensor_A.extent() << "):\n" << tensor_A_physical << "\n\n";
<< "A(physical - stride: " << tensor_A.stride()[0] << ", extent: " << tensor_A.extent() << "):\n" << tensor_A_physical << "\n\n";
std::cout <<"cutlass::sizeof_bits<ElementB>::value = "<<cutlass::sizeof_bits<ElementB>::value<<"\n";
std::cout
<< "B:\n" << tensor_B.host_view() << "\n\n"
<< "B(physical - stride: " << tensor_B.stride() << ", extent: " << tensor_B.extent() <<"):\n" << tensor_B_physical << "\n\n";
<< "B(physical - stride: " << tensor_B.stride()[0] << ", extent: " << tensor_B.extent() <<"):\n" << tensor_B_physical << "\n\n";
std::cout
<< "C:\n" << tensor_C.host_view() << "\n\n"
@@ -1073,11 +1080,11 @@ __global__ void sparse_kernel(
Mma::Shape::kK / Mma::kSparse /
Mma::kElementsPerElementE / Mma::kInterleaved});
typename Mma::IteratorA iter_A({smem_buffer_A.data(), layout_A}, cutlass::LaneId());
typename Mma::IteratorA iter_A({smem_buffer_A.data(), layout_A}, cutlass::arch::LaneId());
typename Mma::IteratorB iter_B({smem_buffer_B.data(), layout_B}, cutlass::LaneId());
typename Mma::IteratorB iter_B({smem_buffer_B.data(), layout_B}, cutlass::arch::LaneId());
typename Mma::IteratorE iter_E({smem_buffer_E.data(), layout_E}, cutlass::LaneId());
typename Mma::IteratorE iter_E({smem_buffer_E.data(), layout_E}, cutlass::arch::LaneId());
FragmentA frag_A;
FragmentB frag_B;
@@ -1108,7 +1115,7 @@ __global__ void sparse_kernel(
}
}
typename Mma::IteratorC iter_C({output_C, layout_C}, cutlass::LaneId());
typename Mma::IteratorC iter_C({output_C, layout_C}, cutlass::arch::LaneId());
iter_C.store(accum);
}