CUTLASS 3.2.1 (#1113)
* Updates for 3.2.1 release. * Minor fix in gemm op profiler for raster order. * Add scheduler mapping for raster order in the kernels.
This commit is contained in:
@@ -44,13 +44,22 @@
|
||||
Map tensor sizes (Conv2d -> ImplicitGemm) : implicit_gemm_tensor_[a|b|c]_size(ConvolutionOperator)
|
||||
Map tensor problem sizes (Conv2d -> ImplicitGemm): implicit_gemm_problem_size(ConvolutionOperator)
|
||||
*/
|
||||
/*
|
||||
Note: CUTLASS 3x increases the host compiler requirements to C++17. However, certain
|
||||
existing integrations of CUTLASS require C++11 host compilers.
|
||||
|
||||
Until this requirement can be lifted, certain headers with this annotation are required
|
||||
to be remain consistent with C++11 syntax.
|
||||
|
||||
C++11 compatibility is enforced by `cutlass_test_unit_core_cpp11`.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/tensor_coord.h"
|
||||
#include "cutlass/fast_math.h"
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/gemm_enumerated_types.h"
|
||||
#include "cutlass/matrix_coord.h"
|
||||
#include "cutlass/conv/convolution.h"
|
||||
#include "cutlass/functional.h"
|
||||
@@ -80,7 +89,7 @@ struct Conv2dProblemSize {
|
||||
|
||||
public:
|
||||
CUTLASS_HOST_DEVICE
|
||||
Conv2dProblemSize():
|
||||
Conv2dProblemSize():
|
||||
N(0), H(0), W(0), C(0), P(0), Q(0), K(0), R(0), S(0),
|
||||
pad_h(0), pad_w(0), stride_h(1), stride_w(1), dilation_h(1), dilation_w(1),
|
||||
mode(Mode::kConvolution), split_k_slices(1), groups(1) { }
|
||||
@@ -125,7 +134,7 @@ public:
|
||||
int split_k_slices = 1,
|
||||
int groups = 1
|
||||
):
|
||||
N(N), H(H), W(W), C(C), K(K), R(R), S(S), P(P), Q(Q),
|
||||
N(N), H(H), W(W), C(C), P(P), Q(Q), K(K), R(R), S(S),
|
||||
pad_h(pad_h), pad_w(pad_w), stride_h(stride_h), stride_w(stride_w),
|
||||
dilation_h(dilation_h), dilation_w(dilation_w),
|
||||
mode(mode), split_k_slices(split_k_slices), groups (groups) { }
|
||||
@@ -145,11 +154,11 @@ public:
|
||||
int groups = 1
|
||||
):
|
||||
N(input_size.n()), H(input_size.h()), W(input_size.w()), C(input_size.c()),
|
||||
P(output_size.h()), Q(output_size.w()),
|
||||
K(filter_size.n()), R(filter_size.h()), S(filter_size.w()),
|
||||
pad_h(padding[0]), pad_w(padding[2]),
|
||||
stride_h(stride.row()), stride_w(stride.column()),
|
||||
dilation_h(dilation.row()), dilation_w(dilation.column()),
|
||||
P(output_size.h()), Q(output_size.w()),
|
||||
mode(mode), split_k_slices(split_k_slices), groups(groups) {}
|
||||
|
||||
/// Constructs convolution problem size from cutlass Tensor4DCoord and MatrixCoord
|
||||
@@ -188,8 +197,8 @@ public:
|
||||
int groups = 1
|
||||
):
|
||||
N(input_size.n()), H(input_size.h()), W(input_size.w()), C(input_size.c()),
|
||||
P(output_size.h()), Q(output_size.w()),
|
||||
K(filter_size.n()), R(filter_size.h()), S(filter_size.w()),
|
||||
P(output_size.h()), Q(output_size.w()),
|
||||
pad_h(R / 2), pad_w(S / 2), stride_h(1), stride_w(1),
|
||||
dilation_h(1), dilation_w(1),
|
||||
mode(mode), split_k_slices(split_k_slices), groups(groups) {}
|
||||
@@ -486,7 +495,6 @@ int depthwise_gemm_k_iterations(
|
||||
CUTLASS_HOST_DEVICE
|
||||
int implicit_gemm_k_iterations_per_channel(
|
||||
Operator conv_operator,
|
||||
int threadblock_K,
|
||||
Conv2dProblemSize const &problem_size,
|
||||
IteratorAlgorithm algorithm = IteratorAlgorithm::kAnalytic) {
|
||||
|
||||
|
||||
@@ -44,6 +44,15 @@
|
||||
Map tensor sizes (Conv3d -> ImplicitGemm) : implicit_gemm_tensor_[a|b|c]_size(ConvolutionOperator)
|
||||
Map tensor problem sizes (Conv3d -> ImplicitGemm): implicit_gemm_problem_size(ConvolutionOperator)
|
||||
*/
|
||||
/*
|
||||
Note: CUTLASS 3x increases the host compiler requirements to C++17. However, certain
|
||||
existing integrations of CUTLASS require C++11 host compilers.
|
||||
|
||||
Until this requirement can be lifted, certain headers with this annotation are required
|
||||
to be remain consistent with C++11 syntax.
|
||||
|
||||
C++11 compatibility is enforced by `cutlass_test_unit_core_cpp11`.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
@@ -80,11 +89,11 @@ struct Conv3dProblemSize : public Conv2dProblemSize {
|
||||
public:
|
||||
CUTLASS_HOST_DEVICE
|
||||
Conv3dProblemSize():
|
||||
Conv2dProblemSize(),
|
||||
D(0), T(0), Z(0),
|
||||
pad_d(0),
|
||||
stride_d(1),
|
||||
dilation_d(1),
|
||||
Conv2dProblemSize() { }
|
||||
dilation_d(1) { }
|
||||
|
||||
/// Constructor for default padding, stride, dilation, and split-K
|
||||
CUTLASS_HOST_DEVICE
|
||||
@@ -102,10 +111,10 @@ public:
|
||||
int R,
|
||||
int S,
|
||||
Mode mode
|
||||
):
|
||||
):
|
||||
Conv2dProblemSize(N, H, W, C, P, Q, K, R, S, mode),
|
||||
D(D), T(T), Z(Z),
|
||||
pad_d(T / 2), stride_d(1), dilation_d(1),
|
||||
Conv2dProblemSize(N, H, W, C, P, Q, K, R, S, mode) { }
|
||||
pad_d(T / 2), stride_d(1), dilation_d(1) { }
|
||||
|
||||
/// Constructor
|
||||
CUTLASS_HOST_DEVICE
|
||||
@@ -134,15 +143,15 @@ public:
|
||||
Mode mode,
|
||||
int split_k_slices = 1,
|
||||
int groups = 1
|
||||
):
|
||||
D(D), T(T), Z(Z),
|
||||
pad_d(pad_d), stride_d(stride_d), dilation_d(dilation_d),
|
||||
):
|
||||
Conv2dProblemSize(
|
||||
N, H, W, C, K, R, S, P, Q,
|
||||
pad_h, pad_w,
|
||||
stride_h, stride_w,
|
||||
dilation_h, dilation_w,
|
||||
mode, split_k_slices, groups) { }
|
||||
N, H, W, C, K, R, S, P, Q,
|
||||
pad_h, pad_w,
|
||||
stride_h, stride_w,
|
||||
dilation_h, dilation_w,
|
||||
mode, split_k_slices, groups),
|
||||
D(D), T(T), Z(Z),
|
||||
pad_d(pad_d), stride_d(stride_d), dilation_d(dilation_d) { }
|
||||
|
||||
/// Constructs convolution problem size from cutlass Tensor5DCoord and Coord3D
|
||||
// set *user-defined* output size and sets Z, P, and Q (include all data members in ctor)
|
||||
@@ -158,8 +167,6 @@ public:
|
||||
int split_k_slices = 1,
|
||||
int groups = 1
|
||||
):
|
||||
D(input_size.d()), T(filter_size.d()), Z(output_size.d()),
|
||||
pad_d(padding[0]), stride_d(stride[0]), dilation_d(dilation[0]),
|
||||
Conv2dProblemSize(
|
||||
{input_size.n(), input_size.h(), input_size.w(), input_size.c()},
|
||||
{filter_size.n(), filter_size.h(), filter_size.w(), filter_size.c()},
|
||||
@@ -167,8 +174,9 @@ public:
|
||||
{stride[1], stride[2]},
|
||||
{dilation[1], dilation[2]},
|
||||
{output_size.n(), output_size.h(), output_size.w(), output_size.c()},
|
||||
mode, split_k_slices, groups
|
||||
) { }
|
||||
mode, split_k_slices, groups),
|
||||
D(input_size.d()), T(filter_size.d()), Z(output_size.d()),
|
||||
pad_d(padding[0]), stride_d(stride[0]), dilation_d(dilation[0]) { }
|
||||
|
||||
/// Constructs convolution problem size from cutlass Tensor5DCoord and Coord3D
|
||||
// *computes* output size and sets Z, P and Q (include all data members in ctor)
|
||||
@@ -183,18 +191,18 @@ public:
|
||||
int split_k_slices = 1,
|
||||
int groups = 1
|
||||
):
|
||||
D(input_size.d()), T(filter_size.d()),
|
||||
pad_d(padding[0]), stride_d(stride[0]), dilation_d(dilation[0]),
|
||||
Conv2dProblemSize(
|
||||
{input_size.n(), input_size.h(), input_size.w(), input_size.c()},
|
||||
{filter_size.n(), filter_size.h(), filter_size.w(), filter_size.c()},
|
||||
{padding[1], padding[1], padding[2], padding[2]},
|
||||
{stride[1], stride[2]},
|
||||
{dilation[1], dilation[2]},
|
||||
mode, split_k_slices, groups
|
||||
) {
|
||||
mode, split_k_slices, groups),
|
||||
D(input_size.d()), T(filter_size.d()),
|
||||
pad_d(padding[0]), stride_d(stride[0]), dilation_d(dilation[0])
|
||||
{
|
||||
// set output Z
|
||||
Z = ((D + pad_d * 2 - T * dilation_d) / stride_d) + 1;
|
||||
Z = ((D + pad_d * 2 - T * dilation_d) / stride_d) + 1;
|
||||
}
|
||||
|
||||
/// Equality operator (ignores mode and split_k_slice)
|
||||
|
||||
@@ -70,13 +70,23 @@ Map elements' data types (ImplicitGemm -> Conv): GemmToConvElementMap
|
||||
Map elements' data types (Conv -> ImplicitGemm): ConvToGemmElementMap
|
||||
*/
|
||||
|
||||
/*
|
||||
Note: CUTLASS 3x increases the host compiler requirements to C++17. However, certain
|
||||
existing integrations of CUTLASS require C++11 host compilers.
|
||||
|
||||
Until this requirement can be lifted, certain headers with this annotation are required
|
||||
to be remain consistent with C++11 syntax.
|
||||
|
||||
C++11 compatibility is enforced by `cutlass_test_unit_core_cpp11`.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/layout/tensor.h"
|
||||
#include "cutlass/tensor_coord.h"
|
||||
#include "cutlass/fast_math.h"
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/gemm_enumerated_types.h"
|
||||
#include "cutlass/matrix_coord.h"
|
||||
|
||||
namespace cutlass {
|
||||
|
||||
@@ -142,7 +142,7 @@ struct DirectConvolutionParams {
|
||||
ThreadblockShape::kN);
|
||||
|
||||
gemm_k_iterations_per_channel = implicit_gemm_k_iterations_per_channel(
|
||||
kConvolutionalOperator, ThreadblockShape::kK, args.problem_size, kIteratorAlgorithm);
|
||||
kConvolutionalOperator, args.problem_size, kIteratorAlgorithm);
|
||||
|
||||
ThreadblockSwizzle threadblock_swizzle;
|
||||
|
||||
|
||||
@@ -250,7 +250,7 @@ struct ImplicitGemmConvolution {
|
||||
ThreadblockShape::kN);
|
||||
|
||||
gemm_k_iterations_per_channel = implicit_gemm_k_iterations_per_channel(
|
||||
kConvolutionalOperator, ThreadblockShape::kK, args.problem_size, kIteratorAlgorithm);
|
||||
kConvolutionalOperator, args.problem_size, kIteratorAlgorithm);
|
||||
|
||||
ThreadblockSwizzle threadblock_swizzle;
|
||||
|
||||
|
||||
@@ -95,11 +95,11 @@ struct StridedDgradHorizontalThreadblockSwizzle :
|
||||
/// Returns the shape of the problem in units of logical tiles
|
||||
/// For ImplicitGemmConvolution Conv2d problem size: conv_operator(NPQK, NHWC, KRSC)
|
||||
CUTLASS_HOST_DEVICE
|
||||
gemm::GemmCoord get_tiled_shape(
|
||||
static gemm::GemmCoord get_tiled_shape(
|
||||
cutlass::conv::Operator conv_operator,
|
||||
cutlass::conv::Conv2dProblemSize const &problem_size,
|
||||
gemm::GemmCoord tile_size,
|
||||
int split_k_slices) const {
|
||||
int split_k_slices) {
|
||||
|
||||
gemm::GemmCoord implicit_gemm_problem_size =
|
||||
cutlass::conv::implicit_gemm_problem_size(conv_operator, problem_size);
|
||||
@@ -136,11 +136,11 @@ struct StridedDgradIdentityThreadblockSwizzle :
|
||||
/// Returns the shape of the problem in units of logical tiles
|
||||
/// For ImplicitGemmConvolution Conv2d problem size: conv_operator(NPQK, NHWC, KRSC)
|
||||
CUTLASS_HOST_DEVICE
|
||||
gemm::GemmCoord get_tiled_shape(
|
||||
static gemm::GemmCoord get_tiled_shape(
|
||||
cutlass::conv::Operator conv_operator,
|
||||
cutlass::conv::Conv2dProblemSize const &problem_size,
|
||||
gemm::GemmCoord tile_size,
|
||||
int split_k_slices) const {
|
||||
int split_k_slices) {
|
||||
|
||||
gemm::GemmCoord implicit_gemm_problem_size =
|
||||
cutlass::conv::implicit_gemm_problem_size(conv_operator, problem_size);
|
||||
@@ -174,10 +174,10 @@ struct DepthwiseDirect2dConvIdentityThreadblockSwizzle
|
||||
|
||||
/// Returns the shape of the problem in units of logical tiles
|
||||
CUTLASS_HOST_DEVICE
|
||||
gemm::GemmCoord get_tiled_shape(cutlass::conv::Operator conv_operator,
|
||||
static gemm::GemmCoord get_tiled_shape(cutlass::conv::Operator conv_operator,
|
||||
cutlass::conv::Conv2dProblemSize const &problem_size,
|
||||
gemm::GemmCoord tile_size,
|
||||
int split_k_slices) const {
|
||||
int split_k_slices) {
|
||||
|
||||
gemm::GemmCoord implicit_gemm_problem_size =
|
||||
cutlass::conv::implicit_gemm_problem_size(conv_operator, problem_size);
|
||||
|
||||
Reference in New Issue
Block a user