CUTLASS v1.0 release
This commit is contained in:
@@ -0,0 +1,92 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#pragma once
|
||||
|
||||
#include <cutlass/matrix_traits.h>
|
||||
#include <tools/util/type_traits.h>
|
||||
|
||||
namespace perf {
|
||||
|
||||
/// Dispatcher for cuBLAS kernels
|
||||
template <typename AType, typename BType, typename CType, typename Accumulator, typename Scalar>
|
||||
struct CublasGemmDispatch {
|
||||
/// Type used for device-side allocations
|
||||
typedef typename cutlass::TypeTraits<AType>::device_type ADeviceType;
|
||||
typedef typename cutlass::TypeTraits<BType>::device_type BDeviceType;
|
||||
typedef typename cutlass::TypeTraits<CType>::device_type CDeviceType;
|
||||
typedef typename cutlass::TypeTraits<Accumulator>::device_type AccumulatorDeviceType;
|
||||
typedef typename cutlass::TypeTraits<Scalar>::device_type ScalarDeviceType;
|
||||
|
||||
static cublasOperation_t convert(cutlass::MatrixLayout::Kind layout) {
|
||||
switch (layout) {
|
||||
case cutlass::MatrixLayout::kRowMajor:
|
||||
return CUBLAS_OP_T;
|
||||
case cutlass::MatrixLayout::kColumnMajor:
|
||||
return CUBLAS_OP_N;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
return CUBLAS_OP_N;
|
||||
}
|
||||
|
||||
/// Launches a cuBLAS GEMM kernel
|
||||
cublasStatus_t operator()(cublasHandle_t handle,
|
||||
cutlass::MatrixLayout::Kind layout_a,
|
||||
cutlass::MatrixLayout::Kind layout_b,
|
||||
int m,
|
||||
int n,
|
||||
int k,
|
||||
Scalar alpha,
|
||||
const ADeviceType *A,
|
||||
int lda,
|
||||
const BDeviceType *B,
|
||||
int ldb,
|
||||
Scalar beta,
|
||||
CDeviceType *C,
|
||||
int ldc,
|
||||
cublasGemmAlgo_t algorithm) {
|
||||
return cublasGemmEx(handle,
|
||||
convert(layout_a),
|
||||
convert(layout_b),
|
||||
m,
|
||||
n,
|
||||
k,
|
||||
reinterpret_cast<ScalarDeviceType const *>(&alpha),
|
||||
A,
|
||||
cutlass::TypeTraits<ADeviceType>::cublas_type,
|
||||
lda,
|
||||
B,
|
||||
cutlass::TypeTraits<BDeviceType>::cublas_type,
|
||||
ldb,
|
||||
reinterpret_cast<ScalarDeviceType const *>(&beta),
|
||||
C,
|
||||
cutlass::TypeTraits<CDeviceType>::cublas_type,
|
||||
ldc,
|
||||
cutlass::TypeTraits<AccumulatorDeviceType>::cublas_type,
|
||||
algorithm);
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace perf
|
||||
@@ -0,0 +1,148 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#pragma once
|
||||
|
||||
template <typename Gemm_,
|
||||
typename Index_,
|
||||
typename ScalarA_,
|
||||
typename ScalarB_,
|
||||
typename ScalarC_,
|
||||
typename ScalarD_,
|
||||
typename Compute_,
|
||||
typename ScalarEpilogue_,
|
||||
bool ThreadMultiplyAdd_>
|
||||
struct CutlassDispatch {
|
||||
typedef typename Gemm_::Params Params;
|
||||
typedef Gemm_ Gemm;
|
||||
typedef Index_ Index;
|
||||
typedef ScalarA_ ScalarA;
|
||||
typedef ScalarB_ ScalarB;
|
||||
typedef ScalarC_ ScalarC;
|
||||
typedef ScalarD_ ScalarD;
|
||||
typedef Compute_ Compute;
|
||||
typedef ScalarEpilogue_ ScalarEpilogue;
|
||||
|
||||
static bool const kThreadMultiplyAdd = ThreadMultiplyAdd_;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Gemm::Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Gemm::Traits::kLayoutB;
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
/// Params argument
|
||||
Params params;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
CutlassDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
CutlassDispatch(Index m,
|
||||
Index n,
|
||||
Index k,
|
||||
ScalarEpilogue alpha,
|
||||
ScalarA const* d_a,
|
||||
Index lda,
|
||||
ScalarB const* d_b,
|
||||
Index ldb,
|
||||
ScalarEpilogue beta,
|
||||
ScalarC const* d_c,
|
||||
Index ldc,
|
||||
ScalarD* d_d,
|
||||
Index ldd) {
|
||||
params.initialize(m, n, k, alpha, d_a, lda, d_b, ldb, beta, d_c, ldc, d_d, ldd);
|
||||
}
|
||||
|
||||
/// Initializes params object
|
||||
CutlassDispatch(Params const& _params) : params(_params) {}
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
|
||||
/// Determines if problem is aligned (assuming no padding)
|
||||
static bool is_problem_aligned(
|
||||
int m,
|
||||
int n,
|
||||
int k) {
|
||||
|
||||
bool aligned = true;
|
||||
|
||||
if (kLayoutA == cutlass::MatrixLayout::kColumnMajor) {
|
||||
aligned = aligned && !(m % Gemm::Traits::GemmConfig::kScalarsPerLdgA);
|
||||
}
|
||||
else {
|
||||
aligned = aligned && !(k % Gemm::Traits::GemmConfig::kScalarsPerLdgA);
|
||||
}
|
||||
|
||||
if (kLayoutB == cutlass::MatrixLayout::kColumnMajor) {
|
||||
aligned = aligned && !(k % Gemm::Traits::GemmConfig::kScalarsPerLdgB);
|
||||
}
|
||||
else {
|
||||
aligned = aligned && !(n % Gemm::Traits::GemmConfig::kScalarsPerLdgB);
|
||||
}
|
||||
|
||||
aligned = aligned && !(m % Gemm::Traits::GemmConfig::kScalarsPerLdgC);
|
||||
|
||||
return aligned;
|
||||
}
|
||||
};
|
||||
|
||||
/// Basic dispatcher inferred from GEMM traits
|
||||
template <typename Traits>
|
||||
struct CutlassDispatchBasic {
|
||||
/// Gemm kernel
|
||||
typedef cutlass::gemm::Gemm<Traits> Gemm;
|
||||
|
||||
/// Index type
|
||||
typedef typename Traits::Index Index;
|
||||
|
||||
/// The scalar for A.
|
||||
typedef typename Traits::ScalarA ScalarA;
|
||||
/// The scalar for B.
|
||||
typedef typename Traits::ScalarB ScalarB;
|
||||
/// The scalar for C.
|
||||
typedef typename Traits::ScalarC ScalarC;
|
||||
/// The scalar for D.
|
||||
typedef typename Traits::ScalarD ScalarD;
|
||||
|
||||
// TODO - support alternative accumulator and scalar types
|
||||
typedef ScalarD Compute;
|
||||
typedef Compute ScalarEpilogue;
|
||||
|
||||
typedef CutlassDispatch<Gemm,
|
||||
Index,
|
||||
ScalarA,
|
||||
ScalarB,
|
||||
ScalarC,
|
||||
ScalarD,
|
||||
Compute,
|
||||
ScalarEpilogue,
|
||||
true>
|
||||
Dispatch;
|
||||
};
|
||||
@@ -0,0 +1,97 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
#include <cutlass/gemm/dgemm_traits.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_dgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef perf::GemmProfiler<double, double, double, double, double> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::DgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_nt", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::DgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_nn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::DgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_tn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::DgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_tt", options);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
@@ -0,0 +1,624 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#pragma once
|
||||
|
||||
// Standard Library includes
|
||||
#include <fstream>
|
||||
#include <ostream>
|
||||
#include <stdexcept>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
|
||||
// CUDA includes
|
||||
#include <cublas_v2.h>
|
||||
#include <curand_kernel.h>
|
||||
|
||||
// Cutlass includes
|
||||
#include <tools/test/perf/gemm/cublas_dispatch.h>
|
||||
#include <tools/test/perf/performance_result.h>
|
||||
#include <tools/test/perf/testbench_options.h>
|
||||
#include <tools/util/device_memory.h>
|
||||
#include <tools/util/type_traits.h>
|
||||
#include <tools/util/host_tensor.h>
|
||||
#include <tools/util/tensor_view_io.h>
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Kernel to determine if two tensors are equal
|
||||
template <typename Type>
|
||||
__global__ void tensor_equals(int *result,
|
||||
int dim_contiguous,
|
||||
int dim_strided,
|
||||
Type const *experimental,
|
||||
int lde,
|
||||
Type const *reference,
|
||||
int ldr) {
|
||||
typedef typename cutlass::TypeTraits<Type>::unsigned_type UnsignedType;
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
experimental += s_idx * lde + c_idx;
|
||||
reference += s_idx * ldr + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
UnsignedType exp = *reinterpret_cast<UnsignedType const *>(experimental);
|
||||
UnsignedType ref = *reinterpret_cast<UnsignedType const *>(reference);
|
||||
|
||||
if (exp != ref) {
|
||||
*result = -1;
|
||||
return;
|
||||
}
|
||||
|
||||
experimental += lde;
|
||||
reference += ldr;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Kernel to initialize tensor to uniform distribution
|
||||
template <typename T>
|
||||
__global__ void initialize_uniform(
|
||||
Distribution dist, int64_t seed, int dim_contiguous, int dim_strided, T *tensor, int ldm) {
|
||||
__shared__ curandState_t rng_state[1024];
|
||||
|
||||
uint64_t gtid = threadIdx.x + blockIdx.x * blockDim.x + blockIdx.y * gridDim.x * blockDim.x;
|
||||
|
||||
curand_init(seed, gtid, 0, &rng_state[threadIdx.x]);
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
tensor += s_idx * ldm + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
double range = dist.uniform.max - dist.uniform.min;
|
||||
|
||||
double rnd = curand_uniform(&rng_state[threadIdx.x]);
|
||||
|
||||
rnd = dist.uniform.min + range * rnd;
|
||||
|
||||
// Random values are cast to integer after scaling by a power of two to facilitate error
|
||||
// testing
|
||||
if (dist.int_scale >= 0) {
|
||||
rnd = double(int(rnd * double(1 << dist.int_scale)));
|
||||
*tensor = T(rnd / double(1 << dist.int_scale));
|
||||
} else {
|
||||
*tensor = T(rnd);
|
||||
}
|
||||
|
||||
tensor += ldm;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kernel to initialize tensor to uniform distribution
|
||||
template <typename T>
|
||||
__global__ void initialize_gaussian(
|
||||
Distribution dist, int64_t seed, int dim_contiguous, int dim_strided, T *tensor, int ldm) {
|
||||
__shared__ curandState_t rng_state[1024];
|
||||
|
||||
uint64_t gtid = threadIdx.x + blockIdx.x * blockDim.x + blockIdx.y * gridDim.x * blockDim.x;
|
||||
|
||||
curand_init(seed, gtid, 0, &rng_state[threadIdx.x]);
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
tensor += s_idx * ldm + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
// Random values are cast to integer after scaling by a power of two to facilitate error
|
||||
// testing
|
||||
|
||||
double rnd = curand_normal(&rng_state[threadIdx.x]);
|
||||
|
||||
rnd = dist.gaussian.mean + dist.gaussian.stddev * rnd;
|
||||
|
||||
if (dist.int_scale >= 0) {
|
||||
rnd = double(int(rnd * double(1 << dist.int_scale)));
|
||||
*tensor = T(rnd / double(1 << dist.int_scale));
|
||||
} else {
|
||||
*tensor = T(rnd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kernel to initialize tensor to an identity matrix
|
||||
template <typename T>
|
||||
__global__ void initialize_linear(
|
||||
Distribution dist, int64_t seed, int dim_contiguous, int dim_strided, T *tensor, int ldm) {
|
||||
__shared__ curandState_t rng_state[1024];
|
||||
|
||||
uint64_t gtid = threadIdx.x + blockIdx.x * blockDim.x + blockIdx.y * gridDim.x * blockDim.x;
|
||||
|
||||
curand_init(seed, gtid, 0, &rng_state[threadIdx.x]);
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
tensor += s_idx * ldm + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
*tensor =
|
||||
dist.linear.offset + dist.linear.delta_row * c_idx + dist.linear.delta_column * s_idx;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kernel to initialize tensor to an identity matrix
|
||||
template <typename T>
|
||||
__global__ void initialize_identity(
|
||||
Distribution dist, int64_t seed, int dim_contiguous, int dim_strided, T *tensor, int ldm) {
|
||||
__shared__ curandState_t rng_state[1024];
|
||||
|
||||
uint64_t gtid = threadIdx.x + blockIdx.x * blockDim.x + blockIdx.y * gridDim.x * blockDim.x;
|
||||
|
||||
curand_init(seed, gtid, 0, &rng_state[threadIdx.x]);
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
tensor += s_idx * ldm + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
*tensor = (c_idx == s_idx ? T(1) : T(0));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Dispatcher to appropriate initialization kernel
|
||||
template <typename T>
|
||||
inline void initialize(Distribution const &dist,
|
||||
int64_t seed,
|
||||
int dim_contiguous,
|
||||
int dim_strided,
|
||||
T *tensor,
|
||||
int ldm) {
|
||||
dim3 block(256, 1, 1);
|
||||
dim3 grid((dim_contiguous + block.x - 1) / block.x, (dim_strided + block.x - 1) / block.x);
|
||||
|
||||
switch (dist.kind) {
|
||||
case Distribution::Uniform:
|
||||
initialize_uniform<<<grid, block>>>(dist, seed, dim_contiguous, dim_strided, tensor, ldm);
|
||||
break;
|
||||
case Distribution::Gaussian:
|
||||
initialize_gaussian<<<grid, block>>>(dist, seed, dim_contiguous, dim_strided, tensor, ldm);
|
||||
break;
|
||||
case Distribution::Linear:
|
||||
initialize_linear<<<grid, block>>>(dist, seed, dim_contiguous, dim_strided, tensor, ldm);
|
||||
break;
|
||||
case Distribution::Identity:
|
||||
initialize_identity<<<grid, block>>>(dist, seed, dim_contiguous, dim_strided, tensor, ldm);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Host-side implementation of performance testbed
|
||||
template <typename AType, typename BType, typename CType, typename Accumulator, typename Scalar>
|
||||
class GemmTestbed {
|
||||
public:
|
||||
/// Type used for device-side allocations
|
||||
typedef typename cutlass::TypeTraits<AType>::device_type ADeviceType;
|
||||
typedef typename cutlass::TypeTraits<BType>::device_type BDeviceType;
|
||||
typedef typename cutlass::TypeTraits<CType>::device_type CDeviceType;
|
||||
typedef typename cutlass::TypeTraits<Accumulator>::device_type AccumulatorDeviceType;
|
||||
typedef typename cutlass::TypeTraits<Scalar>::device_type ScalarDeviceType;
|
||||
|
||||
/// Dispatch object to cuBLAS GEMM
|
||||
typedef CublasGemmDispatch<AType, BType, CType, Accumulator, Scalar> CublasDispatch;
|
||||
|
||||
//
|
||||
// Type definitions
|
||||
//
|
||||
|
||||
/// Host tensor for operand A
|
||||
typedef cutlass::device_memory::allocation<ADeviceType> TensorA;
|
||||
|
||||
/// Host tensor for operand B
|
||||
typedef cutlass::device_memory::allocation<BDeviceType> TensorB;
|
||||
|
||||
/// Host tensor for operand C
|
||||
typedef cutlass::device_memory::allocation<CDeviceType> TensorC;
|
||||
|
||||
private:
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
InitialDistribution initial_distribution;
|
||||
|
||||
/// Status
|
||||
cublasStatus_t status;
|
||||
|
||||
/// cuBLAS handle
|
||||
cublasHandle_t handle;
|
||||
|
||||
/// GEMM problem
|
||||
GemmProblem problem;
|
||||
|
||||
/// A matrix operand
|
||||
TensorA A;
|
||||
|
||||
/// B matrix operand
|
||||
TensorB B;
|
||||
|
||||
/// C matrix operand
|
||||
TensorC C_initial;
|
||||
|
||||
/// Reference result
|
||||
TensorC reference;
|
||||
|
||||
/// Experimental result
|
||||
TensorC experimental;
|
||||
|
||||
private:
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
/// Helper to resize a matrix with a given size and layout if needed
|
||||
template <typename T>
|
||||
static void resize_device_allocation(
|
||||
cutlass::device_memory::allocation<T> &tensor,
|
||||
Distribution const &dist,
|
||||
int64_t seed,
|
||||
int rows,
|
||||
int columns,
|
||||
cutlass::MatrixLayout::Kind layout,
|
||||
int ldm = 0) {
|
||||
if (!ldm) {
|
||||
ldm = (layout == cutlass::MatrixLayout::kColumnMajor ? rows : columns);
|
||||
}
|
||||
|
||||
size_t capacity = ldm * (layout == cutlass::MatrixLayout::kColumnMajor ? columns : rows);
|
||||
|
||||
if (capacity > tensor.capacity) {
|
||||
tensor.reset(cutlass::device_memory::allocate<T>(capacity), capacity);
|
||||
|
||||
int c_dim = (layout == cutlass::MatrixLayout::kColumnMajor ? rows : columns);
|
||||
int s_dim = (layout == cutlass::MatrixLayout::kColumnMajor ? columns : rows);
|
||||
|
||||
initialize(dist, seed, c_dim, s_dim, tensor.get(), ldm);
|
||||
}
|
||||
}
|
||||
|
||||
/// Resizes each tensor
|
||||
void resize_helper(GemmProblem const &problem) {
|
||||
resize_device_allocation(
|
||||
A,
|
||||
initial_distribution.dist_A,
|
||||
initial_distribution.seed,
|
||||
problem.m,
|
||||
problem.k,
|
||||
problem.layout_A);
|
||||
|
||||
resize_device_allocation(
|
||||
B,
|
||||
initial_distribution.dist_B,
|
||||
initial_distribution.seed + 17, // compute distinct value from initial seed
|
||||
problem.k,
|
||||
problem.n,
|
||||
problem.layout_B);
|
||||
|
||||
resize_device_allocation(
|
||||
C_initial,
|
||||
initial_distribution.dist_C,
|
||||
initial_distribution.seed + 101, // compute distinct value from initial seed
|
||||
problem.m,
|
||||
problem.n,
|
||||
cutlass::MatrixLayout::kColumnMajor);
|
||||
|
||||
resize_device_allocation(
|
||||
reference, Distribution(), 0, problem.m, problem.n, cutlass::MatrixLayout::kColumnMajor);
|
||||
|
||||
resize_device_allocation(
|
||||
experimental, Distribution(), 0, problem.m, problem.n, cutlass::MatrixLayout::kColumnMajor);
|
||||
}
|
||||
|
||||
/// Functor to print errors
|
||||
struct PrintErrors {
|
||||
|
||||
/// Equivalently sized integer type
|
||||
typedef typename cutlass::TypeTraits<CType>::integer_type integer_t;
|
||||
|
||||
/// Output stream to write to
|
||||
std::ostream& out;
|
||||
|
||||
/// Reference tensor view
|
||||
cutlass::HostTensorView<CType> const& reference;
|
||||
|
||||
/// Computed tensor view
|
||||
cutlass::HostTensorView<CType> const& experimental;
|
||||
|
||||
/// Errors greater than or this amount result in printing
|
||||
integer_t ulps_threshold;
|
||||
|
||||
///
|
||||
PrintErrors(std::ostream& _out,
|
||||
cutlass::HostTensorView<CType> const& _reference,
|
||||
cutlass::HostTensorView<CType> const& _experimental,
|
||||
integer_t _ulps_threshold = 1)
|
||||
: out(_out),
|
||||
reference(_reference),
|
||||
experimental(_experimental),
|
||||
ulps_threshold(_ulps_threshold) {}
|
||||
|
||||
/// Compares one element
|
||||
void operator()(
|
||||
CType const& element,
|
||||
typename cutlass::HostTensorView<CType>::Coord_t coord) {
|
||||
|
||||
CType exp = experimental.at(coord);
|
||||
CType ref = reference.at(coord);
|
||||
|
||||
int64_t int_exp = 0;
|
||||
int64_t int_ref = 0;
|
||||
|
||||
*reinterpret_cast<CType*>(&int_exp) = exp;
|
||||
*reinterpret_cast<CType*>(&int_ref) = ref;
|
||||
|
||||
integer_t ulps = integer_t(int_exp - int_ref);
|
||||
|
||||
if (std::abs(ulps) >= ulps_threshold) {
|
||||
// width in hexadecimal digits of value
|
||||
int const width = sizeof(integer_t) * 2;
|
||||
|
||||
double relative = double(exp) - double(ref);
|
||||
if (ref != CType(0)) {
|
||||
relative /= double(ref);
|
||||
}
|
||||
|
||||
out << "[" << coord << "] expected: " << ref << " (0x"
|
||||
<< std::hex << std::setw(width) << std::setfill('0') << integer_t(int_ref) << std::dec
|
||||
<< ")"
|
||||
<< ", got: " << exp << " (0x" << std::hex
|
||||
<< std::setw(width) << std::setfill('0') << integer_t(int_exp) << std::dec << ")"
|
||||
<< " relative error: " << relative << ", ulps: " << ulps << "\n";
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
public:
|
||||
/// Resizes tensors to accommodate the given problem
|
||||
void resize(GemmProblem const &_problem) {
|
||||
problem = _problem;
|
||||
|
||||
try {
|
||||
resize_helper(problem);
|
||||
} catch (...) {
|
||||
// If out of memory, clear each allocation then allocate again
|
||||
A.reset();
|
||||
B.reset();
|
||||
C_initial.reset();
|
||||
reference.reset();
|
||||
experimental.reset();
|
||||
|
||||
resize_helper(problem);
|
||||
}
|
||||
}
|
||||
|
||||
/// Constructs a basic workspace
|
||||
GemmTestbed(InitialDistribution const &_dist = InitialDistribution())
|
||||
: initial_distribution(_dist) {
|
||||
status = cublasCreate(&handle);
|
||||
if (status != CUBLAS_STATUS_SUCCESS) {
|
||||
throw cutlass::cuda_exception("Failed to create CUBLAS handle");
|
||||
}
|
||||
}
|
||||
|
||||
/// Constructs a workspace for verifying GEMM, assumes
|
||||
/// dense packing.
|
||||
GemmTestbed(GemmProblem const &_problem,
|
||||
cublasGemmAlgo_t algorithm_ = CUBLAS_GEMM_DEFAULT,
|
||||
InitialDistribution const &_dist = InitialDistribution())
|
||||
: problem(_problem), initial_distribution(_dist) {
|
||||
status = cublasCreate(&handle);
|
||||
if (status != CUBLAS_STATUS_SUCCESS) {
|
||||
throw cutlass::cuda_exception("Failed to create CUBLAS handle");
|
||||
}
|
||||
|
||||
resize(problem);
|
||||
}
|
||||
|
||||
~GemmTestbed() { status = cublasDestroy(handle); }
|
||||
|
||||
/// Returns true if the last CUBLAS call returned successfully
|
||||
bool good() const { return status == CUBLAS_STATUS_SUCCESS; }
|
||||
|
||||
/// Rows of GEMM problem
|
||||
int M() const { return problem.m; }
|
||||
|
||||
/// Columns of GEMM problem
|
||||
int N() const { return problem.n; }
|
||||
|
||||
/// Inner dimension of GEMM problem
|
||||
int K() const { return problem.k; }
|
||||
|
||||
/// Returns a pointer to the A operand
|
||||
ADeviceType *ptr_A() const { return A.get(); }
|
||||
|
||||
/// Leading dimension of A
|
||||
int lda() const { return problem.lda(); }
|
||||
|
||||
/// Returns a pointer to the B operand
|
||||
BDeviceType *ptr_B() const { return B.get(); }
|
||||
|
||||
/// Leading dimension of B
|
||||
int ldb() const { return problem.ldb(); }
|
||||
|
||||
/// Returns a pointer to the initial state of the result tensor in device memory
|
||||
CDeviceType *ptr_C_initial() const { return C_initial.get(); }
|
||||
|
||||
/// Leading dimension of C
|
||||
int ldc() const { return problem.ldc(); }
|
||||
|
||||
/// Returns a pointer to the result tensor in device memory
|
||||
CDeviceType *ptr_experimental() const { return experimental.get(); }
|
||||
|
||||
/// Returns a pointer to the result tensor in device memory
|
||||
CDeviceType *ptr_reference() const { return reference.get(); }
|
||||
|
||||
/// Returns the number of flops implied by the computation (1 multiply-accumulate = 2 flops)
|
||||
uint64_t flops() const {
|
||||
return uint64_t(problem.m) * uint64_t(problem.n) * uint64_t(problem.k) * 2ULL;
|
||||
}
|
||||
|
||||
/// Computes the speed of the computation in GFLOPs/s
|
||||
double GFLOPs_per_sec(double runtime_ms) const { return double(flops()) / runtime_ms / 1.0e6; }
|
||||
|
||||
/// Matrix layout of A
|
||||
cutlass::MatrixLayout::Kind layout_a() const { return problem.layout_A; }
|
||||
|
||||
/// Matrix layout of B
|
||||
cutlass::MatrixLayout::Kind layout_b() const { return problem.layout_B; }
|
||||
|
||||
/// Returns alpha scalar
|
||||
Scalar alpha() const { return Scalar(problem.alpha); }
|
||||
|
||||
/// Returns alpha scalar
|
||||
Scalar beta() const { return Scalar(problem.beta); }
|
||||
|
||||
/// Initializes C matrix by copying from C_initial
|
||||
void prepare_gemm(CDeviceType *target) {
|
||||
size_t count = ldc() * problem.n;
|
||||
cutlass::device_memory::copy_device_to_device(target, ptr_C_initial(), count);
|
||||
}
|
||||
|
||||
/// Initializes output matrix of cublas
|
||||
void prepare_cublas() { prepare_gemm(ptr_reference()); }
|
||||
|
||||
/// Initializes output matrix of cublas
|
||||
void prepare_experimental() { prepare_gemm(ptr_experimental()); }
|
||||
|
||||
/// Launches the cuBLAS GEMM - does not initialize output matrix
|
||||
cublasStatus_t launch_cublas(cublasGemmAlgo_t algo) {
|
||||
CublasDispatch dispatch;
|
||||
|
||||
Scalar alpha(Scalar(problem.alpha));
|
||||
Scalar beta(Scalar(problem.beta));
|
||||
|
||||
status = dispatch(handle,
|
||||
problem.layout_A,
|
||||
problem.layout_B,
|
||||
problem.m,
|
||||
problem.n,
|
||||
problem.k,
|
||||
alpha,
|
||||
ptr_A(),
|
||||
lda(),
|
||||
ptr_B(),
|
||||
ldb(),
|
||||
beta,
|
||||
ptr_reference(),
|
||||
ldc(),
|
||||
algo);
|
||||
|
||||
return status;
|
||||
}
|
||||
|
||||
/// Verifies the 'test' tensor with 'ref'
|
||||
bool verify(TensorC const &test, TensorC const &ref) {
|
||||
cutlass::device_memory::allocation<int> flag_device(1);
|
||||
|
||||
int flag = 0;
|
||||
cutlass::device_memory::copy_to_device(flag_device.get(), &flag, 1);
|
||||
|
||||
dim3 block(256, 1, 1);
|
||||
dim3 grid((problem.m + block.x - 1) / block.x, (problem.n + block.x - 1) / block.x);
|
||||
|
||||
tensor_equals<CDeviceType><<<grid, block>>>(flag_device.get(),
|
||||
problem.m,
|
||||
problem.n,
|
||||
experimental.get(),
|
||||
problem.m,
|
||||
reference.get(),
|
||||
problem.m);
|
||||
|
||||
cutlass::device_memory::copy_to_host(&flag, flag_device.get(), 1);
|
||||
|
||||
return flag == 0;
|
||||
}
|
||||
|
||||
/// Computes the reference output
|
||||
void compute_reference(cublasGemmAlgo_t algorithm) {
|
||||
prepare_cublas();
|
||||
launch_cublas(algorithm);
|
||||
}
|
||||
|
||||
/// Helper to verify with reference
|
||||
bool verify_with_reference() { return verify(experimental, reference); }
|
||||
|
||||
/// Writes the problem to an ostream in human-readable form
|
||||
void write_problem(std::ostream &results_output, std::ostream &errors_output) {
|
||||
|
||||
cutlass::HostTensor<AType, false> host_A;
|
||||
cutlass::HostTensor<BType, false> host_B;
|
||||
cutlass::HostTensor<CType, false> host_C;
|
||||
cutlass::HostTensor<CType, false> host_D;
|
||||
cutlass::HostTensor<CType, false> host_Ref;
|
||||
|
||||
host_A.resize_matrix(M(), K(), layout_a());
|
||||
host_B.resize_matrix(K(), N(), layout_b());
|
||||
host_C.resize_matrix(M(), N(), cutlass::MatrixLayout::kColumnMajor);
|
||||
host_D.resize_matrix(M(), N(), cutlass::MatrixLayout::kColumnMajor);
|
||||
host_Ref.resize_matrix(M(), N(), cutlass::MatrixLayout::kColumnMajor);
|
||||
|
||||
// copy from device allocations
|
||||
host_A.copy_to_host(ptr_A());
|
||||
host_B.copy_to_host(ptr_B());
|
||||
host_C.copy_to_host(ptr_C_initial());
|
||||
host_D.copy_to_host(ptr_experimental());
|
||||
host_Ref.copy_to_host(ptr_reference());
|
||||
|
||||
// write out human readable
|
||||
results_output << "A =\n" << host_A << "\n"
|
||||
<< "B =\n" << host_B << "\n"
|
||||
<< "C = \n" << host_C << "\n"
|
||||
<< "Ref =\n" << host_Ref << "\n"
|
||||
<< "Experimental =\n" << host_D << "\n";
|
||||
|
||||
// write out list of errors
|
||||
PrintErrors printer(errors_output, host_Ref, host_D);
|
||||
|
||||
host_D.visit(printer);
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace perf
|
||||
@@ -0,0 +1,343 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#pragma once
|
||||
|
||||
#include <fstream>
|
||||
#include <map>
|
||||
#include <stdexcept>
|
||||
#include <utility>
|
||||
|
||||
#if defined(WIN32)
|
||||
#include <Windows.h>
|
||||
#else
|
||||
// needed for sleep
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
#include <tools/test/perf/testbench_options.h>
|
||||
#include <tools/test/perf/testbench_output.h>
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Performance measuring testbed
|
||||
template <typename AType,
|
||||
typename BType,
|
||||
typename CType,
|
||||
typename AccumulatorType,
|
||||
typename ScalarType>
|
||||
class GemmProfiler {
|
||||
public:
|
||||
/// Test environment
|
||||
typedef GemmTestbed<AType, BType, CType, AccumulatorType, ScalarType> PerfTestbed;
|
||||
|
||||
private:
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
/// Reference to TestbenchOutput instance
|
||||
TestbenchOutput &output;
|
||||
|
||||
/// Reference to options object
|
||||
TestbenchOptions const &options;
|
||||
|
||||
/// Performance test environment
|
||||
PerfTestbed testbed;
|
||||
|
||||
/// Kernel name
|
||||
std::string kernel_name;
|
||||
|
||||
/// Timing events
|
||||
cudaEvent_t events[2];
|
||||
|
||||
public:
|
||||
/// Delays
|
||||
static void pause(int seconds) {
|
||||
#if defined(WIN32)
|
||||
Sleep(1000 * seconds);
|
||||
#else
|
||||
sleep(seconds);
|
||||
#endif
|
||||
}
|
||||
|
||||
public:
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
/// Constructs performance testebed
|
||||
GemmProfiler(TestbenchOutput &_output,
|
||||
std::string const &_kernel_name,
|
||||
TestbenchOptions const &_options)
|
||||
: output(_output),
|
||||
options(_options),
|
||||
kernel_name(_kernel_name),
|
||||
testbed(_options.initial_distribution) {
|
||||
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
cudaError_t result = cudaEventCreate(&events[i]);
|
||||
if (result != cudaSuccess) {
|
||||
throw std::runtime_error("GemmPerfTestbed() failed to create CUDA events");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
~GemmProfiler() {}
|
||||
|
||||
/// Writes the workspace to text files
|
||||
void write_problem(std::string const &kernel_name) {
|
||||
|
||||
std::stringstream base_filename;
|
||||
|
||||
base_filename
|
||||
<< kernel_name << "_"
|
||||
<< testbed.M() << "x" << testbed.N() << "x" << testbed.K();
|
||||
|
||||
std::string results_name = base_filename.str() + "_results.txt";
|
||||
std::string errors_name = base_filename.str() + "_errors.txt";
|
||||
|
||||
std::ofstream results(results_name.c_str());
|
||||
std::ofstream errors(errors_name.c_str());
|
||||
testbed.write_problem(results, errors);
|
||||
}
|
||||
|
||||
/// Profiles Cutlass
|
||||
template <typename CutlassDispatch>
|
||||
PerformanceResult execute_cutlass(GemmProblem const &problem, cublasGemmAlgo_t algorithm) {
|
||||
PerformanceResult result(kernel_name, problem);
|
||||
|
||||
testbed.compute_reference(algorithm);
|
||||
|
||||
if (cudaDeviceSynchronize() != cudaSuccess) {
|
||||
result.disposition = Disposition::NotVerified;
|
||||
return result;
|
||||
}
|
||||
|
||||
CutlassDispatch dispatch(testbed.M(),
|
||||
testbed.N(),
|
||||
testbed.K(),
|
||||
testbed.alpha(),
|
||||
testbed.ptr_A(),
|
||||
testbed.lda(),
|
||||
testbed.ptr_B(),
|
||||
testbed.ldb(),
|
||||
testbed.beta(),
|
||||
testbed.ptr_C_initial(),
|
||||
testbed.ldc(),
|
||||
testbed.ptr_experimental(),
|
||||
testbed.ldc());
|
||||
|
||||
dispatch();
|
||||
|
||||
if (cudaDeviceSynchronize() != cudaSuccess) {
|
||||
result.disposition = Disposition::Failed;
|
||||
return result;
|
||||
}
|
||||
|
||||
if (testbed.verify_with_reference()) {
|
||||
result.disposition = Disposition::Passed;
|
||||
} else {
|
||||
result.disposition = Disposition::Incorrect;
|
||||
}
|
||||
|
||||
if (options.save_workspace(result.disposition == Disposition::Passed)) {
|
||||
write_problem(kernel_name);
|
||||
}
|
||||
|
||||
if (cudaDeviceSynchronize() != cudaSuccess) {
|
||||
result.disposition = Disposition::Failed;
|
||||
}
|
||||
|
||||
// warmup launch
|
||||
dispatch();
|
||||
|
||||
if (cudaDeviceSynchronize() != cudaSuccess) {
|
||||
result.disposition = Disposition::Failed;
|
||||
return result;
|
||||
}
|
||||
|
||||
if (cudaEventRecord(events[0]) != cudaSuccess) {
|
||||
result.disposition = Disposition::Failed;
|
||||
return result;
|
||||
}
|
||||
|
||||
for (int iter = 0; iter < options.iterations; ++iter) {
|
||||
dispatch();
|
||||
}
|
||||
|
||||
if (cudaEventRecord(events[1]) != cudaSuccess) {
|
||||
result.disposition = Disposition::Failed;
|
||||
return result;
|
||||
}
|
||||
|
||||
if (cudaEventSynchronize(events[1]) != cudaSuccess) {
|
||||
result.disposition = Disposition::Failed;
|
||||
return result;
|
||||
}
|
||||
|
||||
float average_ms = 0;
|
||||
if (cudaEventElapsedTime(&average_ms, events[0], events[1]) != cudaSuccess) {
|
||||
result.disposition = Disposition::Failed;
|
||||
return result;
|
||||
}
|
||||
|
||||
result.runtime = double(average_ms) / double(options.iterations);
|
||||
result.gflops = testbed.GFLOPs_per_sec(result.runtime);
|
||||
|
||||
if (result.disposition != Disposition::Passed) {
|
||||
std::cout << kernel_name << " failed with disposition: " << result.disposition;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/// Executes all kernels for this problem size
|
||||
template <typename CutlassDispatch>
|
||||
std::vector<PerformanceResult> execute(GemmProblem const &problem) {
|
||||
|
||||
// New problem size
|
||||
output.begin_problem();
|
||||
|
||||
cublasGemmAlgo_t algorithm =
|
||||
(CutlassDispatch::kThreadMultiplyAdd ? CUBLAS_GEMM_DEFAULT : CUBLAS_GEMM_DEFAULT_TENSOR_OP);
|
||||
|
||||
testbed.resize(problem);
|
||||
|
||||
std::vector<PerformanceResult> results;
|
||||
|
||||
results.push_back(execute_cutlass<CutlassDispatch>(problem, algorithm));
|
||||
|
||||
// cool-down period
|
||||
pause(2);
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
/// Runs the test and collects performance for all results
|
||||
template <typename CutlassDispatch>
|
||||
void schmoo(Range const &M, Range const &N, Range const &K) {
|
||||
for (int m = M.start; m <= M.end; m += M.increment) {
|
||||
for (int n = N.start; n <= N.end; n += N.increment) {
|
||||
for (int k = K.start; k <= K.end; k += K.increment) {
|
||||
|
||||
// Avoid evaluating problem if problem size does not satisfy alignment
|
||||
if (!CutlassDispatch::is_problem_aligned(m, n, k)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
std::vector<PerformanceResult> results =
|
||||
execute<CutlassDispatch>(GemmProblem(m,
|
||||
n,
|
||||
k,
|
||||
CutlassDispatch::kLayoutA,
|
||||
CutlassDispatch::kLayoutB,
|
||||
options.alpha,
|
||||
options.beta));
|
||||
|
||||
for (std::vector<PerformanceResult>::const_iterator it = results.begin();
|
||||
it != results.end();
|
||||
++it) {
|
||||
output.append(*it);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Runs the test over the problem space and reports only the best performance
|
||||
template <typename CutlassDispatch>
|
||||
void peak(Range const &M, Range const &N, Range const &K) {
|
||||
|
||||
PerformanceResult max_perf;
|
||||
bool first_result = true;
|
||||
|
||||
for (int m = M.start; m <= M.end; m += M.increment) {
|
||||
for (int n = N.start; n <= N.end; n += N.increment) {
|
||||
for (int k = K.start; k <= K.end; k += K.increment) {
|
||||
|
||||
// Avoid evaluating problem if problem size does not satisfy alignment
|
||||
if (!CutlassDispatch::is_problem_aligned(m, n, k)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
std::vector<PerformanceResult> results =
|
||||
execute<CutlassDispatch>(GemmProblem(m,
|
||||
n,
|
||||
k,
|
||||
CutlassDispatch::kLayoutA,
|
||||
CutlassDispatch::kLayoutB,
|
||||
options.alpha,
|
||||
options.beta));
|
||||
|
||||
for (std::vector<PerformanceResult>::const_iterator it = results.begin();
|
||||
it != results.end();
|
||||
++it) {
|
||||
|
||||
/// Writes the output without appending it
|
||||
output.pretty_print(*it);
|
||||
|
||||
/// Updates maximum performing kernel
|
||||
if (first_result || max_perf.gflops > it->gflops) {
|
||||
max_perf = *it;
|
||||
}
|
||||
first_result = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
output.append(max_perf);
|
||||
}
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Dispatches to GEMM performance profiler
|
||||
template <typename Dispatch, typename GemmProfiler>
|
||||
int profile_gemm(TestbenchOutput &output,
|
||||
std::string const &kernel,
|
||||
TestbenchOptions const &options) {
|
||||
if (options.kernel_enabled(kernel)) {
|
||||
GemmProfiler perf(output, kernel, options);
|
||||
if (options.peak_performance) {
|
||||
perf.template peak<Dispatch>(
|
||||
options.problem_range.M, options.problem_range.N, options.problem_range.K);
|
||||
} else {
|
||||
perf.template schmoo<Dispatch>(
|
||||
options.problem_range.M, options.problem_range.N, options.problem_range.K);
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
@@ -0,0 +1,113 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
#include <cutlass/gemm/hgemm_traits.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_hgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef perf::GemmProfiler<
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
>
|
||||
GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_nt", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
>
|
||||
GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_nn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
>
|
||||
GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_tn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
>
|
||||
GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_tt", options);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
#include <cutlass/gemm/igemm_traits.h>
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_igemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef perf::GemmProfiler<int8_t, int8_t, int, int, int> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::IgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "igemm_nt", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::IgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "igemm_nn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::IgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "igemm_tn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::IgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "igemm_tt", options);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
@@ -0,0 +1,101 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
#include <cutlass/gemm/sgemm_traits.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_sgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef perf::GemmProfiler<float, float, float, float, float> SGemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::SgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, SGemmProfiler>(output, "sgemm_nt", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::SgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, SGemmProfiler>(output, "sgemm_nn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::SgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, SGemmProfiler>(output, "sgemm_tn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::SgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, SGemmProfiler>(output, "sgemm_tt", options);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#include <cutlass/wmma_matrix.h>
|
||||
#ifdef CUTLASS_USE_WMMA_API
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
#include <cutlass/gemm/wmma_gemm_traits.h>
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template <typename Traits>
|
||||
struct WmmaGemmDispatch {
|
||||
|
||||
typedef cutlass::gemm::Gemm<Traits> Gemm;
|
||||
|
||||
typedef typename Gemm::Params Params;
|
||||
|
||||
/// Indicate warp-level GEMM
|
||||
static bool const kThreadMultiplyAdd = false;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Traits::kLayoutB;
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
/// Params argument
|
||||
Params params;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
WmmaGemmDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaGemmDispatch(int m, int n, int k, float alpha, half const* d_a, int lda,
|
||||
half const* d_b, int ldb, float beta, float const* d_c, int ldc,
|
||||
float* d_d, int ldd) {
|
||||
|
||||
params.initialize(m, n, k, alpha, d_a, lda, d_b, ldb, beta, d_c, ldc, d_d, ldd);
|
||||
}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaGemmDispatch(Params const& _params) : params(_params) {}
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
|
||||
/// Determines if problem is aligned (assuming no padding)
|
||||
static bool is_problem_aligned(
|
||||
int m,
|
||||
int n,
|
||||
int k) {
|
||||
|
||||
bool aligned = true;
|
||||
|
||||
if (kLayoutA == cutlass::MatrixLayout::kColumnMajor) {
|
||||
aligned = aligned && !(m % Gemm::Traits::GemmConfig::kScalarsPerLdgA);
|
||||
}
|
||||
else {
|
||||
aligned = aligned && !(k % Gemm::Traits::GemmConfig::kScalarsPerLdgA);
|
||||
}
|
||||
|
||||
if (kLayoutB == cutlass::MatrixLayout::kColumnMajor) {
|
||||
aligned = aligned && !(k % Gemm::Traits::GemmConfig::kScalarsPerLdgB);
|
||||
}
|
||||
else {
|
||||
aligned = aligned && !(n % Gemm::Traits::GemmConfig::kScalarsPerLdgB);
|
||||
}
|
||||
|
||||
aligned = aligned && !(m % Gemm::Traits::GemmConfig::kScalarsPerLdgC);
|
||||
|
||||
return aligned;
|
||||
}
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_wmma_gemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef perf::GemmProfiler<cutlass::half_t, cutlass::half_t, float, float, float> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_nt", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_nn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_tn", options);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_tt", options);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#endif // defined CUTLASS_USE_WMMA_API
|
||||
Reference in New Issue
Block a user