Checkpointing CUTLASS 1.1 release.
This commit is contained in:
@@ -34,12 +34,14 @@ set(CUTLASS_PERF_TEST_HEADERS
|
||||
)
|
||||
|
||||
set(CUTLASS_PERF_TEST_SOURCES
|
||||
cutlass_perf_test.cpp
|
||||
cutlass_perf_test.cu
|
||||
gemm/sgemm.cu
|
||||
gemm/dgemm.cu
|
||||
gemm/hgemm.cu
|
||||
gemm/igemm.cu
|
||||
gemm/wmma_gemm.cu
|
||||
gemm/wmma_binary_gemm.cu
|
||||
gemm/wmma_integer_gemm.cu
|
||||
)
|
||||
|
||||
source_group("Source\ Files" FILES ${CUTLASS_PERF_TEST_SOURCES})
|
||||
@@ -56,4 +58,6 @@ cutlass_add_executable(
|
||||
${CUTLASS_PERF_TEST_SOURCES}
|
||||
${CUTLASS_PERF_TEST_HEADERS}
|
||||
)
|
||||
CUDA_ADD_CUBLAS_TO_TARGET(cutlass_perf_test)
|
||||
|
||||
target_link_libraries(cutlass_perf_test ${CUBLAS_LIBRARY})
|
||||
|
||||
|
||||
@@ -27,19 +27,24 @@
|
||||
\brief CUTLASS Performance Tests
|
||||
*/
|
||||
|
||||
#include <tools/test/perf/testbench_options.h>
|
||||
#include <tools/test/perf/testbench_output.h>
|
||||
#include <vector>
|
||||
#include "tools/test/perf/performance_result.h"
|
||||
#include "tools/test/perf/testbench_configs.h"
|
||||
#include "tools/test/perf/testbench_options.h"
|
||||
#include "tools/test/perf/testbench_output.h"
|
||||
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
|
||||
static std::vector<perf::GemmProfileFunc*> GemmProfileFuncs;
|
||||
|
||||
//
|
||||
// Profiling entry points defined in corresponding .cu files
|
||||
//
|
||||
namespace perf {
|
||||
|
||||
int profile_sgemm(TestbenchOutput &output, TestbenchOptions const &options);
|
||||
int profile_dgemm(TestbenchOutput &output, TestbenchOptions const &options);
|
||||
int profile_hgemm(TestbenchOutput &output, TestbenchOptions const &options);
|
||||
int profile_igemm(TestbenchOutput &output, TestbenchOptions const &options);
|
||||
int profile_wmma_gemm(TestbenchOutput &output, TestbenchOptions const &options);
|
||||
void RegisterGemmProfileFunc(GemmProfileFunc * profileFunc) {
|
||||
GemmProfileFuncs.push_back(profileFunc);
|
||||
}
|
||||
|
||||
} // namespace perf
|
||||
|
||||
@@ -47,6 +52,22 @@ int profile_wmma_gemm(TestbenchOutput &output, TestbenchOptions const &options);
|
||||
// Executes profiling functionality
|
||||
//
|
||||
|
||||
template <typename Problem>
|
||||
int profile(int (**functions)(perf::TestbenchOutput<Problem> &,
|
||||
perf::TestbenchOptions const &,
|
||||
perf::Config const &),
|
||||
perf::TestbenchOutput<Problem> &output,
|
||||
perf::TestbenchOptions options,
|
||||
int result) {
|
||||
perf::TestbenchConfigs test_configs(options);
|
||||
for (size_t j = 0; !result && j < test_configs.configs.size(); j++) {
|
||||
for (size_t i = 0; !result && functions[i] != 0; ++i) {
|
||||
result = (functions[i])(output, options, test_configs.configs[j]);
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/// Entry point to CUTLASS performance test
|
||||
int main(int argc, const char **argv) {
|
||||
cutlass::CommandLine args(argc, argv);
|
||||
@@ -57,20 +78,17 @@ int main(int argc, const char **argv) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
perf::TestbenchOutput output(options);
|
||||
|
||||
int (*profile_gemm[])(perf::TestbenchOutput &, perf::TestbenchOptions const &) = {
|
||||
perf::profile_sgemm,
|
||||
perf::profile_dgemm,
|
||||
perf::profile_hgemm,
|
||||
perf::profile_igemm,
|
||||
perf::profile_wmma_gemm,
|
||||
0};
|
||||
|
||||
int result = 0;
|
||||
for (int i = 0; !result && profile_gemm[i]; ++i) {
|
||||
result = (profile_gemm[i])(output, options);
|
||||
if (args.check_cmd_line_flag("version")) {
|
||||
perf::TestbenchOptions::version(std::cout);
|
||||
std::cout << std::endl;
|
||||
return 0;
|
||||
}
|
||||
|
||||
return result;
|
||||
int result = 0;
|
||||
|
||||
std::vector<perf::GemmProfileFunc*> profileFuncs = GemmProfileFuncs;
|
||||
profileFuncs.push_back(0); // Passing as array reference below, so need NULL termination.
|
||||
perf::TestbenchOutput<perf::GemmProblem> output_gemm(options);
|
||||
result = profile(&profileFuncs[0], output_gemm, options, result);
|
||||
return result;
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#pragma once
|
||||
|
||||
#pragma diag_suppress boolean_controlling_expr_is_constant
|
||||
#include <gtest/gtest.h>
|
||||
#pragma diag_warning boolean_controlling_expr_is_constant
|
||||
|
||||
#include "tools/test/perf/testbench_output.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
|
||||
namespace perf {
|
||||
|
||||
typedef int (GemmProfileFunc)(
|
||||
TestbenchOutput <GemmProblem> &output,
|
||||
TestbenchOptions const &options,
|
||||
Config const &config);
|
||||
|
||||
void RegisterGemmProfileFunc(GemmProfileFunc*);
|
||||
|
||||
} // perf
|
||||
@@ -0,0 +1,121 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
/// \file {nv-internal-release}
|
||||
|
||||
#if (defined(__CUDACC__) && (!defined(__CUDA_ARCH__) || __CUDA_ARCH__ >= 750))
|
||||
#pragma warning( disable : 4503)
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/bmma_gemm_traits.h"
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
#include "tools/test/perf/gemm/cutlass_dispatch.h"
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename Traits>
|
||||
struct BmmaGemmDispatch {
|
||||
|
||||
typedef cutlass::gemm::Gemm<Traits> Gemm;
|
||||
|
||||
typedef typename Gemm::Params Params;
|
||||
|
||||
/// Indicate warp-level GEMM
|
||||
static bool const kThreadMultiplyAdd = false;
|
||||
|
||||
static bool const kRunCuBLAS = false;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Traits::kLayoutB;
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
/// Params argument
|
||||
Params params;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
BmmaGemmDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
BmmaGemmDispatch(int m, int n, int k, int alpha,
|
||||
cutlass::Vector<cutlass::bin1_t, 32> const* d_a, int lda,
|
||||
cutlass::Vector<cutlass::bin1_t, 32> const* d_b, int ldb, int beta,
|
||||
int const* d_c, int ldc, int* d_d, int ldd) {
|
||||
|
||||
params.initialize(m, n, k * 32, alpha, d_a, lda, d_b, ldb, beta, d_c, ldc, d_d, ldd);
|
||||
}
|
||||
|
||||
/// Initializes params object
|
||||
BmmaGemmDispatch(Params const& _params) : params(_params) {}
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_bmma_gemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
typedef perf::GemmProfiler<cutlass::Vector<cutlass::bin1_t, 32>, cutlass::Vector<cutlass::bin1_t, 32>, int, int, int> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
{
|
||||
|
||||
typedef cutlass::gemm::BmmaGemmTraits<cutlass::Shape<1024, 128, 128>,
|
||||
cutlass::Shape<1024, 32, 32>,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor>
|
||||
BmmaGemmTraits;
|
||||
|
||||
typedef BmmaGemmDispatch<BmmaGemmTraits> Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "bmma_gemm_tn", options, config);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
struct BmmaGemmRegistrar {
|
||||
BmmaGemmRegistrar() { RegisterGemmProfileFunc(profile_bmma_gemm); }
|
||||
};
|
||||
|
||||
volatile BmmaGemmRegistrar _BmmaGemmRegistrar;
|
||||
|
||||
} // namespace perf
|
||||
|
||||
#endif // if (defined(__CUDACC__) && (!defined(__CUDA_ARCH__) || __CUDA_ARCH__ >= 750)
|
||||
@@ -24,8 +24,8 @@
|
||||
**************************************************************************************************/
|
||||
#pragma once
|
||||
|
||||
#include <cutlass/matrix_traits.h>
|
||||
#include <tools/util/type_traits.h>
|
||||
#include "cutlass/matrix_traits.h"
|
||||
#include "tools/util/type_traits.h"
|
||||
|
||||
namespace perf {
|
||||
|
||||
|
||||
@@ -32,7 +32,8 @@ template <typename Gemm_,
|
||||
typename ScalarD_,
|
||||
typename Compute_,
|
||||
typename ScalarEpilogue_,
|
||||
bool ThreadMultiplyAdd_>
|
||||
bool ThreadMultiplyAdd_,
|
||||
bool RunCuBLAS_ = true>
|
||||
struct CutlassDispatch {
|
||||
typedef typename Gemm_::Params Params;
|
||||
typedef Gemm_ Gemm;
|
||||
@@ -45,6 +46,7 @@ struct CutlassDispatch {
|
||||
typedef ScalarEpilogue_ ScalarEpilogue;
|
||||
|
||||
static bool const kThreadMultiplyAdd = ThreadMultiplyAdd_;
|
||||
static bool const kRunCuBLAS = RunCuBLAS_;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Gemm::Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Gemm::Traits::kLayoutB;
|
||||
@@ -60,7 +62,7 @@ struct CutlassDispatch {
|
||||
// Methods
|
||||
//
|
||||
|
||||
CutlassDispatch() {}
|
||||
// CutlassDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
CutlassDispatch(Index m,
|
||||
@@ -84,33 +86,6 @@ struct CutlassDispatch {
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
|
||||
/// Determines if problem is aligned (assuming no padding)
|
||||
static bool is_problem_aligned(
|
||||
int m,
|
||||
int n,
|
||||
int k) {
|
||||
|
||||
bool aligned = true;
|
||||
|
||||
if (kLayoutA == cutlass::MatrixLayout::kColumnMajor) {
|
||||
aligned = aligned && !(m % Gemm::Traits::GemmConfig::kScalarsPerLdgA);
|
||||
}
|
||||
else {
|
||||
aligned = aligned && !(k % Gemm::Traits::GemmConfig::kScalarsPerLdgA);
|
||||
}
|
||||
|
||||
if (kLayoutB == cutlass::MatrixLayout::kColumnMajor) {
|
||||
aligned = aligned && !(k % Gemm::Traits::GemmConfig::kScalarsPerLdgB);
|
||||
}
|
||||
else {
|
||||
aligned = aligned && !(n % Gemm::Traits::GemmConfig::kScalarsPerLdgB);
|
||||
}
|
||||
|
||||
aligned = aligned && !(m % Gemm::Traits::GemmConfig::kScalarsPerLdgC);
|
||||
|
||||
return aligned;
|
||||
}
|
||||
};
|
||||
|
||||
/// Basic dispatcher inferred from GEMM traits
|
||||
|
||||
@@ -23,26 +23,29 @@
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
#include <cutlass/gemm/dgemm_traits.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/dgemm_traits.h"
|
||||
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
#include "tools/test/perf/gemm/cutlass_dispatch.h"
|
||||
#pragma warning( disable : 4503)
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_dgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
int profile_dgemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
typedef perf::GemmProfiler<double, double, double, double, double> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
|
||||
// compute capability check
|
||||
if (!options.compute_capability(6, 0)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::DgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor
|
||||
@@ -50,11 +53,10 @@ int profile_dgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_nt", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_nt", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::DgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor
|
||||
@@ -62,11 +64,10 @@ int profile_dgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_nn", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_nn", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::DgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor
|
||||
@@ -74,11 +75,10 @@ int profile_dgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_tn", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_tn", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::DgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor
|
||||
@@ -86,12 +86,18 @@ int profile_dgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_tt", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "dgemm_tt", options, config);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
struct DgemmRegistrar {
|
||||
DgemmRegistrar() { RegisterGemmProfileFunc(profile_dgemm); }
|
||||
};
|
||||
|
||||
volatile DgemmRegistrar _DgemmRegistrar;
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
@@ -36,200 +36,35 @@
|
||||
#include <curand_kernel.h>
|
||||
|
||||
// Cutlass includes
|
||||
#include <tools/test/perf/gemm/cublas_dispatch.h>
|
||||
#include <tools/test/perf/performance_result.h>
|
||||
#include <tools/test/perf/testbench_options.h>
|
||||
#include <tools/util/device_memory.h>
|
||||
#include <tools/util/type_traits.h>
|
||||
#include <tools/util/host_tensor.h>
|
||||
#include <tools/util/tensor_view_io.h>
|
||||
#include "tools/test/perf/gemm/cublas_dispatch.h"
|
||||
#include "tools/test/perf/performance_result.h"
|
||||
#include "tools/test/perf/testbench_options.h"
|
||||
#include "tools/util/device_memory.h"
|
||||
#include "tools/util/host_matrix.h"
|
||||
#include "tools/util/reference/device/tensor_elementwise.h"
|
||||
#include "tools/util/tensor_view_io.h"
|
||||
#include "tools/util/type_traits.h"
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Kernel to determine if two tensors are equal
|
||||
template <typename Type>
|
||||
__global__ void tensor_equals(int *result,
|
||||
int dim_contiguous,
|
||||
int dim_strided,
|
||||
Type const *experimental,
|
||||
int lde,
|
||||
Type const *reference,
|
||||
int ldr) {
|
||||
typedef typename cutlass::TypeTraits<Type>::unsigned_type UnsignedType;
|
||||
namespace detail {
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
template <typename T>
|
||||
struct ElementCount {
|
||||
static int const kValue = 1;
|
||||
};
|
||||
|
||||
experimental += s_idx * lde + c_idx;
|
||||
reference += s_idx * ldr + c_idx;
|
||||
template <typename T, int Elements>
|
||||
struct ElementCount<cutlass::Vector<T, Elements> > {
|
||||
static int const kValue = Elements * ElementCount<T>::kValue;
|
||||
};
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
UnsignedType exp = *reinterpret_cast<UnsignedType const *>(experimental);
|
||||
UnsignedType ref = *reinterpret_cast<UnsignedType const *>(reference);
|
||||
|
||||
if (exp != ref) {
|
||||
*result = -1;
|
||||
return;
|
||||
}
|
||||
|
||||
experimental += lde;
|
||||
reference += ldr;
|
||||
}
|
||||
}
|
||||
}
|
||||
} // namespace detail
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Kernel to initialize tensor to uniform distribution
|
||||
template <typename T>
|
||||
__global__ void initialize_uniform(
|
||||
Distribution dist, int64_t seed, int dim_contiguous, int dim_strided, T *tensor, int ldm) {
|
||||
__shared__ curandState_t rng_state[1024];
|
||||
|
||||
uint64_t gtid = threadIdx.x + blockIdx.x * blockDim.x + blockIdx.y * gridDim.x * blockDim.x;
|
||||
|
||||
curand_init(seed, gtid, 0, &rng_state[threadIdx.x]);
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
tensor += s_idx * ldm + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
double range = dist.uniform.max - dist.uniform.min;
|
||||
|
||||
double rnd = curand_uniform(&rng_state[threadIdx.x]);
|
||||
|
||||
rnd = dist.uniform.min + range * rnd;
|
||||
|
||||
// Random values are cast to integer after scaling by a power of two to facilitate error
|
||||
// testing
|
||||
if (dist.int_scale >= 0) {
|
||||
rnd = double(int(rnd * double(1 << dist.int_scale)));
|
||||
*tensor = T(rnd / double(1 << dist.int_scale));
|
||||
} else {
|
||||
*tensor = T(rnd);
|
||||
}
|
||||
|
||||
tensor += ldm;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kernel to initialize tensor to uniform distribution
|
||||
template <typename T>
|
||||
__global__ void initialize_gaussian(
|
||||
Distribution dist, int64_t seed, int dim_contiguous, int dim_strided, T *tensor, int ldm) {
|
||||
__shared__ curandState_t rng_state[1024];
|
||||
|
||||
uint64_t gtid = threadIdx.x + blockIdx.x * blockDim.x + blockIdx.y * gridDim.x * blockDim.x;
|
||||
|
||||
curand_init(seed, gtid, 0, &rng_state[threadIdx.x]);
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
tensor += s_idx * ldm + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
// Random values are cast to integer after scaling by a power of two to facilitate error
|
||||
// testing
|
||||
|
||||
double rnd = curand_normal(&rng_state[threadIdx.x]);
|
||||
|
||||
rnd = dist.gaussian.mean + dist.gaussian.stddev * rnd;
|
||||
|
||||
if (dist.int_scale >= 0) {
|
||||
rnd = double(int(rnd * double(1 << dist.int_scale)));
|
||||
*tensor = T(rnd / double(1 << dist.int_scale));
|
||||
} else {
|
||||
*tensor = T(rnd);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kernel to initialize tensor to an identity matrix
|
||||
template <typename T>
|
||||
__global__ void initialize_linear(
|
||||
Distribution dist, int64_t seed, int dim_contiguous, int dim_strided, T *tensor, int ldm) {
|
||||
__shared__ curandState_t rng_state[1024];
|
||||
|
||||
uint64_t gtid = threadIdx.x + blockIdx.x * blockDim.x + blockIdx.y * gridDim.x * blockDim.x;
|
||||
|
||||
curand_init(seed, gtid, 0, &rng_state[threadIdx.x]);
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
tensor += s_idx * ldm + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
*tensor =
|
||||
dist.linear.offset + dist.linear.delta_row * c_idx + dist.linear.delta_column * s_idx;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Kernel to initialize tensor to an identity matrix
|
||||
template <typename T>
|
||||
__global__ void initialize_identity(
|
||||
Distribution dist, int64_t seed, int dim_contiguous, int dim_strided, T *tensor, int ldm) {
|
||||
__shared__ curandState_t rng_state[1024];
|
||||
|
||||
uint64_t gtid = threadIdx.x + blockIdx.x * blockDim.x + blockIdx.y * gridDim.x * blockDim.x;
|
||||
|
||||
curand_init(seed, gtid, 0, &rng_state[threadIdx.x]);
|
||||
|
||||
int c_idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int s_idx = blockIdx.y * blockDim.x;
|
||||
|
||||
tensor += s_idx * ldm + c_idx;
|
||||
|
||||
for (int s_offset = 0; s_offset < blockDim.x; ++s_offset, ++s_idx) {
|
||||
if (s_idx < dim_strided && c_idx < dim_contiguous) {
|
||||
*tensor = (c_idx == s_idx ? T(1) : T(0));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Dispatcher to appropriate initialization kernel
|
||||
template <typename T>
|
||||
inline void initialize(Distribution const &dist,
|
||||
int64_t seed,
|
||||
int dim_contiguous,
|
||||
int dim_strided,
|
||||
T *tensor,
|
||||
int ldm) {
|
||||
dim3 block(256, 1, 1);
|
||||
dim3 grid((dim_contiguous + block.x - 1) / block.x, (dim_strided + block.x - 1) / block.x);
|
||||
|
||||
switch (dist.kind) {
|
||||
case Distribution::Uniform:
|
||||
initialize_uniform<<<grid, block>>>(dist, seed, dim_contiguous, dim_strided, tensor, ldm);
|
||||
break;
|
||||
case Distribution::Gaussian:
|
||||
initialize_gaussian<<<grid, block>>>(dist, seed, dim_contiguous, dim_strided, tensor, ldm);
|
||||
break;
|
||||
case Distribution::Linear:
|
||||
initialize_linear<<<grid, block>>>(dist, seed, dim_contiguous, dim_strided, tensor, ldm);
|
||||
break;
|
||||
case Distribution::Identity:
|
||||
initialize_identity<<<grid, block>>>(dist, seed, dim_contiguous, dim_strided, tensor, ldm);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Host-side implementation of performance testbed
|
||||
template <typename AType, typename BType, typename CType, typename Accumulator, typename Scalar>
|
||||
class GemmTestbed {
|
||||
@@ -295,14 +130,13 @@ class GemmTestbed {
|
||||
|
||||
/// Helper to resize a matrix with a given size and layout if needed
|
||||
template <typename T>
|
||||
static void resize_device_allocation(
|
||||
cutlass::device_memory::allocation<T> &tensor,
|
||||
Distribution const &dist,
|
||||
int64_t seed,
|
||||
int rows,
|
||||
int columns,
|
||||
cutlass::MatrixLayout::Kind layout,
|
||||
int ldm = 0) {
|
||||
static void resize_device_allocation(cutlass::device_memory::allocation<T> &tensor,
|
||||
cutlass::Distribution const &dist,
|
||||
int64_t seed,
|
||||
int rows,
|
||||
int columns,
|
||||
cutlass::MatrixLayout::Kind layout,
|
||||
int ldm = 0) {
|
||||
if (!ldm) {
|
||||
ldm = (layout == cutlass::MatrixLayout::kColumnMajor ? rows : columns);
|
||||
}
|
||||
@@ -315,65 +149,79 @@ class GemmTestbed {
|
||||
int c_dim = (layout == cutlass::MatrixLayout::kColumnMajor ? rows : columns);
|
||||
int s_dim = (layout == cutlass::MatrixLayout::kColumnMajor ? columns : rows);
|
||||
|
||||
initialize(dist, seed, c_dim, s_dim, tensor.get(), ldm);
|
||||
cutlass::TensorView<T, 2> view(
|
||||
tensor.get(),
|
||||
cutlass::make_Coord(ldm, 1),
|
||||
cutlass::make_Coord(s_dim, c_dim));
|
||||
|
||||
cutlass::reference::device::TensorInitialize(view, seed, dist);
|
||||
}
|
||||
}
|
||||
|
||||
/// Resizes each tensor
|
||||
void resize_helper(GemmProblem const &problem) {
|
||||
resize_device_allocation(
|
||||
A,
|
||||
initial_distribution.dist_A,
|
||||
initial_distribution.seed,
|
||||
problem.m,
|
||||
problem.k,
|
||||
problem.layout_A);
|
||||
resize_device_allocation(A,
|
||||
initial_distribution.dist_A,
|
||||
initial_distribution.seed,
|
||||
problem.m,
|
||||
problem.k,
|
||||
problem.layout_A);
|
||||
|
||||
resize_device_allocation(
|
||||
B,
|
||||
initial_distribution.dist_B,
|
||||
initial_distribution.seed + 17, // compute distinct value from initial seed
|
||||
problem.k,
|
||||
problem.n,
|
||||
problem.layout_B);
|
||||
B,
|
||||
initial_distribution.dist_B,
|
||||
initial_distribution.seed + 17, // compute distinct value from initial seed
|
||||
problem.k,
|
||||
problem.n,
|
||||
problem.layout_B);
|
||||
|
||||
resize_device_allocation(
|
||||
C_initial,
|
||||
initial_distribution.dist_C,
|
||||
initial_distribution.seed + 101, // compute distinct value from initial seed
|
||||
problem.m,
|
||||
problem.n,
|
||||
cutlass::MatrixLayout::kColumnMajor);
|
||||
C_initial,
|
||||
initial_distribution.dist_C,
|
||||
initial_distribution.seed + 101, // compute distinct value from initial seed
|
||||
problem.m,
|
||||
problem.n,
|
||||
cutlass::MatrixLayout::kColumnMajor);
|
||||
|
||||
resize_device_allocation(
|
||||
reference, Distribution(), 0, problem.m, problem.n, cutlass::MatrixLayout::kColumnMajor);
|
||||
resize_device_allocation(reference,
|
||||
cutlass::Distribution(),
|
||||
0,
|
||||
problem.m,
|
||||
problem.n,
|
||||
cutlass::MatrixLayout::kColumnMajor);
|
||||
|
||||
resize_device_allocation(
|
||||
experimental, Distribution(), 0, problem.m, problem.n, cutlass::MatrixLayout::kColumnMajor);
|
||||
resize_device_allocation(experimental,
|
||||
cutlass::Distribution(),
|
||||
0,
|
||||
problem.m,
|
||||
problem.n,
|
||||
cutlass::MatrixLayout::kColumnMajor);
|
||||
}
|
||||
|
||||
/// Functor to print errors
|
||||
struct PrintErrors {
|
||||
|
||||
/// Equivalently sized integer type
|
||||
typedef typename cutlass::TypeTraits<CType>::integer_type integer_t;
|
||||
|
||||
/// Performance testbench defined for a TensorView of rank-2 contiguous matrices
|
||||
typedef cutlass::TensorView<CType, 2, cutlass::MatrixLayout::ContiguousLayout> MatrixView;
|
||||
|
||||
/// Output stream to write to
|
||||
std::ostream& out;
|
||||
std::ostream &out;
|
||||
|
||||
/// Reference tensor view
|
||||
cutlass::HostTensorView<CType> const& reference;
|
||||
MatrixView const &reference;
|
||||
|
||||
/// Computed tensor view
|
||||
cutlass::HostTensorView<CType> const& experimental;
|
||||
MatrixView const &experimental;
|
||||
|
||||
/// Errors greater than or this amount result in printing
|
||||
integer_t ulps_threshold;
|
||||
|
||||
///
|
||||
PrintErrors(std::ostream& _out,
|
||||
cutlass::HostTensorView<CType> const& _reference,
|
||||
cutlass::HostTensorView<CType> const& _experimental,
|
||||
PrintErrors(std::ostream &_out,
|
||||
MatrixView const &_reference,
|
||||
MatrixView const &_experimental,
|
||||
integer_t _ulps_threshold = 1)
|
||||
: out(_out),
|
||||
reference(_reference),
|
||||
@@ -381,18 +229,15 @@ class GemmTestbed {
|
||||
ulps_threshold(_ulps_threshold) {}
|
||||
|
||||
/// Compares one element
|
||||
void operator()(
|
||||
CType const& element,
|
||||
typename cutlass::HostTensorView<CType>::Coord_t coord) {
|
||||
|
||||
void operator()(CType const &element, typename MatrixView::TensorCoord coord) {
|
||||
CType exp = experimental.at(coord);
|
||||
CType ref = reference.at(coord);
|
||||
|
||||
int64_t int_exp = 0;
|
||||
int64_t int_ref = 0;
|
||||
|
||||
*reinterpret_cast<CType*>(&int_exp) = exp;
|
||||
*reinterpret_cast<CType*>(&int_ref) = ref;
|
||||
*reinterpret_cast<CType *>(&int_exp) = exp;
|
||||
*reinterpret_cast<CType *>(&int_ref) = ref;
|
||||
|
||||
integer_t ulps = integer_t(int_exp - int_ref);
|
||||
|
||||
@@ -405,11 +250,10 @@ class GemmTestbed {
|
||||
relative /= double(ref);
|
||||
}
|
||||
|
||||
out << "[" << coord << "] expected: " << ref << " (0x"
|
||||
<< std::hex << std::setw(width) << std::setfill('0') << integer_t(int_ref) << std::dec
|
||||
<< ")"
|
||||
<< ", got: " << exp << " (0x" << std::hex
|
||||
<< std::setw(width) << std::setfill('0') << integer_t(int_exp) << std::dec << ")"
|
||||
out << "[" << coord << "] expected: " << ref << " (0x" << std::hex << std::setw(width)
|
||||
<< std::setfill('0') << integer_t(int_ref) << std::dec << ")"
|
||||
<< ", got: " << exp << " (0x" << std::hex << std::setw(width) << std::setfill('0')
|
||||
<< integer_t(int_exp) << std::dec << ")"
|
||||
<< " relative error: " << relative << ", ulps: " << ulps << "\n";
|
||||
}
|
||||
}
|
||||
@@ -497,7 +341,7 @@ class GemmTestbed {
|
||||
|
||||
/// Returns the number of flops implied by the computation (1 multiply-accumulate = 2 flops)
|
||||
uint64_t flops() const {
|
||||
return uint64_t(problem.m) * uint64_t(problem.n) * uint64_t(problem.k) * 2ULL;
|
||||
return uint64_t(problem.m) * uint64_t(problem.n) * uint64_t(problem.k) * detail::ElementCount<AType>::kValue * 2ULL;
|
||||
}
|
||||
|
||||
/// Computes the speed of the computation in GFLOPs/s
|
||||
@@ -555,25 +399,17 @@ class GemmTestbed {
|
||||
|
||||
/// Verifies the 'test' tensor with 'ref'
|
||||
bool verify(TensorC const &test, TensorC const &ref) {
|
||||
cutlass::device_memory::allocation<int> flag_device(1);
|
||||
|
||||
int flag = 0;
|
||||
cutlass::device_memory::copy_to_device(flag_device.get(), &flag, 1);
|
||||
|
||||
dim3 block(256, 1, 1);
|
||||
dim3 grid((problem.m + block.x - 1) / block.x, (problem.n + block.x - 1) / block.x);
|
||||
|
||||
tensor_equals<CDeviceType><<<grid, block>>>(flag_device.get(),
|
||||
problem.m,
|
||||
problem.n,
|
||||
experimental.get(),
|
||||
problem.m,
|
||||
reference.get(),
|
||||
problem.m);
|
||||
|
||||
cutlass::device_memory::copy_to_host(&flag, flag_device.get(), 1);
|
||||
|
||||
return flag == 0;
|
||||
return cutlass::reference::device::TensorEquals(
|
||||
cutlass::TensorView<CDeviceType, 2>(
|
||||
test.get(),
|
||||
cutlass::make_Coord(problem.m, 1),
|
||||
cutlass::make_Coord(problem.n, problem.m)),
|
||||
cutlass::TensorView<CDeviceType, 2>(
|
||||
ref.get(),
|
||||
cutlass::make_Coord(problem.m, 1),
|
||||
cutlass::make_Coord(problem.n, problem.m))
|
||||
);
|
||||
}
|
||||
|
||||
/// Computes the reference output
|
||||
@@ -587,12 +423,11 @@ class GemmTestbed {
|
||||
|
||||
/// Writes the problem to an ostream in human-readable form
|
||||
void write_problem(std::ostream &results_output, std::ostream &errors_output) {
|
||||
|
||||
cutlass::HostTensor<AType, false> host_A;
|
||||
cutlass::HostTensor<BType, false> host_B;
|
||||
cutlass::HostTensor<CType, false> host_C;
|
||||
cutlass::HostTensor<CType, false> host_D;
|
||||
cutlass::HostTensor<CType, false> host_Ref;
|
||||
cutlass::HostMatrix<AType> host_A;
|
||||
cutlass::HostMatrix<BType> host_B;
|
||||
cutlass::HostMatrix<CType> host_C;
|
||||
cutlass::HostMatrix<CType> host_D;
|
||||
cutlass::HostMatrix<CType> host_Ref;
|
||||
|
||||
host_A.resize_matrix(M(), K(), layout_a());
|
||||
host_B.resize_matrix(K(), N(), layout_b());
|
||||
@@ -608,11 +443,16 @@ class GemmTestbed {
|
||||
host_Ref.copy_to_host(ptr_reference());
|
||||
|
||||
// write out human readable
|
||||
results_output << "A =\n" << host_A << "\n"
|
||||
<< "B =\n" << host_B << "\n"
|
||||
<< "C = \n" << host_C << "\n"
|
||||
<< "Ref =\n" << host_Ref << "\n"
|
||||
<< "Experimental =\n" << host_D << "\n";
|
||||
results_output << "A =\n"
|
||||
<< host_A << "\n"
|
||||
<< "B =\n"
|
||||
<< host_B << "\n"
|
||||
<< "C = \n"
|
||||
<< host_C << "\n"
|
||||
<< "Ref =\n"
|
||||
<< host_Ref << "\n"
|
||||
<< "Experimental =\n"
|
||||
<< host_D << "\n";
|
||||
|
||||
// write out list of errors
|
||||
PrintErrors printer(errors_output, host_Ref, host_D);
|
||||
|
||||
@@ -29,16 +29,18 @@
|
||||
#include <stdexcept>
|
||||
#include <utility>
|
||||
|
||||
#if defined(WIN32)
|
||||
#include "cutlass/util/platform.h"
|
||||
#if defined(CUTLASS_OS_WINDOWS)
|
||||
#include <Windows.h>
|
||||
#else
|
||||
// needed for sleep
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
#include <tools/test/perf/testbench_options.h>
|
||||
#include <tools/test/perf/testbench_output.h>
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
#include "tools/test/perf/testbench_configs.h"
|
||||
#include "tools/test/perf/testbench_options.h"
|
||||
#include "tools/test/perf/testbench_output.h"
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -63,17 +65,23 @@ class GemmProfiler {
|
||||
//
|
||||
|
||||
/// Reference to TestbenchOutput instance
|
||||
TestbenchOutput &output;
|
||||
TestbenchOutput<GemmProblem> &output;
|
||||
|
||||
/// Reference to options object
|
||||
TestbenchOptions const &options;
|
||||
|
||||
// Reference to config object
|
||||
Config const &config;
|
||||
|
||||
/// Performance test environment
|
||||
PerfTestbed testbed;
|
||||
|
||||
/// Kernel name
|
||||
std::string kernel_name;
|
||||
|
||||
/// Cutlass algorithm
|
||||
std::string cutlass_algo;
|
||||
|
||||
/// Timing events
|
||||
cudaEvent_t events[2];
|
||||
|
||||
@@ -93,14 +101,17 @@ class GemmProfiler {
|
||||
//
|
||||
|
||||
/// Constructs performance testebed
|
||||
GemmProfiler(TestbenchOutput &_output,
|
||||
GemmProfiler(TestbenchOutput<GemmProblem> &_output,
|
||||
std::string const &_kernel_name,
|
||||
TestbenchOptions const &_options)
|
||||
std::string const &_cutlass_algo,
|
||||
TestbenchOptions const &_options,
|
||||
Config const &_config)
|
||||
: output(_output),
|
||||
options(_options),
|
||||
config(_config),
|
||||
kernel_name(_kernel_name),
|
||||
cutlass_algo(_cutlass_algo),
|
||||
testbed(_options.initial_distribution) {
|
||||
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
cudaError_t result = cudaEventCreate(&events[i]);
|
||||
if (result != cudaSuccess) {
|
||||
@@ -112,34 +123,47 @@ class GemmProfiler {
|
||||
~GemmProfiler() {}
|
||||
|
||||
/// Writes the workspace to text files
|
||||
void write_problem(std::string const &kernel_name) {
|
||||
void write_problem(Provider::Kind provider, std::string const &kernel_name) {
|
||||
std::stringstream base_filename;
|
||||
|
||||
std::stringstream base_filename;
|
||||
base_filename << provider << "_" << kernel_name << "_" << testbed.M() << "x" << testbed.N()
|
||||
<< "x" << testbed.K();
|
||||
|
||||
base_filename
|
||||
<< kernel_name << "_"
|
||||
<< testbed.M() << "x" << testbed.N() << "x" << testbed.K();
|
||||
std::string results_name = base_filename.str() + "_results.txt";
|
||||
std::string errors_name = base_filename.str() + "_errors.txt";
|
||||
|
||||
std::string results_name = base_filename.str() + "_results.txt";
|
||||
std::string errors_name = base_filename.str() + "_errors.txt";
|
||||
|
||||
std::ofstream results(results_name.c_str());
|
||||
std::ofstream errors(errors_name.c_str());
|
||||
testbed.write_problem(results, errors);
|
||||
std::ofstream results(results_name.c_str());
|
||||
std::ofstream errors(errors_name.c_str());
|
||||
testbed.write_problem(results, errors);
|
||||
}
|
||||
|
||||
/// Profiles Cutlass
|
||||
template <typename CutlassDispatch>
|
||||
PerformanceResult execute_cutlass(GemmProblem const &problem, cublasGemmAlgo_t algorithm) {
|
||||
PerformanceResult result(kernel_name, problem);
|
||||
PerformanceResult<GemmProblem> execute_cutlass(GemmProblem const &problem,
|
||||
cublasGemmAlgo_t algorithm) {
|
||||
PerformanceResult<GemmProblem> result(
|
||||
Provider::Cutlass
|
||||
, kernel_name
|
||||
, problem
|
||||
);
|
||||
|
||||
testbed.compute_reference(algorithm);
|
||||
|
||||
if (cudaDeviceSynchronize() != cudaSuccess) {
|
||||
result.disposition = Disposition::NotVerified;
|
||||
if (options.dry_run) {
|
||||
result.disposition = Disposition::NotRun;
|
||||
return result;
|
||||
}
|
||||
|
||||
if (CutlassDispatch::kRunCuBLAS) {
|
||||
testbed.compute_reference(algorithm);
|
||||
|
||||
if (cudaDeviceSynchronize() != cudaSuccess) {
|
||||
result.disposition = Disposition::NotVerified;
|
||||
return result;
|
||||
}
|
||||
}
|
||||
else {
|
||||
result.disposition = Disposition::Passed;
|
||||
}
|
||||
|
||||
CutlassDispatch dispatch(testbed.M(),
|
||||
testbed.N(),
|
||||
testbed.K(),
|
||||
@@ -161,14 +185,16 @@ class GemmProfiler {
|
||||
return result;
|
||||
}
|
||||
|
||||
if (testbed.verify_with_reference()) {
|
||||
result.disposition = Disposition::Passed;
|
||||
} else {
|
||||
result.disposition = Disposition::Incorrect;
|
||||
if (CutlassDispatch::kRunCuBLAS) {
|
||||
if (testbed.verify_with_reference()) {
|
||||
result.disposition = Disposition::Passed;
|
||||
} else {
|
||||
result.disposition = Disposition::Incorrect;
|
||||
}
|
||||
}
|
||||
|
||||
if (options.save_workspace(result.disposition == Disposition::Passed)) {
|
||||
write_problem(kernel_name);
|
||||
write_problem(Provider::Cutlass, kernel_name);
|
||||
}
|
||||
|
||||
if (cudaDeviceSynchronize() != cudaSuccess) {
|
||||
@@ -212,30 +238,38 @@ class GemmProfiler {
|
||||
result.gflops = testbed.GFLOPs_per_sec(result.runtime);
|
||||
|
||||
if (result.disposition != Disposition::Passed) {
|
||||
std::cout << kernel_name << " failed with disposition: " << result.disposition;
|
||||
std::cout << "[\033[1;31mFAILED\033[0m]: " << kernel_name
|
||||
<< " failed with disposition: " << result.disposition << "\n";
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
template <typename T, typename F>
|
||||
bool contains(T const &container, F const &val) {
|
||||
return std::find(container.begin(), container.end(), val) != container.end();
|
||||
}
|
||||
|
||||
/// Executes all kernels for this problem size
|
||||
template <typename CutlassDispatch>
|
||||
std::vector<PerformanceResult> execute(GemmProblem const &problem) {
|
||||
std::vector<PerformanceResult<GemmProblem> > execute(GemmProblem const &problem) {
|
||||
|
||||
// New problem size
|
||||
output.begin_problem();
|
||||
|
||||
cublasGemmAlgo_t algorithm =
|
||||
(CutlassDispatch::kThreadMultiplyAdd ? CUBLAS_GEMM_DEFAULT : CUBLAS_GEMM_DEFAULT_TENSOR_OP);
|
||||
bool const tensor_op = !(CutlassDispatch::kThreadMultiplyAdd);
|
||||
cublasGemmAlgo_t algorithm = tensor_op ?
|
||||
CUBLAS_GEMM_DEFAULT_TENSOR_OP : CUBLAS_GEMM_DEFAULT;
|
||||
|
||||
testbed.resize(problem);
|
||||
|
||||
std::vector<PerformanceResult> results;
|
||||
|
||||
results.push_back(execute_cutlass<CutlassDispatch>(problem, algorithm));
|
||||
std::vector<PerformanceResult<GemmProblem> > results;
|
||||
|
||||
results.push_back(execute_cutlass<CutlassDispatch>(problem, algorithm));
|
||||
// cool-down period
|
||||
pause(2);
|
||||
if (!options.dry_run) {
|
||||
pause(options.sleep_time);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
@@ -243,25 +277,20 @@ class GemmProfiler {
|
||||
/// Runs the test and collects performance for all results
|
||||
template <typename CutlassDispatch>
|
||||
void schmoo(Range const &M, Range const &N, Range const &K) {
|
||||
for (int m = M.start; m <= M.end; m += M.increment) {
|
||||
for (int n = N.start; n <= N.end; n += N.increment) {
|
||||
for (int k = K.start; k <= K.end; k += K.increment) {
|
||||
for (int m = M.start; m <= M.end; m = M.next(m)) {
|
||||
for (int n = N.start; n <= N.end; n = N.next(n)) {
|
||||
for (int k = K.start; k <= K.end; k = K.next(k)) {
|
||||
|
||||
// Avoid evaluating problem if problem size does not satisfy alignment
|
||||
if (!CutlassDispatch::is_problem_aligned(m, n, k)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
std::vector<PerformanceResult> results =
|
||||
std::vector<PerformanceResult<GemmProblem> > results =
|
||||
execute<CutlassDispatch>(GemmProblem(m,
|
||||
n,
|
||||
k,
|
||||
CutlassDispatch::kLayoutA,
|
||||
CutlassDispatch::kLayoutB,
|
||||
options.alpha,
|
||||
options.beta));
|
||||
config.alpha,
|
||||
config.beta));
|
||||
|
||||
for (std::vector<PerformanceResult>::const_iterator it = results.begin();
|
||||
for (std::vector<PerformanceResult<GemmProblem> >::const_iterator it = results.begin();
|
||||
it != results.end();
|
||||
++it) {
|
||||
output.append(*it);
|
||||
@@ -274,46 +303,53 @@ class GemmProfiler {
|
||||
/// Runs the test over the problem space and reports only the best performance
|
||||
template <typename CutlassDispatch>
|
||||
void peak(Range const &M, Range const &N, Range const &K) {
|
||||
typedef std::map<Provider::Kind, PerformanceResult<GemmProblem> > ProviderPerformanceMap;
|
||||
|
||||
PerformanceResult max_perf;
|
||||
bool first_result = true;
|
||||
ProviderPerformanceMap max_perf;
|
||||
|
||||
for (int m = M.start; m <= M.end; m += M.increment) {
|
||||
for (int n = N.start; n <= N.end; n += N.increment) {
|
||||
for (int k = K.start; k <= K.end; k += K.increment) {
|
||||
|
||||
// Avoid evaluating problem if problem size does not satisfy alignment
|
||||
if (!CutlassDispatch::is_problem_aligned(m, n, k)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
std::vector<PerformanceResult> results =
|
||||
for (int m = M.start; m <= M.end; m += M.next(m)) {
|
||||
for (int n = N.start; n <= N.end; n += N.next(n)) {
|
||||
for (int k = K.start; k <= K.end; k += K.next(k)) {
|
||||
std::vector<PerformanceResult<GemmProblem> > results =
|
||||
execute<CutlassDispatch>(GemmProblem(m,
|
||||
n,
|
||||
k,
|
||||
CutlassDispatch::kLayoutA,
|
||||
CutlassDispatch::kLayoutB,
|
||||
options.alpha,
|
||||
options.beta));
|
||||
config.alpha,
|
||||
config.beta));
|
||||
|
||||
for (std::vector<PerformanceResult>::const_iterator it = results.begin();
|
||||
for (std::vector<PerformanceResult<GemmProblem> >::const_iterator it = results.begin();
|
||||
it != results.end();
|
||||
++it) {
|
||||
|
||||
/// Writes the output without appending it
|
||||
output.pretty_print(*it);
|
||||
|
||||
/// Updates maximum performing kernel
|
||||
if (first_result || max_perf.gflops > it->gflops) {
|
||||
max_perf = *it;
|
||||
if (it->disposition == Disposition::Passed) {
|
||||
/// Updates maximum performing kernel
|
||||
ProviderPerformanceMap::iterator max_perf_it = max_perf.find(it->provider);
|
||||
|
||||
if (max_perf_it == max_perf.end()) {
|
||||
max_perf.insert(std::make_pair(it->provider, *it));
|
||||
} else if (max_perf_it->second.gflops < it->gflops) {
|
||||
max_perf_it->second = *it;
|
||||
}
|
||||
}
|
||||
first_result = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
output.append(max_perf);
|
||||
Provider::Kind providers[] = {
|
||||
Provider::Cutlass,
|
||||
Provider::Invalid
|
||||
};
|
||||
for (int i = 0; providers[i] != Provider::Invalid; ++i) {
|
||||
ProviderPerformanceMap::const_iterator it = max_perf.find(providers[i]);
|
||||
if (it != max_perf.end()) {
|
||||
output.append(it->second);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -321,17 +357,19 @@ class GemmProfiler {
|
||||
|
||||
/// Dispatches to GEMM performance profiler
|
||||
template <typename Dispatch, typename GemmProfiler>
|
||||
int profile_gemm(TestbenchOutput &output,
|
||||
int profile_gemm(TestbenchOutput<GemmProblem> &output,
|
||||
std::string const &kernel,
|
||||
TestbenchOptions const &options) {
|
||||
if (options.kernel_enabled(kernel)) {
|
||||
GemmProfiler perf(output, kernel, options);
|
||||
TestbenchOptions const &options,
|
||||
Config const &config,
|
||||
std::string const &cutlass_algo = "") {
|
||||
if (config.kernel_enabled(kernel)) {
|
||||
GemmProfiler perf(output, kernel, cutlass_algo, options, config);
|
||||
if (options.peak_performance) {
|
||||
perf.template peak<Dispatch>(
|
||||
options.problem_range.M, options.problem_range.N, options.problem_range.K);
|
||||
config.problem_range.M, config.problem_range.N, config.problem_range.K);
|
||||
} else {
|
||||
perf.template schmoo<Dispatch>(
|
||||
options.problem_range.M, options.problem_range.N, options.problem_range.K);
|
||||
config.problem_range.M, config.problem_range.N, config.problem_range.K);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -22,62 +22,62 @@
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
#include <cutlass/gemm/hgemm_traits.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/hgemm_traits.h"
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
#include "tools/test/perf/gemm/cutlass_dispatch.h"
|
||||
|
||||
#pragma warning( disable : 4503)
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_hgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
int profile_hgemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
typedef perf::GemmProfiler<
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
>
|
||||
GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_nt", options);
|
||||
// compute capability check
|
||||
if (!options.compute_capability(6, 0)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
>
|
||||
GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_nn", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_nt", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
>
|
||||
GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_nn", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
@@ -87,11 +87,10 @@ int profile_hgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_tn", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_tn", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::HgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
@@ -101,13 +100,18 @@ int profile_hgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_tt", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "hgemm_tt", options, config);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
struct HgemmRegistrar {
|
||||
HgemmRegistrar() { RegisterGemmProfileFunc(profile_hgemm); }
|
||||
};
|
||||
|
||||
volatile HgemmRegistrar _HgemmRegistrar;
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
|
||||
@@ -23,24 +23,31 @@
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
#include <cutlass/gemm/igemm_traits.h>
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/igemm_traits.h"
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
#include "tools/test/perf/gemm/cutlass_dispatch.h"
|
||||
|
||||
#pragma warning( disable : 4503)
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_igemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
int profile_igemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
|
||||
typedef perf::GemmProfiler<int8_t, int8_t, int, int, int> GemmProfiler;
|
||||
|
||||
// compute capability check
|
||||
if (!options.compute_capability(6, 1)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::IgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor
|
||||
@@ -48,11 +55,10 @@ int profile_igemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "igemm_nt", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "igemm_nt", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::IgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor
|
||||
@@ -60,11 +66,10 @@ int profile_igemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "igemm_nn", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "igemm_nn", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::IgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor
|
||||
@@ -72,11 +77,10 @@ int profile_igemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "igemm_tn", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "igemm_tn", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::IgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor
|
||||
@@ -84,12 +88,62 @@ int profile_igemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "igemm_tt", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "igemm_tt", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::IgemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor, cutlass::Shape<128, 32, 32>, int,
|
||||
cutlass::gemm::LinearScaling<int>, cutlass::Shape<32, 8, 4> > GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "igemm_32x32x128_nn",
|
||||
options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::IgemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor, cutlass::Shape<128, 32, 32>, int,
|
||||
cutlass::gemm::LinearScaling<int>, cutlass::Shape<32, 8, 4> > GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "igemm_32x32x128_nt",
|
||||
options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::IgemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor, cutlass::Shape<128, 32, 32>, int,
|
||||
cutlass::gemm::LinearScaling<int>, cutlass::Shape<32, 8, 4> > GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "igemm_32x32x128_tn",
|
||||
options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::IgemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor, cutlass::Shape<128, 32, 32>, int,
|
||||
cutlass::gemm::LinearScaling<int>, cutlass::Shape<32, 8, 4> > GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
results = profile_gemm<Dispatch, GemmProfiler>(output, "igemm_32x32x128_tt",
|
||||
options, config);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
struct IgemmRegistrar {
|
||||
IgemmRegistrar() { RegisterGemmProfileFunc(profile_igemm); }
|
||||
};
|
||||
|
||||
volatile IgemmRegistrar _IgemmRegistrar;
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
@@ -22,80 +22,96 @@
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
#include <cutlass/gemm/sgemm_traits.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/sgemm_traits.h"
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
#include "tools/test/perf/gemm/cutlass_dispatch.h"
|
||||
#pragma warning( disable : 4503)
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_sgemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
template <typename OutputTile>
|
||||
int profile_sgemm_kernel(
|
||||
TestbenchOutput<GemmProblem> &output,
|
||||
TestbenchOptions const &options,
|
||||
Config const &config,
|
||||
std::string const &name,
|
||||
std::string const &algo) {
|
||||
|
||||
typedef perf::GemmProfiler<float, float, float, float, float> SGemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::SgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
OutputTile
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, SGemmProfiler>(output, "sgemm_nt", options);
|
||||
results |= profile_gemm<Dispatch, SGemmProfiler>(output, name + "_nt", options, config, algo);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::SgemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
OutputTile
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, SGemmProfiler>(output, "sgemm_nn", options);
|
||||
results |= profile_gemm<Dispatch, SGemmProfiler>(output, name + "_nn", options, config, algo);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::SgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
OutputTile
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, SGemmProfiler>(output, "sgemm_tn", options);
|
||||
results |= profile_gemm<Dispatch, SGemmProfiler>(output, name + "_tn", options, config, algo);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::SgemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<8, 128, 128>
|
||||
OutputTile
|
||||
> GemmTraits;
|
||||
|
||||
typedef typename CutlassDispatchBasic<GemmTraits>::Dispatch Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, SGemmProfiler>(output, "sgemm_tt", options);
|
||||
results |= profile_gemm<Dispatch, SGemmProfiler>(output, name + "_tt", options, config, algo);
|
||||
}
|
||||
return results;
|
||||
}
|
||||
|
||||
/// Profiles all SGEMM tile sizes
|
||||
int profile_sgemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
int results = 0;
|
||||
|
||||
results |= profile_sgemm_kernel<cutlass::Shape<8, 128, 128> >(output, options, config, "sgemm", "128x128");
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
struct SgemmRegistrar {
|
||||
SgemmRegistrar() { RegisterGemmProfileFunc(profile_sgemm); }
|
||||
};
|
||||
|
||||
volatile SgemmRegistrar _SgemmRegistrar;
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
|
||||
#include "cutlass/wmma_matrix.h"
|
||||
#ifdef CUTLASS_USE_WMMA_API
|
||||
#ifdef CUTLASS_USE_SUBBYTE_WMMA
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/wmma_gemm_traits.h"
|
||||
#include "tools/test/perf/gemm/cutlass_dispatch.h"
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename Traits>
|
||||
struct WmmaBinaryGemmDispatch {
|
||||
|
||||
typedef cutlass::gemm::Gemm<Traits> Gemm;
|
||||
|
||||
typedef typename Gemm::Params Params;
|
||||
|
||||
/// Indicate warp-level GEMM
|
||||
static bool const kThreadMultiplyAdd = false;
|
||||
|
||||
static bool const kRunCuBLAS = false;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Traits::kLayoutB;
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
/// Params argument
|
||||
Params params;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
WmmaBinaryGemmDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaBinaryGemmDispatch(int m, int n, int k, int alpha,
|
||||
cutlass::Vector<cutlass::bin1_t, 32> const* d_a, int lda,
|
||||
cutlass::Vector<cutlass::bin1_t, 32> const* d_b, int ldb, int beta,
|
||||
int const* d_c, int ldc, int* d_d, int ldd) {
|
||||
|
||||
params.initialize(m, n, k * 32, alpha, d_a, lda, d_b, ldb, beta, d_c, ldc, d_d, ldd);
|
||||
}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaBinaryGemmDispatch(Params const& _params) : params(_params) {}
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_wmma_binary_gemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
typedef perf::GemmProfiler<cutlass::Vector<cutlass::bin1_t, 32>, cutlass::Vector<cutlass::bin1_t, 32>, int, int, int> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
// compute capability check
|
||||
if (!options.compute_capability_exact(7, 5)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<1024, 128, 128>,
|
||||
cutlass::Vector<cutlass::bin1_t, 32>,
|
||||
cutlass::Vector<cutlass::bin1_t, 32>,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<1024, 32, 64>,
|
||||
cutlass::Shape<128, 8, 8>,
|
||||
128,
|
||||
128>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaBinaryGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_binary_gemm_tn", options, config);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#else // ! CUTLASS_USE_SUBBYTE_WMMA
|
||||
|
||||
namespace perf {
|
||||
|
||||
int profile_wmma_binary_gemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
} // namespace perf
|
||||
|
||||
#endif
|
||||
|
||||
struct WmmaBinaryGemmRegistrar {
|
||||
WmmaBinaryGemmRegistrar() { perf::RegisterGemmProfileFunc(perf::profile_wmma_binary_gemm); }
|
||||
};
|
||||
|
||||
volatile WmmaBinaryGemmRegistrar _WmmaBinaryGemmRegistrar;
|
||||
|
||||
#endif // CUTLASS_USE_WMMA_API
|
||||
@@ -23,17 +23,19 @@
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#include <cutlass/wmma_matrix.h>
|
||||
#include "cutlass/wmma_matrix.h"
|
||||
#ifdef CUTLASS_USE_WMMA_API
|
||||
|
||||
#pragma warning( disable : 4503)
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
|
||||
#include <tools/test/perf/gemm/gemm_profiler.h>
|
||||
#include <tools/test/perf/gemm/cutlass_dispatch.h>
|
||||
#include <tools/test/perf/gemm/gemm_perf_testbed.h>
|
||||
#include <cutlass/gemm/wmma_gemm_traits.h>
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/wmma_gemm_traits.h"
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
#include "tools/test/perf/gemm/cutlass_dispatch.h"
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -47,9 +49,17 @@ struct WmmaGemmDispatch {
|
||||
/// Indicate warp-level GEMM
|
||||
static bool const kThreadMultiplyAdd = false;
|
||||
|
||||
static bool const kRunCuBLAS = true;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Traits::kLayoutB;
|
||||
|
||||
typedef typename Traits::ScalarA ScalarA;
|
||||
typedef typename Traits::ScalarB ScalarB;
|
||||
typedef typename Traits::ScalarC ScalarC;
|
||||
typedef typename Traits::ScalarD ScalarD;
|
||||
typedef typename Traits::Epilogue::Functor::Scalar Scalar;
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
@@ -64,9 +74,20 @@ struct WmmaGemmDispatch {
|
||||
WmmaGemmDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaGemmDispatch(int m, int n, int k, float alpha, half const* d_a, int lda,
|
||||
half const* d_b, int ldb, float beta, float const* d_c, int ldc,
|
||||
float* d_d, int ldd) {
|
||||
WmmaGemmDispatch(
|
||||
int m,
|
||||
int n,
|
||||
int k,
|
||||
Scalar alpha,
|
||||
ScalarA const* d_a,
|
||||
int lda,
|
||||
ScalarB const* d_b,
|
||||
int ldb,
|
||||
Scalar beta,
|
||||
ScalarC const* d_c,
|
||||
int ldc,
|
||||
ScalarD* d_d,
|
||||
int ldd) {
|
||||
|
||||
params.initialize(m, n, k, alpha, d_a, lda, d_b, ldb, beta, d_c, ldc, d_d, ldd);
|
||||
}
|
||||
@@ -76,33 +97,6 @@ struct WmmaGemmDispatch {
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
|
||||
/// Determines if problem is aligned (assuming no padding)
|
||||
static bool is_problem_aligned(
|
||||
int m,
|
||||
int n,
|
||||
int k) {
|
||||
|
||||
bool aligned = true;
|
||||
|
||||
if (kLayoutA == cutlass::MatrixLayout::kColumnMajor) {
|
||||
aligned = aligned && !(m % Gemm::Traits::GemmConfig::kScalarsPerLdgA);
|
||||
}
|
||||
else {
|
||||
aligned = aligned && !(k % Gemm::Traits::GemmConfig::kScalarsPerLdgA);
|
||||
}
|
||||
|
||||
if (kLayoutB == cutlass::MatrixLayout::kColumnMajor) {
|
||||
aligned = aligned && !(k % Gemm::Traits::GemmConfig::kScalarsPerLdgB);
|
||||
}
|
||||
else {
|
||||
aligned = aligned && !(n % Gemm::Traits::GemmConfig::kScalarsPerLdgB);
|
||||
}
|
||||
|
||||
aligned = aligned && !(m % Gemm::Traits::GemmConfig::kScalarsPerLdgC);
|
||||
|
||||
return aligned;
|
||||
}
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
@@ -111,54 +105,49 @@ namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_wmma_gemm(TestbenchOutput &output, TestbenchOptions const &options) {
|
||||
|
||||
int profile_wmma_gemm_f32(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
typedef perf::GemmProfiler<cutlass::half_t, cutlass::half_t, float, float, float> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_nt", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_nt", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_nn", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_nn", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_tn", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_tn", options, config);
|
||||
}
|
||||
|
||||
if (!results) {
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_tt", options);
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_tt", options, config);
|
||||
}
|
||||
|
||||
return results;
|
||||
@@ -166,6 +155,112 @@ int profile_wmma_gemm(TestbenchOutput &output, TestbenchOptions const &options)
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_wmma_gemm_f16(
|
||||
TestbenchOutput<GemmProblem> &output,
|
||||
TestbenchOptions const &options,
|
||||
Config const &config) {
|
||||
|
||||
typedef perf::GemmProfiler<
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t,
|
||||
cutlass::half_t> GemmProfiler;
|
||||
|
||||
int results = 0;
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<32, 128, 128>,
|
||||
half,
|
||||
half,
|
||||
half,
|
||||
cutlass::gemm::LinearScaling<half>,
|
||||
half,
|
||||
cutlass::Shape<32, 64, 64>
|
||||
>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_f16_nt", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<32, 128, 128>,
|
||||
half,
|
||||
half,
|
||||
half,
|
||||
cutlass::gemm::LinearScaling<half>,
|
||||
half,
|
||||
cutlass::Shape<32, 64, 64>
|
||||
>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_f16_nn", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<32, 128, 128>,
|
||||
half,
|
||||
half,
|
||||
half,
|
||||
cutlass::gemm::LinearScaling<half>,
|
||||
half,
|
||||
cutlass::Shape<32, 64, 64>
|
||||
>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_f16_tn", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<32, 128, 128>,
|
||||
half,
|
||||
half,
|
||||
half,
|
||||
cutlass::gemm::LinearScaling<half>,
|
||||
half,
|
||||
cutlass::Shape<32, 64, 64>
|
||||
>
|
||||
WmmaGemmTraits;
|
||||
|
||||
typedef WmmaGemmDispatch<WmmaGemmTraits> Dispatch;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_gemm_f16_tt", options, config);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
struct WmmaGemmRegistrar {
|
||||
WmmaGemmRegistrar() {
|
||||
RegisterGemmProfileFunc(profile_wmma_gemm_f32);
|
||||
RegisterGemmProfileFunc(profile_wmma_gemm_f16);
|
||||
}
|
||||
};
|
||||
|
||||
volatile WmmaGemmRegistrar _WmmaGemmRegistrar;
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -0,0 +1,455 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#include "tools/test/perf/cutlass_perf_test.h"
|
||||
#include "tools/test/perf/gemm/gemm_perf_testbed.h"
|
||||
#include "tools/test/perf/gemm/gemm_profiler.h"
|
||||
|
||||
#include "cutlass/wmma_matrix.h"
|
||||
#ifdef CUTLASS_USE_WMMA_API
|
||||
#ifdef CUTLASS_USE_SUBBYTE_WMMA
|
||||
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/wmma_gemm_traits.h"
|
||||
#include "tools/test/perf/gemm/cutlass_dispatch.h"
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename Traits, typename ScalarA, typename ScalarB>
|
||||
struct WmmaIntegerGemmDispatch {
|
||||
|
||||
typedef cutlass::gemm::Gemm<Traits> Gemm;
|
||||
|
||||
typedef typename Gemm::Params Params;
|
||||
|
||||
/// Indicate warp-level GEMM
|
||||
static bool const kThreadMultiplyAdd = false;
|
||||
|
||||
static bool const kRunCuBLAS = false;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Traits::kLayoutB;
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
/// Params argument
|
||||
Params params;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
WmmaIntegerGemmDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaIntegerGemmDispatch(int m, int n, int k, int alpha,
|
||||
ScalarA const* d_a, int lda,
|
||||
ScalarB const* d_b, int ldb, int beta,
|
||||
int const* d_c, int ldc, int* d_d, int ldd) {
|
||||
|
||||
params.initialize(m, n, k, alpha, d_a, lda, d_b, ldb, beta, d_c, ldc, d_d, ldd);
|
||||
}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaIntegerGemmDispatch(Params const& _params) : params(_params) {}
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename Traits>
|
||||
struct WmmaIntegerGemmDispatch<Traits,
|
||||
cutlass::Vector<cutlass::int4_t, 8>,
|
||||
cutlass::Vector<cutlass::int4_t, 8> > {
|
||||
|
||||
typedef typename cutlass::Vector<cutlass::int4_t, 8> ScalarA;
|
||||
typedef typename cutlass::Vector<cutlass::int4_t, 8> ScalarB;
|
||||
|
||||
typedef cutlass::gemm::Gemm<Traits> Gemm;
|
||||
|
||||
typedef typename Gemm::Params Params;
|
||||
|
||||
/// Indicate warp-level GEMM
|
||||
static bool const kThreadMultiplyAdd = false;
|
||||
|
||||
static bool const kRunCuBLAS = false;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Traits::kLayoutB;
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
/// Params argument
|
||||
Params params;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
WmmaIntegerGemmDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaIntegerGemmDispatch(int m, int n, int k, int alpha,
|
||||
ScalarA const* d_a, int lda,
|
||||
ScalarB const* d_b, int ldb, int beta,
|
||||
int const* d_c, int ldc, int* d_d, int ldd) {
|
||||
|
||||
params.initialize(m, n, k * 8, alpha, d_a, lda, d_b, ldb, beta, d_c, ldc, d_d, ldd);
|
||||
}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaIntegerGemmDispatch(Params const& _params) : params(_params) {}
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template<typename Traits>
|
||||
struct WmmaIntegerGemmDispatch<Traits,
|
||||
cutlass::Vector<cutlass::uint4_t, 8>,
|
||||
cutlass::Vector<cutlass::uint4_t, 8> > {
|
||||
|
||||
typedef typename cutlass::Vector<cutlass::uint4_t, 8> ScalarA;
|
||||
typedef typename cutlass::Vector<cutlass::uint4_t, 8> ScalarB;
|
||||
|
||||
typedef cutlass::gemm::Gemm<Traits> Gemm;
|
||||
|
||||
typedef typename Gemm::Params Params;
|
||||
|
||||
/// Indicate warp-level GEMM
|
||||
static bool const kThreadMultiplyAdd = false;
|
||||
|
||||
static bool const kRunCuBLAS = false;
|
||||
|
||||
static cutlass::MatrixLayout::Kind const kLayoutA = Traits::kLayoutA;
|
||||
static cutlass::MatrixLayout::Kind const kLayoutB = Traits::kLayoutB;
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
/// Params argument
|
||||
Params params;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
WmmaIntegerGemmDispatch() {}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaIntegerGemmDispatch(int m, int n, int k, int alpha,
|
||||
ScalarA const* d_a, int lda,
|
||||
ScalarB const* d_b, int ldb, int beta,
|
||||
int const* d_c, int ldc, int* d_d, int ldd) {
|
||||
|
||||
params.initialize(m, n, k * 8, alpha, d_a, lda, d_b, ldb, beta, d_c, ldc, d_d, ldd);
|
||||
}
|
||||
|
||||
/// Initializes params object
|
||||
WmmaIntegerGemmDispatch(Params const& _params) : params(_params) {}
|
||||
|
||||
/// Launches kernel
|
||||
cudaError_t operator()() { return Gemm::launch(params); }
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
int profile_wmma_integer_gemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
|
||||
int results = 0;
|
||||
|
||||
// compute capability check
|
||||
if (!options.compute_capability(7, 5)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<128, 128, 128>,
|
||||
signed char,
|
||||
signed char,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<128, 32, 32>,
|
||||
cutlass::Shape<16, 16, 16>,
|
||||
16,
|
||||
16> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits, signed char, signed char> Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<signed char, signed char, int, int, int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_s8_16x16x16_nn", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<128, 128, 128>,
|
||||
signed char,
|
||||
signed char,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<128, 32, 32>,
|
||||
cutlass::Shape<16, 16, 16>,
|
||||
16,
|
||||
16> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits, signed char, signed char> Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<signed char, signed char, int, int, int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_s8_16x16x16_nt", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<128, 128, 128>,
|
||||
signed char,
|
||||
signed char,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<128, 32, 32>,
|
||||
cutlass::Shape<16, 16, 16>,
|
||||
16,
|
||||
16> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits, signed char, signed char> Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<signed char, signed char, int, int, int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_s8_16x16x16_tn", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<128, 128, 128>,
|
||||
signed char,
|
||||
signed char,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<128, 32, 32>,
|
||||
cutlass::Shape<16, 16, 16>,
|
||||
16,
|
||||
16> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits, signed char, signed char> Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<signed char, signed char, int, int, int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_s8_16x16x16_tt", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<128, 128, 128>,
|
||||
unsigned char,
|
||||
unsigned char,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<128, 32, 32>,
|
||||
cutlass::Shape<16, 16, 16>,
|
||||
16,
|
||||
16> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits, unsigned char, unsigned char> Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<unsigned char, unsigned char, int, int, int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_u8_16x16x16_nn", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<128, 128, 128>,
|
||||
unsigned char,
|
||||
unsigned char,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<128, 32, 32>,
|
||||
cutlass::Shape<16, 16, 16>,
|
||||
16,
|
||||
16> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits, unsigned char, unsigned char> Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<unsigned char, unsigned char, int, int, int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_u8_16x16x16_nt", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<128, 128, 128>,
|
||||
unsigned char,
|
||||
unsigned char,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<128, 32, 32>,
|
||||
cutlass::Shape<16, 16, 16>,
|
||||
16,
|
||||
16> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits, unsigned char, unsigned char> Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<unsigned char, unsigned char, int, int, int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_u8_16x16x16_tn", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::Shape<128, 128, 128>,
|
||||
unsigned char,
|
||||
unsigned char,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<128, 32, 32>,
|
||||
cutlass::Shape<16, 16, 16>,
|
||||
16,
|
||||
16> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits, unsigned char, unsigned char> Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<unsigned char, unsigned char, int, int, int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_u8_16x16x16_tt", options, config);
|
||||
}
|
||||
|
||||
// compute capability check
|
||||
if (!options.compute_capability_exact(7, 5)) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<256, 128, 128>,
|
||||
cutlass::Vector<cutlass::int4_t, 8>,
|
||||
cutlass::Vector<cutlass::int4_t, 8>,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<256, 32, 32>,
|
||||
cutlass::Shape<32, 8, 8>,
|
||||
32,
|
||||
32> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits,
|
||||
cutlass::Vector<cutlass::int4_t, 8>,
|
||||
cutlass::Vector<cutlass::int4_t, 8> > Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<cutlass::Vector<cutlass::int4_t, 8>,
|
||||
cutlass::Vector<cutlass::int4_t, 8>,
|
||||
int,
|
||||
int,
|
||||
int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_s4_tn", options, config);
|
||||
}
|
||||
|
||||
{
|
||||
typedef cutlass::gemm::WmmaGemmTraits<cutlass::MatrixLayout::kRowMajor,
|
||||
cutlass::MatrixLayout::kColumnMajor,
|
||||
cutlass::Shape<256, 128, 128>,
|
||||
cutlass::Vector<cutlass::uint4_t, 8>,
|
||||
cutlass::Vector<cutlass::uint4_t, 8>,
|
||||
int,
|
||||
cutlass::gemm::LinearScaling<int>,
|
||||
int,
|
||||
cutlass::Shape<256, 32, 32>,
|
||||
cutlass::Shape<32, 8, 8>,
|
||||
32,
|
||||
32> WmmaGemmTraits;
|
||||
|
||||
typedef WmmaIntegerGemmDispatch<WmmaGemmTraits,
|
||||
cutlass::Vector<cutlass::uint4_t, 8>,
|
||||
cutlass::Vector<cutlass::uint4_t, 8> > Dispatch;
|
||||
|
||||
typedef perf::GemmProfiler<cutlass::Vector<cutlass::uint4_t, 8>,
|
||||
cutlass::Vector<cutlass::uint4_t, 8>,
|
||||
int,
|
||||
int,
|
||||
int> GemmProfiler;
|
||||
|
||||
results |= profile_gemm<Dispatch, GemmProfiler>(output, "wmma_integer_gemm_u4_tn", options, config);
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#else // ! CUTLASS_USE_SUBBYTE_WMMA
|
||||
|
||||
namespace perf {
|
||||
|
||||
int profile_wmma_integer_gemm(TestbenchOutput<GemmProblem> &output, TestbenchOptions const &options, Config const &config) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
struct WmmaIntegerGemmRegistrar {
|
||||
WmmaIntegerGemmRegistrar() { perf::RegisterGemmProfileFunc(perf::profile_wmma_integer_gemm); }
|
||||
};
|
||||
|
||||
volatile WmmaIntegerGemmRegistrar _WmmaIntegerGemmRegistrar;
|
||||
|
||||
#endif // ifdef CUTLASS_USE_WMMA_API
|
||||
@@ -25,25 +25,39 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cutlass/matrix_traits.h>
|
||||
#include <tools/util/command_line.h>
|
||||
#include "cutlass/matrix_traits.h"
|
||||
#include "tools/util/command_line.h"
|
||||
#include "tools/test/perf/provider.h"
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Outcome of test
|
||||
struct Disposition {
|
||||
enum Kind { Unknown = 0, NotRun, Passed, Incorrect, Failed, NotVerified, Invalid };
|
||||
enum Kind {
|
||||
Unknown = 0,
|
||||
NotRun,
|
||||
Passed,
|
||||
Incorrect,
|
||||
Failed,
|
||||
NotVerified,
|
||||
Invalid
|
||||
};
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
inline std::ostream &operator<<(std::ostream &out, perf::Disposition::Kind value) {
|
||||
char const *str[] = {
|
||||
"unknown", "not_run", "passed", "incorrect", "failed", "not_verified", "invalid"};
|
||||
inline std::ostream &operator<<(std::ostream &out, Disposition::Kind value) {
|
||||
char const *str[] = {"unknown",
|
||||
"not_run",
|
||||
"passed",
|
||||
"incorrect",
|
||||
"failed",
|
||||
"not_verified",
|
||||
"invalid"};
|
||||
if (value >= perf::Disposition::Unknown && value < perf::Disposition::Invalid) {
|
||||
out << str[value];
|
||||
} else {
|
||||
@@ -62,10 +76,6 @@ inline std::ostream &operator<<(std::ostream &out, cutlass::MatrixLayout::Kind l
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Size and layout of a GEMM problem
|
||||
struct GemmProblem {
|
||||
//
|
||||
@@ -86,7 +96,7 @@ struct GemmProblem {
|
||||
//
|
||||
|
||||
/// Static method to print GemmProblem headers
|
||||
static std::string header() { return "M, N, K, Layout_A, Layout_B, Beta"; }
|
||||
static std::string header() { return "M,N,K,Layout_A,Layout_B,Beta"; }
|
||||
|
||||
//
|
||||
// Methods
|
||||
@@ -129,34 +139,27 @@ struct GemmProblem {
|
||||
}
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Prints a problem to an output stream
|
||||
inline std::ostream &operator<<(std::ostream &out, perf::GemmProblem const &problem) {
|
||||
out << problem.m << ", " << problem.n << ", " << problem.k << ", " << problem.layout_A << ", "
|
||||
<< problem.layout_B << ", " << problem.beta;
|
||||
inline std::ostream &operator<<(std::ostream &out, GemmProblem const &problem) {
|
||||
out << problem.m << "," << problem.n << "," << problem.k << "," << problem.layout_A << ","
|
||||
<< problem.layout_B << "," << problem.beta;
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Result object
|
||||
template <typename Problem>
|
||||
struct PerformanceResult {
|
||||
/// Provider of GEMM implementation
|
||||
Provider::Kind provider;
|
||||
|
||||
/// Name of kernel
|
||||
std::string kernel_name;
|
||||
|
||||
/// Problem size
|
||||
GemmProblem problem;
|
||||
Problem problem;
|
||||
|
||||
/// Outcome of test
|
||||
Disposition::Kind disposition;
|
||||
@@ -166,40 +169,45 @@ struct PerformanceResult {
|
||||
|
||||
/// Throughput in units of GFLOPs
|
||||
double gflops;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
PerformanceResult(
|
||||
std::string const &_kernel_name = "",
|
||||
GemmProblem const &_problem = GemmProblem(),
|
||||
Disposition::Kind _disposition = Disposition::NotRun,
|
||||
double _runtime = 0,
|
||||
double _gflops = 0)
|
||||
:
|
||||
kernel_name(_kernel_name),
|
||||
problem(_problem),
|
||||
disposition(_disposition),
|
||||
runtime(_runtime),
|
||||
gflops(_gflops) {}
|
||||
PerformanceResult(Provider::Kind _provider = Provider::Cutlass
|
||||
, std::string const &_kernel_name = ""
|
||||
, Problem const &_problem = Problem()
|
||||
, Disposition::Kind _disposition = Disposition::NotRun
|
||||
, double _runtime = 0
|
||||
, double _gflops = 0
|
||||
):
|
||||
provider(_provider)
|
||||
, kernel_name(_kernel_name)
|
||||
, problem(_problem)
|
||||
, disposition(_disposition)
|
||||
, runtime(_runtime)
|
||||
, gflops(_gflops)
|
||||
{}
|
||||
|
||||
/// Displays headers
|
||||
static std::string header() {
|
||||
return std::string("Kernel, ") + GemmProblem::header() +
|
||||
", Disposition, Runtime, GFLOPs";
|
||||
std::stringstream ss;
|
||||
|
||||
ss << "Provider,Kernel," << Problem::header();
|
||||
ss << ",Disposition,Runtime,GFLOPs";
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
/// Prints human-readable results
|
||||
std::ostream &pretty_print(std::ostream &out) const {
|
||||
|
||||
out << "Kernel: \033[1m" << kernel_name << "\033[0m\n"
|
||||
<< " provider: " << provider << "\n"
|
||||
<< " problem: ";
|
||||
|
||||
std::stringstream disposition_str;
|
||||
if (disposition == Disposition::Passed) {
|
||||
disposition_str << "\033[1m";
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
disposition_str << "\033[1;31m";
|
||||
}
|
||||
disposition_str << disposition << "\033[0m";
|
||||
@@ -215,15 +223,16 @@ struct PerformanceResult {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
/// Outputs result
|
||||
inline std::ostream &operator<<(std::ostream &out, perf::PerformanceResult const &result) {
|
||||
template <typename Problem>
|
||||
inline std::ostream &operator<<(std::ostream &out, PerformanceResult<Problem> const &result) {
|
||||
|
||||
out << result.kernel_name << ", " << result.problem << ", "
|
||||
<< result.disposition << ", " << result.runtime << ", " << result.gflops;
|
||||
out << result.provider << "," << result.kernel_name << "," << result.problem << ","
|
||||
<< result.disposition << "," << result.runtime << "," << result.gflops;
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <iosfwd>
|
||||
|
||||
namespace perf {
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Implementation under test
|
||||
struct Provider {
|
||||
enum Kind {
|
||||
Unknown = 0,
|
||||
Cutlass,
|
||||
Invalid
|
||||
};
|
||||
|
||||
static Provider::Kind from_string(std::string const &str) {
|
||||
if (str == "cutlass" || str == "Cutlass") {
|
||||
return Cutlass;
|
||||
}
|
||||
else {
|
||||
return Invalid;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// Prints provider
|
||||
inline std::ostream &operator<<(std::ostream &out, Provider::Kind provider) {
|
||||
char const *str[] = {
|
||||
"unknown",
|
||||
"Cutlass",
|
||||
"invalid"
|
||||
};
|
||||
if (provider >= perf::Provider::Unknown && provider < perf::Provider::Invalid) {
|
||||
out << str[provider];
|
||||
} else {
|
||||
out << str[perf::Provider::Invalid];
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace perf
|
||||
|
||||
|
||||
@@ -0,0 +1,189 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without modification, are permitted
|
||||
* provided that the following conditions are met:
|
||||
* * Redistributions of source code must retain the above copyright notice, this list of
|
||||
* conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright notice, this list of
|
||||
* conditions and the following disclaimer in the documentation and/or other materials
|
||||
* provided with the distribution.
|
||||
* * Neither the name of the NVIDIA CORPORATION nor the names of its contributors may be used
|
||||
* to endorse or promote products derived from this software without specific prior written
|
||||
* permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND ANY EXPRESS OR
|
||||
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND
|
||||
* FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL NVIDIA CORPORATION BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
|
||||
* BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
|
||||
* OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
|
||||
* STRICT LIABILITY, OR TOR (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <algorithm>
|
||||
#include <fstream>
|
||||
#include <string>
|
||||
|
||||
#include "tools/test/perf/testbench_options.h"
|
||||
|
||||
namespace perf {
|
||||
|
||||
// Structure of configurations to run
|
||||
struct Config {
|
||||
// Scalar value for GEMM
|
||||
double alpha;
|
||||
|
||||
/// Scalar value for GEMM
|
||||
double beta;
|
||||
|
||||
// kernel to run
|
||||
std::vector<std::string> kernels;
|
||||
|
||||
/// Range of problem sizes
|
||||
GemmProblemRange problem_range;
|
||||
|
||||
// Reference GFLOPs
|
||||
double gflops_ref;
|
||||
|
||||
// Reference Runtime
|
||||
double runtime_ref;
|
||||
|
||||
// Reference Peak Throughput
|
||||
double peak_throughput_ref;
|
||||
|
||||
// Returns true if the kernel name appears among the enabled kernels
|
||||
bool kernel_enabled(std::string const &kernel) const {
|
||||
typedef std::vector<std::string>::const_iterator kernel_iterator;
|
||||
|
||||
for (kernel_iterator it = kernels.begin(); it != kernels.end(); ++it) {
|
||||
if (kernel.find(*it) != std::string::npos) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
};
|
||||
|
||||
// Class to set the configurations to run
|
||||
struct TestbenchConfigs {
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
// Vector of configurations to run
|
||||
std::vector<perf::Config> configs;
|
||||
|
||||
// Options to test environment
|
||||
TestbenchOptions options;
|
||||
|
||||
// Input CSV file to read (if applicable)
|
||||
std::ifstream threshold_file;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
// Determines the configurations to run from the threshold file
|
||||
void configs_from_file() {
|
||||
// Set the values of kernels, M, N, K and beta based off of values read from CSVs
|
||||
threshold_file.open(options.threshold_filename.c_str());
|
||||
if (threshold_file.is_open()) {
|
||||
std::string line;
|
||||
int provider_idx = -1;
|
||||
int kernel_idx = -1;
|
||||
int beta_idx = -1;
|
||||
int m_idx = -1;
|
||||
int n_idx = -1;
|
||||
int k_idx = -1;
|
||||
int gflops_idx = -1;
|
||||
int runtime_idx = -1;
|
||||
int peak_throughput_idx = -1;
|
||||
|
||||
// Read the header and get the indices of the columns
|
||||
if (getline(threshold_file, line)) {
|
||||
char delim = ',';
|
||||
size_t s_idx = 0;
|
||||
size_t d_idx = std::string::npos;
|
||||
int idx = 0;
|
||||
line.erase(std::remove(line.begin(), line.end(), ' '), line.end());
|
||||
while (s_idx < line.size()) {
|
||||
d_idx = line.find_first_of(delim, s_idx);
|
||||
size_t end_idx = (d_idx != std::string::npos ? d_idx : line.size());
|
||||
std::string item = line.substr(s_idx, end_idx - s_idx);
|
||||
if (item.compare("Provider") == 0) provider_idx = idx;
|
||||
if (item.compare("Kernel") == 0) kernel_idx = idx;
|
||||
if (item.compare("Beta") == 0) beta_idx = idx;
|
||||
if (item.compare("M") == 0) m_idx = idx;
|
||||
if (item.compare("N") == 0) n_idx = idx;
|
||||
if (item.compare("K") == 0) k_idx = idx;
|
||||
if (item.compare("GFLOPs") == 0) gflops_idx = idx;
|
||||
if (item.compare("Runtime") == 0) runtime_idx = idx;
|
||||
if (item.compare("SOL") == 0) peak_throughput_idx = idx;
|
||||
s_idx = end_idx + 1; // For comma
|
||||
idx++;
|
||||
}
|
||||
}
|
||||
|
||||
while (getline(threshold_file, line)) {
|
||||
char delim = ',';
|
||||
size_t s_idx = 0;
|
||||
size_t d_idx = std::string::npos;
|
||||
std::vector<std::string> tokens;
|
||||
line.erase(std::remove(line.begin(), line.end(), ' '), line.end());
|
||||
while (s_idx < line.size()) {
|
||||
d_idx = line.find_first_of(delim, s_idx);
|
||||
size_t end_idx = (d_idx != std::string::npos ? d_idx : line.size());
|
||||
std::string item = line.substr(s_idx, end_idx - s_idx);
|
||||
tokens.push_back(item);
|
||||
s_idx = end_idx + 1; // For comma
|
||||
}
|
||||
if (tokens[provider_idx].compare("Cutlass") == 0) {
|
||||
// Create a new config
|
||||
Config config = Config();
|
||||
config.alpha = options.alpha;
|
||||
config.beta = strtod(tokens[beta_idx].c_str(), NULL);
|
||||
config.kernels.push_back(tokens[kernel_idx]);
|
||||
config.problem_range.M = Range((int)strtol(tokens[m_idx].c_str(), NULL, 10));
|
||||
config.problem_range.N = Range((int)strtol(tokens[n_idx].c_str(), NULL, 10));
|
||||
config.problem_range.K = Range((int)strtol(tokens[k_idx].c_str(), NULL, 10));
|
||||
config.gflops_ref = strtod(tokens[gflops_idx].c_str(), NULL);
|
||||
config.runtime_ref = strtod(tokens[runtime_idx].c_str(), NULL);
|
||||
config.peak_throughput_ref = strtod(tokens[peak_throughput_idx].c_str(), NULL);
|
||||
configs.push_back(config);
|
||||
}
|
||||
}
|
||||
} else { // !threshold_file.is_open()
|
||||
std::cout << "ERROR: Could not open threshold file " << options.threshold_filename << "\n";
|
||||
}
|
||||
}
|
||||
|
||||
// Determines the configurations to run from the command line arguments
|
||||
void configs_from_args() {
|
||||
Config config = Config();
|
||||
config.alpha = options.alpha;
|
||||
config.beta = options.beta;
|
||||
for (int i = 0; i < options.kernels.size(); i++) {
|
||||
config.kernels.push_back(options.kernels[i]);
|
||||
}
|
||||
config.problem_range = options.problem_range;
|
||||
configs.push_back(config);
|
||||
}
|
||||
|
||||
// Constructor
|
||||
TestbenchConfigs(TestbenchOptions const &_options) : options(_options) {
|
||||
if (!options.threshold_filename.empty()) {
|
||||
configs_from_file();
|
||||
} else {
|
||||
configs_from_args();
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace perf
|
||||
+240
-173
@@ -25,8 +25,16 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
#include <cublas_v2.h>
|
||||
|
||||
#include <stdint.h>
|
||||
#include <tools/util/command_line.h>
|
||||
#include <stdexcept>
|
||||
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "tools/util/command_line.h"
|
||||
#include "tools/util/distribution.h"
|
||||
#include "tools/test/perf/provider.h"
|
||||
|
||||
namespace perf {
|
||||
|
||||
@@ -34,14 +42,73 @@ namespace perf {
|
||||
|
||||
/// Range of problem sizes
|
||||
struct Range {
|
||||
|
||||
enum Operator {
|
||||
Add,
|
||||
Multiply
|
||||
};
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
|
||||
int start;
|
||||
int end;
|
||||
int increment;
|
||||
Operator increment_op;
|
||||
|
||||
Range(int _start = 0) : start(_start), end(_start), increment(1) {}
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
Range(int _start, int _end, int _increment = 1)
|
||||
: start(_start), end(_end), increment(_increment) {}
|
||||
Range(int _start = 0) : start(_start), end(_start), increment(1), increment_op(Add) {}
|
||||
|
||||
Range(int _start, int _end, int _increment = 1, Operator _op = Add)
|
||||
: start(_start), end(_end), increment(_increment), increment_op(_op) {}
|
||||
|
||||
/// Returns the next item in series
|
||||
int next(int val) const {
|
||||
switch (increment_op) {
|
||||
case Add: val += increment; break;
|
||||
case Multiply: val *= increment; break;
|
||||
default: val = end; break;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
void import_from_strings(const std::vector<std::string>& values) {
|
||||
if (values.size() > 0) {
|
||||
std::stringstream ss;
|
||||
ss << values.at(0);
|
||||
ss >> start;
|
||||
}
|
||||
|
||||
if (values.size() > 1) {
|
||||
std::stringstream ss;
|
||||
ss << values.at(1);
|
||||
ss >> end;
|
||||
} else {
|
||||
end = start;
|
||||
}
|
||||
|
||||
if (values.size() > 2 && !values.at(2).empty()) {
|
||||
std::stringstream ss;
|
||||
|
||||
char first = values.at(2).at(0);
|
||||
if (first == '*' || first == '+') {
|
||||
ss << values.at(2).substr(1);
|
||||
switch (first) {
|
||||
case '*': increment_op = Multiply; break;
|
||||
case '+': increment_op = Add; break;
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
else {
|
||||
ss << values.at(2);
|
||||
}
|
||||
ss >> increment;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
@@ -77,25 +144,7 @@ struct GemmProblemRange {
|
||||
std::vector<std::string> values;
|
||||
args.get_cmd_line_arguments(arg.c_str(), values, ':');
|
||||
|
||||
if (values.size() > 0) {
|
||||
std::stringstream ss;
|
||||
ss << values.at(0);
|
||||
ss >> range.start;
|
||||
}
|
||||
|
||||
if (values.size() > 1) {
|
||||
std::stringstream ss;
|
||||
ss << values.at(1);
|
||||
ss >> range.end;
|
||||
} else {
|
||||
range.end = range.start;
|
||||
}
|
||||
|
||||
if (values.size() > 2) {
|
||||
std::stringstream ss;
|
||||
ss << values.at(2);
|
||||
ss >> range.increment;
|
||||
}
|
||||
range.import_from_strings(values);
|
||||
} else {
|
||||
range = _default;
|
||||
}
|
||||
@@ -111,105 +160,6 @@ struct GemmProblemRange {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Distribution type
|
||||
struct Distribution {
|
||||
/// Variant types
|
||||
enum Kind { Invalid, Uniform, Gaussian, Linear, Identity };
|
||||
|
||||
/// Distribution state
|
||||
union {
|
||||
/// Uniform distribution
|
||||
struct {
|
||||
double min;
|
||||
double max;
|
||||
} uniform;
|
||||
|
||||
/// Gaussian distribution
|
||||
struct {
|
||||
double mean;
|
||||
double stddev;
|
||||
} gaussian;
|
||||
|
||||
/// Elements are linear combination of row and column index
|
||||
struct {
|
||||
double offset;
|
||||
double delta_row;
|
||||
double delta_column;
|
||||
} linear;
|
||||
};
|
||||
|
||||
/// Active variant kind
|
||||
Kind kind;
|
||||
|
||||
/// Random values are cast to integer after scaling by this power of two
|
||||
int int_scale;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
Distribution() : kind(Invalid), int_scale(0) {}
|
||||
|
||||
/// Configures distribution as uniform random
|
||||
Distribution &set_uniform(double _min, double _max, int _int_scale = 0) {
|
||||
kind = Uniform;
|
||||
uniform.min = _min;
|
||||
uniform.max = _max;
|
||||
int_scale = _int_scale;
|
||||
return *this;
|
||||
}
|
||||
|
||||
/// Configures distribution as Gaussian distribution
|
||||
Distribution &set_gaussian(double _mean, double _stddev, int _int_scale = 0) {
|
||||
kind = Gaussian;
|
||||
gaussian.mean = _mean;
|
||||
gaussian.stddev = _stddev;
|
||||
int_scale = _int_scale;
|
||||
return *this;
|
||||
}
|
||||
|
||||
|
||||
/// Sets identity
|
||||
Distribution &set_identity() {
|
||||
kind = Identity;
|
||||
return *this;
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace perf
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Prints a Distribution to ostream
|
||||
inline std::ostream &operator<<(std::ostream &out, perf::Distribution const &dist) {
|
||||
switch (dist.kind) {
|
||||
case perf::Distribution::Uniform:
|
||||
out << "uniorm, min: " << dist.uniform.min << ", max: " << dist.uniform.max;
|
||||
break;
|
||||
case perf::Distribution::Gaussian:
|
||||
out << "gaussian, mean: " << dist.gaussian.mean << ", stddev: " << dist.gaussian.stddev;
|
||||
break;
|
||||
case perf::Distribution::Linear:
|
||||
out << "linear, mean: " << dist.linear.offset << ", delta_row: " << dist.linear.delta_row
|
||||
<< ", delta_column: " << dist.linear.delta_column;
|
||||
break;
|
||||
case perf::Distribution::Identity:
|
||||
break;
|
||||
default:
|
||||
out << "unknown";
|
||||
}
|
||||
|
||||
out << ", int_scale: " << dist.int_scale;
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Defines a vector of string pairs
|
||||
typedef std::vector<std::pair<std::string, std::string> > KeyValueVector;
|
||||
|
||||
@@ -219,13 +169,13 @@ typedef KeyValueVector::const_iterator KeyValueIterator;
|
||||
/// Structure captures the initial configuration of matrices
|
||||
struct InitialDistribution {
|
||||
/// Distribution of A matrix operand
|
||||
Distribution dist_A;
|
||||
cutlass::Distribution dist_A;
|
||||
|
||||
/// Distribution of B matrix operand
|
||||
Distribution dist_B;
|
||||
cutlass::Distribution dist_B;
|
||||
|
||||
/// Distribution of C matrix operand
|
||||
Distribution dist_C;
|
||||
/// cutlass::Distribution of C matrix operand
|
||||
cutlass::Distribution dist_C;
|
||||
|
||||
/// Seed for random number generation
|
||||
int64_t seed;
|
||||
@@ -237,15 +187,15 @@ struct InitialDistribution {
|
||||
/// Gets the initial distribution
|
||||
static void get_distribution(cutlass::CommandLine const &args,
|
||||
std::string const &arg,
|
||||
Distribution &dist) {
|
||||
cutlass::Distribution &dist) {
|
||||
struct {
|
||||
const char *label;
|
||||
Distribution::Kind kind;
|
||||
} distribution_kinds[] = {{"uniform", Distribution::Uniform},
|
||||
{"gaussian", Distribution::Gaussian},
|
||||
{"linear", Distribution::Linear},
|
||||
{"identity", Distribution::Identity},
|
||||
{0, Distribution::Invalid}};
|
||||
cutlass::Distribution::Kind kind;
|
||||
} distribution_kinds[] = {{"uniform", cutlass::Distribution::Uniform},
|
||||
{"gaussian", cutlass::Distribution::Gaussian},
|
||||
{"linear", cutlass::Distribution::Linear},
|
||||
{"identity", cutlass::Distribution::Identity},
|
||||
{0, cutlass::Distribution::Invalid}};
|
||||
|
||||
struct {
|
||||
char const *label;
|
||||
@@ -276,13 +226,17 @@ struct InitialDistribution {
|
||||
|
||||
// Subsequent key-value pairs update the named field of the distribution struct.
|
||||
for (; it != values.end(); ++it) {
|
||||
|
||||
// Integer scaling factor - if < 0, no integer rounding is performed.
|
||||
if (it->first == "scale" && !it->second.empty()) {
|
||||
std::stringstream ss;
|
||||
ss << it->second;
|
||||
ss >> dist.int_scale;
|
||||
continue; // next token
|
||||
}
|
||||
|
||||
// Casts as integer without scaling
|
||||
if (it->first == "integer") {
|
||||
dist.int_scale = 0;
|
||||
continue; // next token
|
||||
}
|
||||
|
||||
@@ -326,12 +280,12 @@ struct InitialDistribution {
|
||||
args.get_cmd_line_argument("seed", seed, seed);
|
||||
|
||||
// Update all distributions at once
|
||||
Distribution dist_all;
|
||||
cutlass::Distribution dist_all;
|
||||
if (args.check_cmd_line_flag("dist")) {
|
||||
get_distribution(args, "dist", dist_all);
|
||||
dist_A = dist_all;
|
||||
dist_B = dist_all;
|
||||
dist_C = dist_all;
|
||||
get_distribution(args, "dist", dist_all);
|
||||
dist_A = dist_all;
|
||||
dist_B = dist_all;
|
||||
dist_C = dist_all;
|
||||
}
|
||||
|
||||
get_distribution(args, "dist_A", dist_A);
|
||||
@@ -344,19 +298,18 @@ struct InitialDistribution {
|
||||
|
||||
/// Defines how to execute the benchmarks
|
||||
struct ExecutionMode {
|
||||
enum Kind {
|
||||
Profile,
|
||||
Verify,
|
||||
Single,
|
||||
Invalid
|
||||
};
|
||||
enum Kind { Profile, Verify, Single, Invalid };
|
||||
|
||||
static std::string to_string(Kind kind) {
|
||||
switch (kind) {
|
||||
case Profile: return "profile";
|
||||
case Verify: return "verify";
|
||||
case Single: return "single";
|
||||
default: return "invalid";
|
||||
case Profile:
|
||||
return "profile";
|
||||
case Verify:
|
||||
return "verify";
|
||||
case Single:
|
||||
return "single";
|
||||
default:
|
||||
return "invalid";
|
||||
}
|
||||
}
|
||||
|
||||
@@ -370,18 +323,18 @@ struct ExecutionMode {
|
||||
|
||||
/// Indicates when the workspace is saved
|
||||
struct WorkspaceSaveMode {
|
||||
enum Kind {
|
||||
Never,
|
||||
Incorrect,
|
||||
Always
|
||||
};
|
||||
enum Kind { Never, Incorrect, Always };
|
||||
|
||||
static std::string to_string(Kind kind) {
|
||||
switch (kind) {
|
||||
case Never: return "never";
|
||||
case Incorrect: return "incorrect";
|
||||
case Always: return "always";
|
||||
default: return "incorrect";
|
||||
case Never:
|
||||
return "never";
|
||||
case Incorrect:
|
||||
return "incorrect";
|
||||
case Always:
|
||||
return "always";
|
||||
default:
|
||||
return "incorrect";
|
||||
}
|
||||
}
|
||||
|
||||
@@ -397,7 +350,6 @@ struct WorkspaceSaveMode {
|
||||
|
||||
/// Class holding testbench command line options
|
||||
struct TestbenchOptions {
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
@@ -408,18 +360,24 @@ struct TestbenchOptions {
|
||||
// Path to output file name
|
||||
std::string output_filename;
|
||||
|
||||
// Path to input file name
|
||||
std::string threshold_filename;
|
||||
|
||||
/// If true, output is appended
|
||||
bool append;
|
||||
|
||||
/// Number of iterations
|
||||
int iterations;
|
||||
|
||||
|
||||
/// Defines how to run the benchmark
|
||||
ExecutionMode::Kind execution_mode;
|
||||
|
||||
/// Indicates when the workspace is saved
|
||||
WorkspaceSaveMode::Kind save_workspace_mode;
|
||||
|
||||
/// Properties of CUDA device
|
||||
cudaDeviceProp device_properties;
|
||||
|
||||
/// Enabled kernel names
|
||||
std::vector<std::string> kernels;
|
||||
|
||||
@@ -432,12 +390,21 @@ struct TestbenchOptions {
|
||||
/// Range of problem sizes
|
||||
GemmProblemRange problem_range;
|
||||
|
||||
/// If true, kernels are not executed, and no sleep waits are inserted
|
||||
bool dry_run;
|
||||
|
||||
/// Tags to describe the profiler output
|
||||
KeyValueVector pivot_tags;
|
||||
|
||||
/// If enabled, only the peak performance for a given kernel is reported
|
||||
bool peak_performance;
|
||||
|
||||
/// Performance Degradatiom Margin before flagging as test failure
|
||||
double perf_margin;
|
||||
|
||||
/// Cool-down period
|
||||
int sleep_time;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
@@ -447,26 +414,47 @@ struct TestbenchOptions {
|
||||
: initial_distribution(args),
|
||||
execution_mode(ExecutionMode::Profile),
|
||||
save_workspace_mode(WorkspaceSaveMode::Never),
|
||||
problem_range(args) {
|
||||
problem_range(args),
|
||||
dry_run(false),
|
||||
sleep_time(1) {
|
||||
|
||||
// Set the CUDA device and/or specify clock rate
|
||||
configure_cuda_device(args);
|
||||
|
||||
// fetch command line arguments
|
||||
args.get_cmd_line_argument("iterations", iterations, 25);
|
||||
args.get_cmd_line_argument("append", append, false);
|
||||
args.get_cmd_line_argument("output", output_filename);
|
||||
args.get_cmd_line_argument("threshold", threshold_filename);
|
||||
args.get_cmd_line_argument("alpha", alpha, 1.0);
|
||||
args.get_cmd_line_argument("beta", beta, 0.0);
|
||||
args.get_cmd_line_argument("peak", peak_performance, false);
|
||||
args.get_cmd_line_argument_pairs("tags", pivot_tags);
|
||||
args.get_cmd_line_argument("perf-margin", perf_margin, 0.97);
|
||||
args.get_cmd_line_argument("dry-run", dry_run, false);
|
||||
args.get_cmd_line_argument("sleep-time", sleep_time, 1);
|
||||
|
||||
if (args.check_cmd_line_flag("execution_mode")) {
|
||||
if (args.check_cmd_line_flag("execution-mode")) {
|
||||
std::string str;
|
||||
args.get_cmd_line_argument("execution_mode", str);
|
||||
args.get_cmd_line_argument("execution-mode", str);
|
||||
execution_mode = ExecutionMode::from_string(str);
|
||||
}
|
||||
|
||||
if (args.check_cmd_line_flag("save_workspace")) {
|
||||
if (args.check_cmd_line_flag("save-workspace")) {
|
||||
std::string str;
|
||||
args.get_cmd_line_argument("save_workspace", str);
|
||||
args.get_cmd_line_argument("save-workspace", str);
|
||||
save_workspace_mode = WorkspaceSaveMode::from_string(str);
|
||||
}
|
||||
|
||||
if (args.check_cmd_line_flag("execution-mode")) {
|
||||
std::string str;
|
||||
args.get_cmd_line_argument("execution-mode", str);
|
||||
execution_mode = ExecutionMode::from_string(str);
|
||||
}
|
||||
|
||||
if (args.check_cmd_line_flag("save-workspace")) {
|
||||
std::string str;
|
||||
args.get_cmd_line_argument("save-workspace", str);
|
||||
save_workspace_mode = WorkspaceSaveMode::from_string(str);
|
||||
}
|
||||
|
||||
@@ -474,13 +462,50 @@ struct TestbenchOptions {
|
||||
if (args.check_cmd_line_flag("kernels")) {
|
||||
args.get_cmd_line_arguments("kernels", kernels, ',');
|
||||
} else {
|
||||
char const *gemms[] = {"sgemm", "dgemm", "hgemm", "igemm", "wmma_gemm", 0};
|
||||
char const *gemms[] = {
|
||||
"sgemm",
|
||||
"dgemm",
|
||||
"hgemm",
|
||||
"igemm",
|
||||
"wmma_gemm",
|
||||
"wmma_gemm_f16",
|
||||
"wmma_binary_gemm",
|
||||
"wmma_integer_gemm",
|
||||
0
|
||||
};
|
||||
char const *layouts[] = {"nn", "nt", "tn", "tt", 0};
|
||||
for (int i = 0; gemms[i]; ++i) {
|
||||
for (int j = 0; layouts[j]; ++j) {
|
||||
if ((std::string(gemms[i]).compare("wmma_binary_gemm") == 0 ||
|
||||
std::string(gemms[i]).compare("wmma_integer_gemm") == 0)
|
||||
&& std::string(layouts[j]).compare("tn") != 0) {
|
||||
continue;
|
||||
}
|
||||
kernels.push_back(std::string(gemms[i]) + "_" + layouts[j]);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
void configure_cuda_device(cutlass::CommandLine const &args) {
|
||||
int device_id = 0;
|
||||
args.get_cmd_line_argument("device", device_id, 0);
|
||||
|
||||
cudaError_t result;
|
||||
result = cudaGetDeviceProperties(&device_properties, device_id);
|
||||
if (result != cudaSuccess) {
|
||||
throw std::runtime_error("cudaGetDeviceProperties() failed for given device.");
|
||||
}
|
||||
result = cudaSetDevice(device_id);
|
||||
if (result != cudaSuccess) {
|
||||
throw std::runtime_error("cudaSetDevice() failed for given device.");
|
||||
}
|
||||
|
||||
// Get the clock rate (specified in cmd line in MHz)
|
||||
if (args.check_cmd_line_flag("clock")) {
|
||||
args.get_cmd_line_argument("clock", device_properties.clockRate);
|
||||
device_properties.clockRate *= 1000;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -501,15 +526,31 @@ struct TestbenchOptions {
|
||||
/// be saved to the file system.
|
||||
bool save_workspace(bool correct) const {
|
||||
if (save_workspace_mode == WorkspaceSaveMode::Always ||
|
||||
(save_workspace_mode == WorkspaceSaveMode::Incorrect && !correct)) {
|
||||
(save_workspace_mode == WorkspaceSaveMode::Incorrect && !correct)) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Returns true if the selected device can satisfy the given compute capability
|
||||
bool compute_capability(int major, int minor) const {
|
||||
return (device_properties.major > major ||
|
||||
(device_properties.major == major && device_properties.minor >= minor));
|
||||
}
|
||||
|
||||
/// Requires an exact match of compute capability
|
||||
bool compute_capability_exact(int major, int minor) const {
|
||||
return major == device_properties.major && minor == device_properties.minor;
|
||||
}
|
||||
|
||||
/// Prints version
|
||||
static void version(std::ostream &out) {
|
||||
out << "CUTLASS " << CUTLASS_MAJOR << "." << CUTLASS_MINOR << "." << CUTLASS_PATCH
|
||||
<< " built on " << __DATE__ << " at " << __TIME__;
|
||||
}
|
||||
|
||||
/// Prints the usage statement
|
||||
static void usage(std::ostream &out) {
|
||||
|
||||
out << "cutlass_perf_test [options]\n\n"
|
||||
|
||||
<< " --help\n"
|
||||
@@ -523,15 +564,27 @@ struct TestbenchOptions {
|
||||
<< " --beta=<beta> "
|
||||
<< " Value for beta to be used in GEMM experiments\n"
|
||||
|
||||
<< " --dist_{A,B,C}=<distribution> "
|
||||
<< " --device=<int> "
|
||||
<< " Specifies the CUDA device to use. Default is device 0.\n"
|
||||
|
||||
<< " --clock=<MHz> "
|
||||
<< " Specifies the SM clock rate in MHz.\n"
|
||||
|
||||
<< " --dist-{A,B,C}=<distribution> "
|
||||
<< " Describes the random distribution of each of the input matrix operands.\n"
|
||||
|
||||
<< " --execution_mode=<mode> "
|
||||
<< " --dry-run=<bool> "
|
||||
<< " If true, kernels are not executed and sleep is not inserted.\n"
|
||||
|
||||
<< " --execution-mode=<mode> "
|
||||
<< " Specifies execution mode: profile, verify, single\n"
|
||||
|
||||
<< " --output=<filename.csv> "
|
||||
<< " Writes summary of profiling to specified .csv file\n"
|
||||
|
||||
<< " --threshold=<filename.csv> "
|
||||
<< " Reads previous output summary and re-executes the same configurations.\n"
|
||||
|
||||
<< " --iterations=<timing iterations> "
|
||||
<< " maximum number of iterations to execute when profiling\n"
|
||||
|
||||
@@ -546,14 +599,19 @@ struct TestbenchOptions {
|
||||
<< " --k=<depth>[:max depth[:step]] "
|
||||
<< " Size of inner dimension of A and B. May specify a range with optional step size.\n"
|
||||
|
||||
<< " --kernels={s|d|h|i|wmma_}gemm_{nn,nt,tn,tt} "
|
||||
<< " --kernels=<{s|d|h|i|wmma_|wmma_binary_|wmma_integer_}gemm_{nn,nt,tn,tt}>\n"
|
||||
<< " "
|
||||
<< " Select GEMM datatype and layout to use for tests\n"
|
||||
|
||||
<< " --peak=<bool> "
|
||||
<< " If true, only reports peak performance per kernel after profiling specified "
|
||||
"problem space.\n"
|
||||
|
||||
<< " --save_workspace={*never,incorrect,always} "
|
||||
<< " --perf-margin=<perf-margin> "
|
||||
<< " Allowable performance degradation before flagging test as failure (e.g. 3% slowdown"
|
||||
" = 0.97).\n"
|
||||
|
||||
<< " --save-workspace={*never,incorrect,always} "
|
||||
<< " Specifies when to save the GEMM inputs and results to the filesystem.\n"
|
||||
|
||||
<< " --seed=<seed> "
|
||||
@@ -563,8 +621,17 @@ struct TestbenchOptions {
|
||||
<< " Inserts leading columns in output table and uniform values for each column. Useful "
|
||||
"for generating pivot tables.\n"
|
||||
|
||||
<< "\n\n"
|
||||
<< " --sleep-time=<second> "
|
||||
<< " Sleep period between profiling kernels to cool down the device.\n"
|
||||
|
||||
<< " --version "
|
||||
<< " ";
|
||||
|
||||
version(out);
|
||||
|
||||
out << "\n\n";
|
||||
|
||||
out << "\n\n"
|
||||
<< "Example usage:\n\n"
|
||||
|
||||
<< "# Runs one problem size for all kernels\n"
|
||||
|
||||
@@ -27,15 +27,16 @@
|
||||
|
||||
#include <fstream>
|
||||
|
||||
#include <tools/test/perf/performance_result.h>
|
||||
#include <tools/test/perf/testbench_options.h>
|
||||
#include <tools/util/command_line.h>
|
||||
#include "tools/test/perf/performance_result.h"
|
||||
#include "tools/test/perf/testbench_options.h"
|
||||
#include "tools/util/command_line.h"
|
||||
|
||||
namespace perf {
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// Wraps an output stream and constructs a comma-separated value table of results
|
||||
template <typename Problem>
|
||||
class TestbenchOutput {
|
||||
public:
|
||||
/// Options to test environment
|
||||
@@ -51,7 +52,7 @@ class TestbenchOutput {
|
||||
bool buffer_csv_output;
|
||||
|
||||
/// Vector holding performance results
|
||||
std::vector<PerformanceResult> buffered_perf_results;
|
||||
std::vector<PerformanceResult<Problem> > buffered_perf_results;
|
||||
|
||||
private:
|
||||
/// Opens the output file and updates output_ptr
|
||||
@@ -74,11 +75,11 @@ class TestbenchOutput {
|
||||
// pivot tags
|
||||
for (KeyValueIterator tag_it = options.pivot_tags.begin(); tag_it != options.pivot_tags.end();
|
||||
++tag_it) {
|
||||
ss << tag_it->first << ", ";
|
||||
ss << tag_it->first << ",";
|
||||
}
|
||||
|
||||
// performance result header
|
||||
ss << PerformanceResult::header();
|
||||
ss << PerformanceResult<Problem>::header();
|
||||
|
||||
return ss.str();
|
||||
}
|
||||
@@ -95,14 +96,23 @@ class TestbenchOutput {
|
||||
|
||||
/// Writes output to CSV
|
||||
~TestbenchOutput() {
|
||||
std::cout << std::endl;
|
||||
if (buffer_csv_output) {
|
||||
out() << "\n\n" << header() << std::endl;
|
||||
for (std::vector<PerformanceResult>::const_iterator it = buffered_perf_results.begin();
|
||||
it != buffered_perf_results.end();
|
||||
++it) {
|
||||
write_csv(*it);
|
||||
if (buffered_perf_results.size() != 0) {
|
||||
std::cout << std::endl;
|
||||
if (buffer_csv_output) {
|
||||
out() << "\n\n" << header() << std::endl;
|
||||
for (typename std::vector<PerformanceResult<Problem> >::const_iterator it =
|
||||
buffered_perf_results.begin();
|
||||
it != buffered_perf_results.end();
|
||||
++it) {
|
||||
write_csv(*it);
|
||||
}
|
||||
}
|
||||
std::cout << "\n[\033[1;32mPASSED\033[0m]";
|
||||
if (!options.threshold_filename.empty()) {
|
||||
std::cout << " - Performance Test Successful" << std::endl;
|
||||
} else {
|
||||
std::cout << std::endl;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -122,11 +132,11 @@ class TestbenchOutput {
|
||||
}
|
||||
|
||||
/// Writes a performance result to CSV output
|
||||
TestbenchOutput &write_csv(PerformanceResult const &result) {
|
||||
TestbenchOutput &write_csv(PerformanceResult<Problem> const &result) {
|
||||
// pivot tags
|
||||
for (KeyValueIterator tag_it = options.pivot_tags.begin(); tag_it != options.pivot_tags.end();
|
||||
++tag_it) {
|
||||
out() << tag_it->second << ", ";
|
||||
out() << tag_it->second << ",";
|
||||
}
|
||||
|
||||
out() << result << std::endl;
|
||||
@@ -134,24 +144,26 @@ class TestbenchOutput {
|
||||
}
|
||||
|
||||
/// Prints the output without appending it for CSV writing
|
||||
TestbenchOutput &pretty_print(PerformanceResult const &result) {
|
||||
TestbenchOutput &pretty_print(PerformanceResult<Problem> const &result) {
|
||||
result.pretty_print(std::cout) << std::endl;
|
||||
|
||||
return *this;
|
||||
}
|
||||
|
||||
/// Emits the result as output
|
||||
TestbenchOutput &append(PerformanceResult const &result) {
|
||||
TestbenchOutput &append(PerformanceResult<Problem> const &result) {
|
||||
if (buffer_csv_output) {
|
||||
buffered_perf_results.push_back(result);
|
||||
} else {
|
||||
write_csv(result);
|
||||
buffered_perf_results.push_back(result);
|
||||
}
|
||||
|
||||
pretty_print(result);
|
||||
|
||||
return *this;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
Reference in New Issue
Block a user