CUTLASS 3.8 Release (#2059)
* CUTLASS 3.8 Release * update * Update README.md * Revert "Update README.md" This reverts commit b353e36fe83e0815f99b44e46c0c95494c44726b. * update * update --------- Co-authored-by: Haicheng Wu <57973641+hwu36@users.noreply.github.com> Co-authored-by: Haicheng Wu <haichengw@nvidia.com>
This commit is contained in:
co-authored by
Haicheng Wu
Haicheng Wu
parent
9eb01fa0b0
commit
389e493055
@@ -31,6 +31,7 @@ cutlass_test_unit_add_executable(
|
||||
pipeline_tma_async.cu
|
||||
pipeline_tma_async_warp_specialized.cu
|
||||
pipeline_tma_async_warp_specialized_persistent.cu
|
||||
pipeline_cluster_launch_control_async_warp_specialized_blackwell.cu
|
||||
pipeline_async.cu
|
||||
sequence_barrier.cu
|
||||
)
|
||||
|
||||
+381
@@ -0,0 +1,381 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Unit test for the PipelineCLCFetchAsync class
|
||||
*/
|
||||
|
||||
//
|
||||
|
||||
//
|
||||
|
||||
#define KERNEL_DBG_TRACE false
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include "../common/cutlass_unit_test.h"
|
||||
#include <thrust/host_vector.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cute/tensor.hpp>
|
||||
#include <cute/arch/cluster_sm90.hpp>
|
||||
|
||||
#include <cutlass/util/reference/host/gemm.h>
|
||||
#include <cutlass/cluster_launch.hpp>
|
||||
|
||||
#include "cutlass/core_io.h"
|
||||
#include "cutlass/util/print_error.hpp"
|
||||
#include "cutlass/util/GPU_Clock.hpp"
|
||||
|
||||
#include "testbed_cluster_launch_control.h"
|
||||
#include "cutlass/pipeline/pipeline.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cute/arch/cluster_sm90.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cutlass/arch/reg_reconfig.h"
|
||||
#include "cutlass/gemm/kernel/sm100_tile_scheduler.hpp"
|
||||
|
||||
|
||||
using namespace cute;
|
||||
using namespace cutlass;
|
||||
using namespace cutlass::gemm::kernel::detail;
|
||||
|
||||
//////////////////// Shared Memory /////////////////////////
|
||||
|
||||
template <uint32_t Stages, typename ClusterShape>
|
||||
struct SharedStorage
|
||||
{
|
||||
alignas(16) typename PersistentTileSchedulerSm100<ClusterShape, Stages>::CLCResponse clc_response[Stages];
|
||||
alignas(8) typename PersistentTileSchedulerSm100<ClusterShape, Stages>::PipelineStorage storage ;
|
||||
};
|
||||
|
||||
//////////////////// Kernel /////////////////////////
|
||||
template <typename ClusterShape, uint32_t Stages>
|
||||
__launch_bounds__(256, 1)
|
||||
__global__ static
|
||||
void pipeline_device(int *d_workerCount)
|
||||
{
|
||||
extern __shared__ char shared_memory[];
|
||||
|
||||
// single producer, multiple consumers
|
||||
// producer: WG0
|
||||
// consumer: WG1
|
||||
|
||||
using SharedStorage = SharedStorage<Stages, ClusterShape>;
|
||||
using Scheduler = PersistentTileSchedulerSm100<ClusterShape, Stages>;
|
||||
using TileSchedulingPipeline = typename Scheduler::Pipeline;
|
||||
SharedStorage& shared_storage = *reinterpret_cast<SharedStorage*>(shared_memory);
|
||||
|
||||
// Logistics
|
||||
int warp_idx = canonical_warp_idx();
|
||||
auto cluster_shape = ClusterShape{};
|
||||
|
||||
typename TileSchedulingPipeline::Params params;
|
||||
params.transaction_bytes = 16;
|
||||
|
||||
constexpr int NUM_PRODUCER = 32;
|
||||
constexpr int NUM_CONSUMERS_PER_CTA = 32;
|
||||
params.consumer_arv_count = NUM_PRODUCER + NUM_CONSUMERS_PER_CTA * cute::size<0>(cluster_shape) * cute::size<1>(cluster_shape);
|
||||
params.producer_arv_count = 1;
|
||||
// Only the first CTA in the Cluster is producing.
|
||||
params.producer_blockid = 0;
|
||||
|
||||
dim3 block_id_in_cluster = cute::block_id_in_cluster();
|
||||
// mbarrier.init
|
||||
TileSchedulingPipeline scheduler_pipeline(shared_storage.storage, params );
|
||||
Scheduler scheduler(&shared_storage.clc_response[0], typename Scheduler::Params{}, block_id_in_cluster);
|
||||
|
||||
// Ensure All CTAs in Cluster have completed init before issuing commits
|
||||
cute::cluster_arrive_relaxed();
|
||||
cute::cluster_wait();
|
||||
|
||||
uint32_t is_first_block_in_cluster = block_id_in_cluster.x == 0 && block_id_in_cluster.y == 0;
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
|
||||
uint32_t is_producer = (is_first_block_in_cluster && warp_idx == 0);
|
||||
uint32_t is_consumer = (warp_idx == 4);
|
||||
|
||||
PipelineState<Stages> scheduler_pipe_state;
|
||||
PipelineState<Stages> scheduler_pipe_state_write = cutlass::make_producer_start_state<TileSchedulingPipeline>();
|
||||
typename Scheduler::WorkTileInfo work_tile_info = {
|
||||
static_cast<int32_t>(blockIdx.x),
|
||||
static_cast<int32_t>(blockIdx.y),
|
||||
static_cast<int32_t>(blockIdx.z),
|
||||
false
|
||||
};
|
||||
|
||||
// Persistent loop
|
||||
do {
|
||||
// Producer
|
||||
if (is_producer) {
|
||||
// Only 1 thread of the entire cluster issues the query.
|
||||
scheduler_pipe_state_write = scheduler.advance_to_next_work(scheduler_pipeline, scheduler_pipe_state_write);
|
||||
}
|
||||
|
||||
// Consumers
|
||||
if (is_consumer) {
|
||||
int linearCLC = work_tile_info.N_idx * gridDim.x + work_tile_info.M_idx;
|
||||
// Atomically increment the worker count for the linearCLC by 1.
|
||||
if (lane_predicate) {
|
||||
atomicAdd(&d_workerCount[linearCLC], 1);
|
||||
}
|
||||
}
|
||||
|
||||
// Union of all consumers. Note that the producer here is its own consumer.
|
||||
if (is_producer || is_consumer) {
|
||||
scheduler_pipeline.consumer_wait(scheduler_pipe_state);
|
||||
work_tile_info = scheduler.get_current_work(scheduler_pipe_state);
|
||||
scheduler_pipeline.consumer_release(scheduler_pipe_state);
|
||||
++scheduler_pipe_state;
|
||||
|
||||
// Add block offset since the scheduler works at cluster level.
|
||||
dim3 block_id_in_cluster = cute::block_id_in_cluster();
|
||||
work_tile_info.M_idx += block_id_in_cluster.x;
|
||||
work_tile_info.N_idx += block_id_in_cluster.y;
|
||||
work_tile_info.L_idx += block_id_in_cluster.z;
|
||||
|
||||
}
|
||||
} while (work_tile_info.is_valid_tile);
|
||||
|
||||
// End of kernel
|
||||
cute::cluster_sync();
|
||||
}
|
||||
/////////////////////////////////////////////////////
|
||||
|
||||
template<uint32_t Stages_, typename ClusterShape_>
|
||||
struct PipelineTest {
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
static constexpr uint32_t Stages = Stages_;
|
||||
static constexpr uint32_t BlockSize = 128 * 2;
|
||||
using ClusterShape = ClusterShape_;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
bool check_results(int *h_workerCount, int size ) {
|
||||
for (int i = 0 ; i< size; i++ ){
|
||||
if ( h_workerCount[i] != 1 )
|
||||
{
|
||||
std::cout << "linearCLC " << i << " has worker count " << h_workerCount[i] << "\n";
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// Run CuTe GEMM kernel
|
||||
cudaError_t run(bool &success, dim3 grid_dim,
|
||||
cudaStream_t stream = 0 ) {
|
||||
|
||||
//
|
||||
// Configure and launch
|
||||
//
|
||||
cudaError_t result;
|
||||
|
||||
int smem_size = 192 * 1024; // 192kB to force 1CTA/SM
|
||||
auto cluster_shape = Shape<Int<ClusterShape::kM>, Int<ClusterShape::kN>, _1>{};
|
||||
// Launch a single Cluster, with BlockSize threads per CTA
|
||||
dim3 dimCluster(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimGrid = grid_dim;
|
||||
dim3 dimBlock(BlockSize,1,1);
|
||||
|
||||
result = cudaFuncSetAttribute(
|
||||
pipeline_device<
|
||||
decltype(cluster_shape),
|
||||
Stages>,
|
||||
cudaFuncAttributeMaxDynamicSharedMemorySize,
|
||||
smem_size
|
||||
);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to set Shared Memory size." << std::endl;
|
||||
return result;
|
||||
}
|
||||
|
||||
int array_size = dimGrid.x * dimGrid.y;
|
||||
int *d_workerCount, *h_workerCount;
|
||||
|
||||
/* Allocate memory. workerCount[i] counts the number of worker(s) which work
|
||||
on linear t i. The expectation is that workerCount[i] == 1 for all i.
|
||||
*/
|
||||
h_workerCount = (int*)malloc(array_size * sizeof(int));
|
||||
|
||||
result = cudaMalloc(&d_workerCount, array_size * sizeof(int));
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Failed to do cudaMalloc." << result << "\n";
|
||||
return result;
|
||||
}
|
||||
|
||||
for(int i = 0 ; i < array_size; i++)
|
||||
{
|
||||
h_workerCount[i] = 0; // Initialize workerCount[i] to 0 for all i.
|
||||
}
|
||||
|
||||
result = cudaMemcpy(d_workerCount, h_workerCount, array_size * sizeof(int), cudaMemcpyHostToDevice);
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Failed to do cudaMemcpy." << result << "\n";
|
||||
return result;
|
||||
}
|
||||
|
||||
// Extended launch API
|
||||
const void* kernel = (const void*)pipeline_device<decltype(cluster_shape), Stages>;
|
||||
void* kernel_params[] = {&d_workerCount};
|
||||
cutlass::ClusterLauncher::launch(dimGrid, dimCluster, dimBlock, smem_size, stream, kernel, kernel_params);
|
||||
|
||||
result = cudaDeviceSynchronize();
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: cudaDeviceSynchronize() failed" << std::endl;
|
||||
return result;
|
||||
}
|
||||
|
||||
result = cudaMemcpy(h_workerCount, d_workerCount, array_size * sizeof(int), cudaMemcpyDeviceToHost);
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Failed to do cudaMemcpy." << result << "\n";
|
||||
return result;
|
||||
}
|
||||
|
||||
success = check_results(h_workerCount, array_size);
|
||||
|
||||
free(h_workerCount);
|
||||
|
||||
result = cudaFree(d_workerCount);
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Failed to do cudaFree." << result << "\n";
|
||||
return result;
|
||||
}
|
||||
|
||||
return cudaSuccess;
|
||||
}
|
||||
};
|
||||
|
||||
#if defined(CUTLASS_ARCH_MMA_SM100_SUPPORTED)
|
||||
//Cluster1x2 Stage4
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster1x2_Stage4) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
//Cluster2x1 Stage4
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster2x1_Stage4) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
//Cluster2x2 Stage4
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster2x2_Stage4) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
//Cluster1x1 Stage3
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster1x1_Stage3) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 3;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
//Cluster1x4 Stage4
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster1x4_Stage4) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
//Cluster4x1 Stage4
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster4x1_Stage4) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
//Cluster2x4 Stage4
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster2x4_Stage4) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
//Cluster4x2 Stage4
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster4x2_Stage4) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
//Cluster4x4 Stage4
|
||||
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster4x4_Stage4) {
|
||||
Options options;
|
||||
options.grid_dim = {32,32,1};
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,154 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Testbed file used by cluster launch control pipeline unit test
|
||||
*/
|
||||
|
||||
//
|
||||
|
||||
//
|
||||
|
||||
#if CUDA_12_0_SM90_FEATURES_SUPPORTED
|
||||
#define CUTLASS_UNIT_TEST_PIPELINE true
|
||||
#else
|
||||
#define CUTLASS_UNIT_TEST_PIPELINE false
|
||||
#endif
|
||||
|
||||
#include <cstdlib>
|
||||
#include <cstdio>
|
||||
#include <cassert>
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
|
||||
#include "cutlass/util/command_line.h"
|
||||
|
||||
// Command line test options
|
||||
struct Options {
|
||||
//
|
||||
// Data Members
|
||||
//
|
||||
bool help = false;
|
||||
bool verification_enabled = true;
|
||||
int SM_count = 116;
|
||||
int clock_MHz = 1477;
|
||||
dim3 grid_dim = {0,0,0};
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
void parse(int argc, char const **args) {
|
||||
cutlass::CommandLine cmd(argc, args);
|
||||
|
||||
if (cmd.check_cmd_line_flag("help")) {
|
||||
help = true;
|
||||
}
|
||||
|
||||
cmd.get_cmd_line_argument("verification-enabled", verification_enabled, verification_enabled);
|
||||
cmd.get_cmd_line_argument("sm-count", SM_count, SM_count);
|
||||
cmd.get_cmd_line_argument("clock", clock_MHz, clock_MHz);
|
||||
}
|
||||
|
||||
/// Prints the usage statement.
|
||||
std::ostream & print_usage(std::ostream &out) const {
|
||||
|
||||
out << "Options:\n\n"
|
||||
<< " --help If specified, displays this usage statement.\n\n"
|
||||
<< " --verification-enabled=<bool> Enable/Disable verification\n"
|
||||
<< " --sm-count=<int> Number of SMs on the chip\n"
|
||||
<< " --clock=<int> Locked clock value in Mhz\n";
|
||||
|
||||
return out;
|
||||
}
|
||||
};
|
||||
|
||||
//
|
||||
// Testbed
|
||||
//
|
||||
|
||||
template<typename Pipeline>
|
||||
class Testbed {
|
||||
private:
|
||||
// Commandline options
|
||||
Options options;
|
||||
|
||||
bool run_test() {
|
||||
|
||||
// Run CuTe Gemm
|
||||
Pipeline pipeline;
|
||||
|
||||
bool success = false;
|
||||
cudaError_t result = pipeline.run(success, this->options.grid_dim);
|
||||
|
||||
CUTE_CHECK_LAST();
|
||||
return success;
|
||||
}
|
||||
|
||||
|
||||
public:
|
||||
Testbed(Options const &options_) : options(options_) {
|
||||
int device_id = 0;
|
||||
cudaDeviceProp device_prop;
|
||||
CUTE_CHECK_ERROR(cudaSetDevice(device_id));
|
||||
CUTE_CHECK_ERROR(cudaGetDeviceProperties(&device_prop, device_id));
|
||||
|
||||
if (device_prop.major < 1) {
|
||||
fprintf(stderr, "Device does not support CUDA.\n");
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
/// Run verification Gemm problem sizes
|
||||
bool verification() {
|
||||
|
||||
#if !defined(CUTLASS_ARCH_MMA_SM100_SUPPORTED)
|
||||
printf(
|
||||
"CUTLASS_ARCH_MMA_SM100_SUPPORTED must be set, but it is not. \n"
|
||||
"This test is waived.\n"
|
||||
);
|
||||
return true;
|
||||
#endif
|
||||
|
||||
#if 1
|
||||
bool is_success = false;
|
||||
for (int i = 0; i< 10; i++){
|
||||
printf("iteration = %d\n", i);
|
||||
is_success = run_test();
|
||||
if ( not is_success )
|
||||
return is_success;
|
||||
}
|
||||
return is_success;
|
||||
#else
|
||||
// Run the test with single launch
|
||||
return run_test();
|
||||
#endif
|
||||
}
|
||||
};
|
||||
Reference in New Issue
Block a user