CUTLASS 3.8 Release (#2059)

* CUTLASS 3.8 Release

* update

* Update README.md

* Revert "Update README.md"

This reverts commit b353e36fe83e0815f99b44e46c0c95494c44726b.

* update

* update

---------

Co-authored-by: Haicheng Wu <57973641+hwu36@users.noreply.github.com>
Co-authored-by: Haicheng Wu <haichengw@nvidia.com>
This commit is contained in:
mihir-awatramani
2025-01-25 02:44:06 -05:00
committed by GitHub
co-authored by Haicheng Wu Haicheng Wu
parent 9eb01fa0b0
commit 389e493055
290 changed files with 91222 additions and 291 deletions
+1
View File
@@ -31,6 +31,7 @@ cutlass_test_unit_add_executable(
pipeline_tma_async.cu
pipeline_tma_async_warp_specialized.cu
pipeline_tma_async_warp_specialized_persistent.cu
pipeline_cluster_launch_control_async_warp_specialized_blackwell.cu
pipeline_async.cu
sequence_barrier.cu
)
@@ -0,0 +1,381 @@
/***************************************************************************************************
* Copyright (c) 2017 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: BSD-3-Clause
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are met:
*
* 1. Redistributions of source code must retain the above copyright notice, this
* list of conditions and the following disclaimer.
*
* 2. Redistributions in binary form must reproduce the above copyright notice,
* this list of conditions and the following disclaimer in the documentation
* and/or other materials provided with the distribution.
*
* 3. Neither the name of the copyright holder nor the names of its
* contributors may be used to endorse or promote products derived from
* this software without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*
**************************************************************************************************/
/*! \file
\brief Unit test for the PipelineCLCFetchAsync class
*/
//
//
#define KERNEL_DBG_TRACE false
#include <cuda/atomic>
#include "../common/cutlass_unit_test.h"
#include <thrust/host_vector.h>
#include <thrust/device_vector.h>
#include <cute/tensor.hpp>
#include <cute/arch/cluster_sm90.hpp>
#include <cutlass/util/reference/host/gemm.h>
#include <cutlass/cluster_launch.hpp>
#include "cutlass/core_io.h"
#include "cutlass/util/print_error.hpp"
#include "cutlass/util/GPU_Clock.hpp"
#include "testbed_cluster_launch_control.h"
#include "cutlass/pipeline/pipeline.hpp"
#include "cutlass/arch/barrier.h"
#include "cute/arch/cluster_sm90.hpp"
#include "cutlass/arch/barrier.h"
#include "cutlass/arch/reg_reconfig.h"
#include "cutlass/gemm/kernel/sm100_tile_scheduler.hpp"
using namespace cute;
using namespace cutlass;
using namespace cutlass::gemm::kernel::detail;
//////////////////// Shared Memory /////////////////////////
template <uint32_t Stages, typename ClusterShape>
struct SharedStorage
{
alignas(16) typename PersistentTileSchedulerSm100<ClusterShape, Stages>::CLCResponse clc_response[Stages];
alignas(8) typename PersistentTileSchedulerSm100<ClusterShape, Stages>::PipelineStorage storage ;
};
//////////////////// Kernel /////////////////////////
template <typename ClusterShape, uint32_t Stages>
__launch_bounds__(256, 1)
__global__ static
void pipeline_device(int *d_workerCount)
{
extern __shared__ char shared_memory[];
// single producer, multiple consumers
// producer: WG0
// consumer: WG1
using SharedStorage = SharedStorage<Stages, ClusterShape>;
using Scheduler = PersistentTileSchedulerSm100<ClusterShape, Stages>;
using TileSchedulingPipeline = typename Scheduler::Pipeline;
SharedStorage& shared_storage = *reinterpret_cast<SharedStorage*>(shared_memory);
// Logistics
int warp_idx = canonical_warp_idx();
auto cluster_shape = ClusterShape{};
typename TileSchedulingPipeline::Params params;
params.transaction_bytes = 16;
constexpr int NUM_PRODUCER = 32;
constexpr int NUM_CONSUMERS_PER_CTA = 32;
params.consumer_arv_count = NUM_PRODUCER + NUM_CONSUMERS_PER_CTA * cute::size<0>(cluster_shape) * cute::size<1>(cluster_shape);
params.producer_arv_count = 1;
// Only the first CTA in the Cluster is producing.
params.producer_blockid = 0;
dim3 block_id_in_cluster = cute::block_id_in_cluster();
// mbarrier.init
TileSchedulingPipeline scheduler_pipeline(shared_storage.storage, params );
Scheduler scheduler(&shared_storage.clc_response[0], typename Scheduler::Params{}, block_id_in_cluster);
// Ensure All CTAs in Cluster have completed init before issuing commits
cute::cluster_arrive_relaxed();
cute::cluster_wait();
uint32_t is_first_block_in_cluster = block_id_in_cluster.x == 0 && block_id_in_cluster.y == 0;
int lane_predicate = cute::elect_one_sync();
uint32_t is_producer = (is_first_block_in_cluster && warp_idx == 0);
uint32_t is_consumer = (warp_idx == 4);
PipelineState<Stages> scheduler_pipe_state;
PipelineState<Stages> scheduler_pipe_state_write = cutlass::make_producer_start_state<TileSchedulingPipeline>();
typename Scheduler::WorkTileInfo work_tile_info = {
static_cast<int32_t>(blockIdx.x),
static_cast<int32_t>(blockIdx.y),
static_cast<int32_t>(blockIdx.z),
false
};
// Persistent loop
do {
// Producer
if (is_producer) {
// Only 1 thread of the entire cluster issues the query.
scheduler_pipe_state_write = scheduler.advance_to_next_work(scheduler_pipeline, scheduler_pipe_state_write);
}
// Consumers
if (is_consumer) {
int linearCLC = work_tile_info.N_idx * gridDim.x + work_tile_info.M_idx;
// Atomically increment the worker count for the linearCLC by 1.
if (lane_predicate) {
atomicAdd(&d_workerCount[linearCLC], 1);
}
}
// Union of all consumers. Note that the producer here is its own consumer.
if (is_producer || is_consumer) {
scheduler_pipeline.consumer_wait(scheduler_pipe_state);
work_tile_info = scheduler.get_current_work(scheduler_pipe_state);
scheduler_pipeline.consumer_release(scheduler_pipe_state);
++scheduler_pipe_state;
// Add block offset since the scheduler works at cluster level.
dim3 block_id_in_cluster = cute::block_id_in_cluster();
work_tile_info.M_idx += block_id_in_cluster.x;
work_tile_info.N_idx += block_id_in_cluster.y;
work_tile_info.L_idx += block_id_in_cluster.z;
}
} while (work_tile_info.is_valid_tile);
// End of kernel
cute::cluster_sync();
}
/////////////////////////////////////////////////////
template<uint32_t Stages_, typename ClusterShape_>
struct PipelineTest {
//
// Data members
//
static constexpr uint32_t Stages = Stages_;
static constexpr uint32_t BlockSize = 128 * 2;
using ClusterShape = ClusterShape_;
//
// Methods
//
bool check_results(int *h_workerCount, int size ) {
for (int i = 0 ; i< size; i++ ){
if ( h_workerCount[i] != 1 )
{
std::cout << "linearCLC " << i << " has worker count " << h_workerCount[i] << "\n";
return false;
}
}
return true;
}
// Run CuTe GEMM kernel
cudaError_t run(bool &success, dim3 grid_dim,
cudaStream_t stream = 0 ) {
//
// Configure and launch
//
cudaError_t result;
int smem_size = 192 * 1024; // 192kB to force 1CTA/SM
auto cluster_shape = Shape<Int<ClusterShape::kM>, Int<ClusterShape::kN>, _1>{};
// Launch a single Cluster, with BlockSize threads per CTA
dim3 dimCluster(size<0>(cluster_shape), size<1>(cluster_shape), 1);
dim3 dimGrid = grid_dim;
dim3 dimBlock(BlockSize,1,1);
result = cudaFuncSetAttribute(
pipeline_device<
decltype(cluster_shape),
Stages>,
cudaFuncAttributeMaxDynamicSharedMemorySize,
smem_size
);
if (result != cudaSuccess) {
std::cerr << "Error: Failed to set Shared Memory size." << std::endl;
return result;
}
int array_size = dimGrid.x * dimGrid.y;
int *d_workerCount, *h_workerCount;
/* Allocate memory. workerCount[i] counts the number of worker(s) which work
on linear t i. The expectation is that workerCount[i] == 1 for all i.
*/
h_workerCount = (int*)malloc(array_size * sizeof(int));
result = cudaMalloc(&d_workerCount, array_size * sizeof(int));
if (result != cudaSuccess) {
std::cerr << "Failed to do cudaMalloc." << result << "\n";
return result;
}
for(int i = 0 ; i < array_size; i++)
{
h_workerCount[i] = 0; // Initialize workerCount[i] to 0 for all i.
}
result = cudaMemcpy(d_workerCount, h_workerCount, array_size * sizeof(int), cudaMemcpyHostToDevice);
if (result != cudaSuccess) {
std::cerr << "Failed to do cudaMemcpy." << result << "\n";
return result;
}
// Extended launch API
const void* kernel = (const void*)pipeline_device<decltype(cluster_shape), Stages>;
void* kernel_params[] = {&d_workerCount};
cutlass::ClusterLauncher::launch(dimGrid, dimCluster, dimBlock, smem_size, stream, kernel, kernel_params);
result = cudaDeviceSynchronize();
if (result != cudaSuccess) {
std::cerr << "Error: cudaDeviceSynchronize() failed" << std::endl;
return result;
}
result = cudaMemcpy(h_workerCount, d_workerCount, array_size * sizeof(int), cudaMemcpyDeviceToHost);
if (result != cudaSuccess) {
std::cerr << "Failed to do cudaMemcpy." << result << "\n";
return result;
}
success = check_results(h_workerCount, array_size);
free(h_workerCount);
result = cudaFree(d_workerCount);
if (result != cudaSuccess) {
std::cerr << "Failed to do cudaFree." << result << "\n";
return result;
}
return cudaSuccess;
}
};
#if defined(CUTLASS_ARCH_MMA_SM100_SUPPORTED)
//Cluster1x2 Stage4
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster1x2_Stage4) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
static constexpr uint32_t Stages = 4;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
//Cluster2x1 Stage4
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster2x1_Stage4) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
static constexpr uint32_t Stages = 4;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
//Cluster2x2 Stage4
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster2x2_Stage4) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
static constexpr uint32_t Stages = 4;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
//Cluster1x1 Stage3
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster1x1_Stage3) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
static constexpr uint32_t Stages = 3;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
//Cluster1x4 Stage4
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster1x4_Stage4) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
static constexpr uint32_t Stages = 4;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
//Cluster4x1 Stage4
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster4x1_Stage4) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
static constexpr uint32_t Stages = 4;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
//Cluster2x4 Stage4
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster2x4_Stage4) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
static constexpr uint32_t Stages = 4;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
//Cluster4x2 Stage4
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster4x2_Stage4) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
static constexpr uint32_t Stages = 4;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
//Cluster4x4 Stage4
TEST(SM100_Verify_PipelineClusterLaunchControlAsync_WS, Cluster4x4_Stage4) {
Options options;
options.grid_dim = {32,32,1};
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
static constexpr uint32_t Stages = 4;
using Test = PipelineTest<Stages, ClusterShape>;
Testbed<Test> testbed(options);
EXPECT_TRUE(testbed.verification());
}
#endif
@@ -0,0 +1,154 @@
/***************************************************************************************************
* Copyright (c) 2017 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
* SPDX-License-Identifier: BSD-3-Clause
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the following conditions are met:
*
* 1. Redistributions of source code must retain the above copyright notice, this
* list of conditions and the following disclaimer.
*
* 2. Redistributions in binary form must reproduce the above copyright notice,
* this list of conditions and the following disclaimer in the documentation
* and/or other materials provided with the distribution.
*
* 3. Neither the name of the copyright holder nor the names of its
* contributors may be used to endorse or promote products derived from
* this software without specific prior written permission.
*
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
*
**************************************************************************************************/
/*! \file
\brief Testbed file used by cluster launch control pipeline unit test
*/
//
//
#if CUDA_12_0_SM90_FEATURES_SUPPORTED
#define CUTLASS_UNIT_TEST_PIPELINE true
#else
#define CUTLASS_UNIT_TEST_PIPELINE false
#endif
#include <cstdlib>
#include <cstdio>
#include <cassert>
#include <cutlass/gemm/gemm.h>
#include "cutlass/util/command_line.h"
// Command line test options
struct Options {
//
// Data Members
//
bool help = false;
bool verification_enabled = true;
int SM_count = 116;
int clock_MHz = 1477;
dim3 grid_dim = {0,0,0};
//
// Methods
//
void parse(int argc, char const **args) {
cutlass::CommandLine cmd(argc, args);
if (cmd.check_cmd_line_flag("help")) {
help = true;
}
cmd.get_cmd_line_argument("verification-enabled", verification_enabled, verification_enabled);
cmd.get_cmd_line_argument("sm-count", SM_count, SM_count);
cmd.get_cmd_line_argument("clock", clock_MHz, clock_MHz);
}
/// Prints the usage statement.
std::ostream & print_usage(std::ostream &out) const {
out << "Options:\n\n"
<< " --help If specified, displays this usage statement.\n\n"
<< " --verification-enabled=<bool> Enable/Disable verification\n"
<< " --sm-count=<int> Number of SMs on the chip\n"
<< " --clock=<int> Locked clock value in Mhz\n";
return out;
}
};
//
// Testbed
//
template<typename Pipeline>
class Testbed {
private:
// Commandline options
Options options;
bool run_test() {
// Run CuTe Gemm
Pipeline pipeline;
bool success = false;
cudaError_t result = pipeline.run(success, this->options.grid_dim);
CUTE_CHECK_LAST();
return success;
}
public:
Testbed(Options const &options_) : options(options_) {
int device_id = 0;
cudaDeviceProp device_prop;
CUTE_CHECK_ERROR(cudaSetDevice(device_id));
CUTE_CHECK_ERROR(cudaGetDeviceProperties(&device_prop, device_id));
if (device_prop.major < 1) {
fprintf(stderr, "Device does not support CUDA.\n");
exit(1);
}
}
/// Run verification Gemm problem sizes
bool verification() {
#if !defined(CUTLASS_ARCH_MMA_SM100_SUPPORTED)
printf(
"CUTLASS_ARCH_MMA_SM100_SUPPORTED must be set, but it is not. \n"
"This test is waived.\n"
);
return true;
#endif
#if 1
bool is_success = false;
for (int i = 0; i< 10; i++){
printf("iteration = %d\n", i);
is_success = run_test();
if ( not is_success )
return is_success;
}
return is_success;
#else
// Run the test with single launch
return run_test();
#endif
}
};