@@ -0,0 +1,36 @@
|
||||
# Copyright (c) 2017 - 2023 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: BSD-3-Clause
|
||||
#
|
||||
# Redistribution and use in source and binary forms, with or without
|
||||
# modification, are permitted provided that the following conditions are met:
|
||||
#
|
||||
# 1. Redistributions of source code must retain the above copyright notice, this
|
||||
# list of conditions and the following disclaimer.
|
||||
#
|
||||
# 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
# this list of conditions and the following disclaimer in the documentation
|
||||
# and/or other materials provided with the distribution.
|
||||
#
|
||||
# 3. Neither the name of the copyright holder nor the names of its
|
||||
# contributors may be used to endorse or promote products derived from
|
||||
# this software without specific prior written permission.
|
||||
#
|
||||
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
# AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
# DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
# FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
# DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
# SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
# CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
# OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
|
||||
cutlass_test_unit_add_executable(
|
||||
cutlass_test_unit_pipeline
|
||||
pipeline_tma_async.cu
|
||||
pipeline_tma_async_warp_specialized.cu
|
||||
pipeline_tma_async_warp_specialized_persistent.cu
|
||||
pipeline_async.cu
|
||||
sequence_barrier.cu
|
||||
)
|
||||
@@ -0,0 +1,468 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2023 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Unit test for the PipelineAsync class
|
||||
*/
|
||||
|
||||
#define KERNEL_DBG_TRACE false
|
||||
|
||||
#include "../common/cutlass_unit_test.h"
|
||||
#include <thrust/host_vector.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cute/tensor.hpp>
|
||||
#include <cute/arch/cluster_sm90.hpp>
|
||||
|
||||
#include <cutlass/util/reference/host/gemm.h>
|
||||
#include <cutlass/cluster_launch.hpp>
|
||||
|
||||
#include "cutlass/core_io.h"
|
||||
|
||||
#include "cutlass/util/print_error.hpp"
|
||||
#include "cutlass/util/GPU_Clock.hpp"
|
||||
|
||||
#include "testbed.h"
|
||||
#include "cutlass/pipeline.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cute/arch/cluster_sm90.hpp"
|
||||
|
||||
using namespace cute;
|
||||
|
||||
//////////////////// KERNEL /////////////////////////
|
||||
|
||||
template <uint32_t Stages>
|
||||
struct SharedStorage
|
||||
{
|
||||
typename cutlass::PipelineAsync<Stages>::SharedStorage storage;
|
||||
};
|
||||
|
||||
// Goal of this kernel is to complete deadlock-free
|
||||
// Simple 1 producer warp, one consumer warp scenario
|
||||
template <class ClusterShape, uint32_t NumStages>
|
||||
__global__ static
|
||||
void pipeline_async_basic_device(uint32_t const num_iterations)
|
||||
{
|
||||
|
||||
extern __shared__ char shared_memory[];
|
||||
using MainloopPipeline = typename cutlass::PipelineAsync<NumStages>;
|
||||
using PipelineState = typename cutlass::PipelineState<NumStages>;
|
||||
|
||||
using SharedStorage = SharedStorage<NumStages>;
|
||||
SharedStorage& shared_storage = *reinterpret_cast<SharedStorage*>(shared_memory);
|
||||
|
||||
|
||||
auto cta_layout = Layout<ClusterShape>{}; // (m,n) -> cta_id
|
||||
|
||||
int warp_idx = __shfl_sync(0xffffffff, threadIdx.x / 32, 0);
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
dim3 block_id_in_cluster = cute::block_id_in_cluster();
|
||||
auto cluster_shape = ClusterShape{};
|
||||
|
||||
// This example showcases 2 producer 1 consumer example
|
||||
typename MainloopPipeline::Params params;
|
||||
params.producer_arv_count = 2;
|
||||
params.consumer_arv_count = 1;
|
||||
MainloopPipeline pipeline(shared_storage.storage, params);
|
||||
|
||||
// Ensure All CTAs in Cluster have completed init before issuing commits
|
||||
cute::cluster_arrive_relaxed();
|
||||
cute::cluster_wait();
|
||||
__syncthreads();
|
||||
|
||||
if (lane_predicate) {
|
||||
// Producer Warps
|
||||
if (warp_idx==0 || warp_idx==1) {
|
||||
|
||||
int prologue_iterations = min(NumStages, num_iterations);
|
||||
for ( int i = 0; i < prologue_iterations; ++i) {
|
||||
// Can also specify stage to commit directly
|
||||
pipeline.producer_commit(i);
|
||||
}
|
||||
|
||||
int mainloop_iterations = num_iterations - prologue_iterations;
|
||||
|
||||
// Only the mainloop needs a PipelineState because this is where we start "waiting" (acquiring)
|
||||
PipelineState smem_pipe_write;
|
||||
|
||||
for ( ; mainloop_iterations > 0; --mainloop_iterations) {
|
||||
pipeline.producer_acquire(smem_pipe_write);
|
||||
pipeline.producer_commit(smem_pipe_write);
|
||||
++smem_pipe_write;
|
||||
}
|
||||
}
|
||||
else {
|
||||
PipelineState smem_pipe_read;
|
||||
for (int iter=0 ; iter < num_iterations; ++iter) {
|
||||
pipeline.consumer_wait(smem_pipe_read);
|
||||
pipeline.consumer_release(smem_pipe_read.index());
|
||||
++smem_pipe_read;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// To make sure remote SMEM doesn't get destroyed
|
||||
cute::cluster_arrive();
|
||||
cute::cluster_wait();
|
||||
}
|
||||
/////////////////////////////////////////////////////
|
||||
|
||||
template<uint32_t Stages_, typename ClusterShape_>
|
||||
struct PipelineTest {
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
static constexpr uint32_t Stages = Stages_;
|
||||
static constexpr uint32_t kBlockSize = 96;
|
||||
using ClusterShape = ClusterShape_;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
// Ctor
|
||||
PipelineTest() = default;
|
||||
|
||||
|
||||
// Run CuTe GEMM kernel
|
||||
cudaError_t run(uint32_t const kNumIters,
|
||||
cudaStream_t stream = nullptr) {
|
||||
|
||||
// Pipeline (multistage pipeline)
|
||||
auto num_stages = Int<Stages>{};
|
||||
|
||||
auto cluster_shape = Shape<Int<ClusterShape::kM>, Int<ClusterShape::kN>, _1>{};
|
||||
|
||||
//
|
||||
// Configure and launch
|
||||
//
|
||||
int iterations = 2;
|
||||
cudaError_t result;
|
||||
|
||||
for (int iter = 0; iter < iterations; ++iter) {
|
||||
|
||||
// Define the tiled MMA layout (static, 4warps)
|
||||
using MainloopPipeline = typename cutlass::PipelineAsync<Stages>;
|
||||
|
||||
int smem_size = int(sizeof(SharedStorage<Stages>));
|
||||
|
||||
result = cudaFuncSetAttribute(
|
||||
pipeline_async_basic_device<decltype(cluster_shape), Stages>,
|
||||
cudaFuncAttributeMaxDynamicSharedMemorySize,
|
||||
smem_size);
|
||||
|
||||
// Launch a single Cluster, with 128 thread per CTA
|
||||
dim3 dimCluster(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimGrid(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimBlock(kBlockSize,1,1);
|
||||
|
||||
const void* kernel = (const void*)pipeline_async_basic_device<decltype(cluster_shape), Stages>;
|
||||
int iters = kNumIters;
|
||||
void* kernel_params[] = {reinterpret_cast<void*>(&iters)};
|
||||
cutlass::ClusterLauncher::launch(dimGrid, dimCluster, dimBlock, smem_size, stream, kernel, kernel_params);
|
||||
|
||||
} // profiling loop ends
|
||||
|
||||
result = cudaDeviceSynchronize();
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: cudaDeviceSynchronize() failed" << std::endl;
|
||||
return result;
|
||||
}
|
||||
|
||||
return cudaSuccess;
|
||||
}
|
||||
|
||||
};
|
||||
|
||||
#if CUDA_12_0_SM90_FEATURES_SUPPORTED
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster1x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster1x1_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster1x1_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster2x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster2x2_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster2x2_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster1x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster1x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster1x2_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster2x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster2x1_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x1_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster1x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster1x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster2x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster2x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage3) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 3;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage4) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 4;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage6) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 6;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage8) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 8;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage9) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 9;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineAsync, Cluster4x4_Stage11) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 11;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,469 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2023 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Unit test for the PipelineTmaAsync class
|
||||
*/
|
||||
|
||||
|
||||
#define KERNEL_DBG_TRACE false
|
||||
|
||||
#include "../common/cutlass_unit_test.h"
|
||||
#include <thrust/host_vector.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cute/tensor.hpp>
|
||||
#include <cute/arch/cluster_sm90.hpp>
|
||||
|
||||
#include <cutlass/util/reference/host/gemm.h>
|
||||
#include <cutlass/cluster_launch.hpp>
|
||||
|
||||
#include "cutlass/core_io.h"
|
||||
|
||||
#include "cutlass/util/print_error.hpp"
|
||||
#include "cutlass/util/GPU_Clock.hpp"
|
||||
|
||||
#include "testbed.h"
|
||||
#include "cutlass/pipeline.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cute/arch/cluster_sm90.hpp"
|
||||
|
||||
using namespace cute;
|
||||
|
||||
//////////////////// KERNEL /////////////////////////
|
||||
|
||||
template <uint32_t Stages, typename ClusterShape>
|
||||
struct SharedStorage
|
||||
{
|
||||
typename cutlass::PipelineTmaAsync<Stages, ClusterShape>::SharedStorage storage;
|
||||
};
|
||||
|
||||
// Goal of this kernel is to complete deadlock-free
|
||||
template <class ClusterShape, uint32_t NumStages>
|
||||
__global__ static
|
||||
void pipeline_device(uint32_t const NumIterations)
|
||||
{
|
||||
|
||||
extern __shared__ char shared_memory[];
|
||||
using DispatchPolicy = cutlass::gemm::MainloopSm90TmaGmma<NumStages, ClusterShape>;
|
||||
using MainloopPipeline = cutlass::PipelineTmaAsync<NumStages, ClusterShape>;
|
||||
using PipelineState = cutlass::PipelineState<NumStages>;
|
||||
|
||||
using SharedStorage = SharedStorage<NumStages, ClusterShape>;
|
||||
SharedStorage& shared_storage = *reinterpret_cast<SharedStorage*>(shared_memory);
|
||||
|
||||
auto cta_layout = Layout<ClusterShape>{}; // (m,n) -> cta_id
|
||||
int warp_idx = __shfl_sync(0xffffffff, threadIdx.x / 32, 0);
|
||||
int warp_group_thread_idx = threadIdx.x % 128;
|
||||
dim3 block_id_in_cluster = cute::block_id_in_cluster();
|
||||
|
||||
auto cluster_shape = ClusterShape{};
|
||||
|
||||
// #Producers = #RowsInCluster + #ColsInCluster - 1
|
||||
uint32_t const NumProducers = cute::size<0>(cluster_shape) + cute::size<1>(cluster_shape) - 1;
|
||||
uint32_t const TmaTransactionBytes = sizeof(uint32_t) * NumProducers;
|
||||
uint32_t const per_cta_bytes = sizeof(uint32_t);
|
||||
|
||||
// mbarrier.init
|
||||
typename MainloopPipeline::Params params;
|
||||
params.transaction_bytes = TmaTransactionBytes;
|
||||
params.role = MainloopPipeline::ThreadCategory::ProducerConsumer;
|
||||
params.is_leader = warp_group_thread_idx == 0;
|
||||
params.num_consumers = 128;
|
||||
|
||||
MainloopPipeline pipeline(shared_storage.storage, params);
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Ensure All CTAs in Cluster have completed init before issuing commits
|
||||
cute::cluster_arrive_relaxed();
|
||||
cute::cluster_wait();
|
||||
|
||||
// Total number of gemm_k_iterations
|
||||
auto mma_k_iterations = NumIterations;
|
||||
auto tma_k_iterations = NumIterations;
|
||||
|
||||
PipelineState smem_pipe_read;
|
||||
// For the DMA (prologue) - we start with an opposite phase - since we skip all waits
|
||||
// i.e., we know that the buffer is indeed empty
|
||||
PipelineState smem_pipe_write = cutlass::make_producer_start_state<MainloopPipeline>();
|
||||
PipelineState smem_pipe_release;
|
||||
int K_TILE_MMAS = 1;
|
||||
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
int k_pipe_tma_prologue = min(NumStages, tma_k_iterations);
|
||||
|
||||
// DMA Prologue (Loads)
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for(int i = 0; i < k_pipe_tma_prologue; ++i) {
|
||||
pipeline.producer_acquire(smem_pipe_write);
|
||||
// cp.async.bulk.tensor would typically happen here
|
||||
pipeline.producer_commit(smem_pipe_write.index(), per_cta_bytes);
|
||||
++smem_pipe_write;
|
||||
}
|
||||
tma_k_iterations -= k_pipe_tma_prologue;
|
||||
|
||||
// MMA Prologue (Compute) - modeling inflight MMAs
|
||||
for (int iter = 0; iter < K_TILE_MMAS; ++iter)
|
||||
{
|
||||
pipeline.consumer_wait(smem_pipe_read);
|
||||
warpgroup_arrive();
|
||||
// GMMA would typically happen here
|
||||
|
||||
++smem_pipe_read;
|
||||
}
|
||||
|
||||
mma_k_iterations -= K_TILE_MMAS;
|
||||
|
||||
CUTLASS_PRAGMA_NO_UNROLL
|
||||
for (int iter = 0; iter < mma_k_iterations; ++iter)
|
||||
{
|
||||
pipeline.consumer_wait(smem_pipe_read);
|
||||
|
||||
warpgroup_arrive();
|
||||
// GMMA would typically happen here
|
||||
|
||||
pipeline.consumer_release(smem_pipe_release);
|
||||
|
||||
if (lane_predicate && (warp_idx == 0) && (tma_k_iterations > 0)) {
|
||||
pipeline.producer_acquire(smem_pipe_write);
|
||||
// cp.async.bulk.tensor would typically happen here
|
||||
pipeline.producer_commit(smem_pipe_write.index(), per_cta_bytes);
|
||||
++smem_pipe_write;
|
||||
--tma_k_iterations;
|
||||
}
|
||||
|
||||
// next read stage
|
||||
++smem_pipe_read;
|
||||
++smem_pipe_release;
|
||||
}
|
||||
|
||||
// To make sure remote SMEM doesn't get destoryed
|
||||
cute::cluster_arrive();
|
||||
cute::cluster_wait();
|
||||
}
|
||||
/////////////////////////////////////////////////////
|
||||
|
||||
/// Device NT GMMA + TMA specialized
|
||||
template<uint32_t Stages_, typename ClusterShape_>
|
||||
struct PipelineTest {
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
static constexpr uint32_t Stages = Stages_;
|
||||
static constexpr uint32_t kBlockSize = 128;
|
||||
using ClusterShape = ClusterShape_;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
// Ctor
|
||||
PipelineTest(){};
|
||||
|
||||
|
||||
// Run CuTe GEMM kernel
|
||||
cudaError_t run(uint32_t const kNumIters,
|
||||
cudaStream_t stream = 0) {
|
||||
|
||||
float elapsed_ms = 0.0f;
|
||||
// Pipeline (multistage pipeline)
|
||||
auto num_stages = Int<Stages>{};
|
||||
|
||||
auto cluster_shape = Shape<Int<ClusterShape::kM>, Int<ClusterShape::kN>, _1>{};
|
||||
|
||||
//
|
||||
// Configure and launch
|
||||
//
|
||||
int iterations = 1;
|
||||
cudaEvent_t events[2];
|
||||
cudaError_t result;
|
||||
|
||||
for (cudaEvent_t & event : events) {
|
||||
result = cudaEventCreate(&event);
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to create event.";
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
result = cudaEventRecord(events[0]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to record start event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
for (int iter = 0; iter < iterations; ++iter) {
|
||||
|
||||
// Define the tiled MMA layout (static, 4warps)
|
||||
using DispatchPolicy = cutlass::gemm::MainloopSm90TmaGmma<Stages, decltype(cluster_shape)>;
|
||||
using MainloopPipeline = typename cutlass::PipelineTmaAsync<Stages, decltype(cluster_shape)>;
|
||||
|
||||
int smem_size = int(sizeof(SharedStorage<Stages, decltype(cluster_shape)>));
|
||||
|
||||
result = cudaFuncSetAttribute(
|
||||
pipeline_device<decltype(cluster_shape), Stages>,
|
||||
cudaFuncAttributeMaxDynamicSharedMemorySize,
|
||||
smem_size);
|
||||
|
||||
// Launch a single Cluster, with 128 thread per CTA
|
||||
dim3 dimCluster(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimGrid(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimBlock(kBlockSize,1,1);
|
||||
|
||||
const void* kernel = (const void*)pipeline_device<decltype(cluster_shape), Stages>;
|
||||
int iters = kNumIters;
|
||||
void* kernel_params[] = {reinterpret_cast<void*>(&iters)};
|
||||
cutlass::ClusterLauncher::launch(dimGrid, dimCluster, dimBlock, smem_size, stream, kernel, kernel_params);
|
||||
|
||||
} // profiling loop ends
|
||||
|
||||
result = cudaEventRecord(events[1]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to record stop event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
result = cudaDeviceSynchronize();
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: cudaDeviceSynchronize() failed" << std::endl;
|
||||
return result;
|
||||
}
|
||||
|
||||
result = cudaEventElapsedTime(&elapsed_ms, events[0], events[1]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Failed to create event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
for (cudaEvent_t & event : events) {
|
||||
(void)cudaEventDestroy(event);
|
||||
}
|
||||
|
||||
return cudaSuccess;
|
||||
}
|
||||
};
|
||||
|
||||
#if CUDA_12_0_SM90_FEATURES_SUPPORTED
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster1x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster1x1_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster1x1_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster2x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster2x2_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster2x2_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster4x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster4x4_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster1x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster1x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster1x2_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster2x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster2x1_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster4x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster4x1_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster1x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster1x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster2x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster2x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster4x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync, Cluster4x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,525 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2023 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Unit test for the PipelineTmaAsync class as it would be used in a Warp specialized loop
|
||||
*/
|
||||
|
||||
#define KERNEL_DBG_TRACE false
|
||||
|
||||
#include "../common/cutlass_unit_test.h"
|
||||
#include <thrust/host_vector.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cute/tensor.hpp>
|
||||
#include <cute/arch/cluster_sm90.hpp>
|
||||
|
||||
#include <cutlass/util/reference/host/gemm.h>
|
||||
#include <cutlass/cluster_launch.hpp>
|
||||
|
||||
#include "cutlass/core_io.h"
|
||||
#include "cutlass/util/print_error.hpp"
|
||||
#include "cutlass/util/GPU_Clock.hpp"
|
||||
|
||||
#include "testbed.h"
|
||||
#include "cutlass/pipeline.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cute/arch/cluster_sm90.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cutlass/arch/reg_reconfig.h"
|
||||
|
||||
|
||||
using namespace cute;
|
||||
using namespace cutlass;
|
||||
|
||||
//////////////////// KERNEL /////////////////////////
|
||||
|
||||
template <uint32_t Stages, typename ClusterShape>
|
||||
struct SharedStorage
|
||||
{
|
||||
typename cutlass::PipelineTmaAsync<Stages, ClusterShape>::SharedStorage storage ;
|
||||
};
|
||||
|
||||
struct KernelParams
|
||||
{
|
||||
uint32_t num_iterations;
|
||||
int* data_ptr;
|
||||
};
|
||||
|
||||
// Goal of this kernel is to complete deadlock-free
|
||||
template <typename ClusterShape, uint32_t Stages>
|
||||
__launch_bounds__(384, 1)
|
||||
__global__ static
|
||||
void pipeline_device(KernelParams const kernel_params)
|
||||
{
|
||||
extern __shared__ char shared_memory[];
|
||||
using MainloopPipeline = typename cutlass::PipelineTmaAsync<Stages, ClusterShape>;
|
||||
using PipelineState = typename cutlass::PipelineState<Stages>;
|
||||
|
||||
using SharedStorage = SharedStorage<Stages, ClusterShape>;
|
||||
SharedStorage& shared_storage = *reinterpret_cast<SharedStorage*>(shared_memory);
|
||||
|
||||
auto cta_layout = Layout<ClusterShape>{}; // (m,n) -> cta_id
|
||||
int warp_group_idx = __shfl_sync(0xffffffff, threadIdx.x / 128, 0);
|
||||
int warp_idx_in_warpgroup = __shfl_sync(0xffffffff, (threadIdx.x / 32) % 4, 0);
|
||||
int warp_group_thread_idx = threadIdx.x % 128;
|
||||
dim3 block_id_in_cluster = cute::block_id_in_cluster();
|
||||
|
||||
auto cluster_shape = ClusterShape{};
|
||||
|
||||
// #Producers = #RowsInCluster + #ColsInCluster - 1
|
||||
uint32_t const NumProducers = cute::size<0>(cluster_shape) + cute::size<1>(cluster_shape) - 1;
|
||||
uint32_t const TmaTransactionBytes = static_cast<uint32_t>(sizeof(uint32_t) * NumProducers);
|
||||
uint32_t const per_cta_bytes = sizeof(uint32_t);
|
||||
|
||||
// mbarrier.init
|
||||
typename MainloopPipeline::Params params;
|
||||
params.transaction_bytes = TmaTransactionBytes;
|
||||
if (warp_group_idx == 0) {
|
||||
params.role = MainloopPipeline::ThreadCategory::Producer;
|
||||
}
|
||||
else {
|
||||
params.role = MainloopPipeline::ThreadCategory::Consumer;
|
||||
}
|
||||
params.is_leader = warp_group_thread_idx == 0;
|
||||
params.num_consumers = 128;
|
||||
|
||||
MainloopPipeline pipeline(shared_storage.storage, params);
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Ensure All CTAs in Cluster have completed init before issuing commits
|
||||
cute::cluster_arrive_relaxed();
|
||||
cute::cluster_wait();
|
||||
|
||||
|
||||
// Producer WarpGroup
|
||||
if (warp_group_idx == 0) {
|
||||
cutlass::arch::warpgroup_reg_alloc<232>();
|
||||
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
if (warp_idx_in_warpgroup == 0 && lane_predicate) {
|
||||
|
||||
int tma_k_prologue = min(Stages, kernel_params.num_iterations);
|
||||
|
||||
// Simulating Prologue TMA Loads
|
||||
// For the DMA (prologue) - we start with an opposite phase - since we skip all waits
|
||||
// i.e., we know that the buffer is indeed empty
|
||||
PipelineState smem_pipe_write = make_producer_start_state<MainloopPipeline>();
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for(int i = 0; i < tma_k_prologue; ++i) {
|
||||
pipeline.producer_acquire(smem_pipe_write);
|
||||
// Simulating cp.async.bulk.tensor behavior
|
||||
pipeline.producer_commit(smem_pipe_write.index(), per_cta_bytes);
|
||||
++smem_pipe_write;
|
||||
}
|
||||
int tma_k_iter = kernel_params.num_iterations - tma_k_prologue;
|
||||
|
||||
// Simulating Mainloop TMA Loads
|
||||
CUTE_NO_UNROLL
|
||||
for ( ; tma_k_iter > 0; --tma_k_iter) {
|
||||
|
||||
pipeline.producer_acquire(smem_pipe_write);
|
||||
|
||||
// Simulating cp.async.bulk.tensor behavior
|
||||
pipeline.producer_commit(smem_pipe_write.index(), per_cta_bytes);
|
||||
|
||||
// Advance write stage
|
||||
++smem_pipe_write;
|
||||
}
|
||||
|
||||
// Tail Loop
|
||||
// Handles the case where we never enter the mainloop
|
||||
PipelineState tail = tma_k_prologue == Stages ? smem_pipe_write : PipelineState{};
|
||||
for ( int i = 0; i < tma_k_prologue; ++i) {
|
||||
pipeline.producer_acquire(tail);
|
||||
++tail;
|
||||
}
|
||||
}
|
||||
// Consumer WarpGroup
|
||||
} else if(warp_group_idx == 1) {
|
||||
cutlass::arch::warpgroup_reg_alloc<232>();
|
||||
|
||||
PipelineState smem_pipe_read;
|
||||
PipelineState smem_pipe_release;
|
||||
|
||||
// simulates accumulators + extra reg. pressure
|
||||
int arr[168];
|
||||
|
||||
// Init Shared Memory read stages & PhaseBit
|
||||
static constexpr uint32_t K_PIPE_MMAS = 1;
|
||||
static_assert( K_PIPE_MMAS < Stages, "ERROR : Too many MMAs in flight");
|
||||
|
||||
// Total number of gemm iterations
|
||||
auto gemm_k_iterations = kernel_params.num_iterations;
|
||||
|
||||
// Simulating Prologue MMAs
|
||||
int mma_k_prologue = min(K_PIPE_MMAS, gemm_k_iterations);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int iter = 0; iter < mma_k_prologue; ++iter) {
|
||||
pipeline.consumer_wait(smem_pipe_read);
|
||||
|
||||
warpgroup_arrive();
|
||||
// GMMA would typically happen here
|
||||
|
||||
++smem_pipe_read;
|
||||
}
|
||||
gemm_k_iterations -= mma_k_prologue;
|
||||
|
||||
// Simulating Mainloop MMAs
|
||||
CUTLASS_PRAGMA_NO_UNROLL
|
||||
for ( ; gemm_k_iterations > 0; --gemm_k_iterations) {
|
||||
|
||||
/// Wait on the smem_pipe_read stage / phase
|
||||
pipeline.consumer_wait(smem_pipe_read);
|
||||
|
||||
warpgroup_arrive();
|
||||
// GMMA would typically happen here
|
||||
|
||||
// Dummy op - which will never happen
|
||||
// But simulates high register usage.
|
||||
CUTE_UNROLL
|
||||
for(int i = 0; i < 168; ++i){
|
||||
if (threadIdx.x > 256){
|
||||
arr[i] += kernel_params.data_ptr[i];
|
||||
}
|
||||
}
|
||||
|
||||
pipeline.consumer_release(smem_pipe_release);
|
||||
|
||||
// Advance stages
|
||||
++smem_pipe_read;
|
||||
++smem_pipe_release;
|
||||
}
|
||||
|
||||
// Dummy op - which will never happen
|
||||
CUTE_UNROLL
|
||||
for(int i = 0; i < 168; ++i){
|
||||
if (threadIdx.x > 256){
|
||||
kernel_params.data_ptr[i] = arr[i];
|
||||
}
|
||||
}
|
||||
|
||||
// Tail Loop
|
||||
for (int i = 0; i < K_PIPE_MMAS; ++i){
|
||||
pipeline.consumer_release(smem_pipe_release);
|
||||
++smem_pipe_release;
|
||||
}
|
||||
|
||||
// Warp-Group #2
|
||||
} else {
|
||||
cutlass::arch::warpgroup_reg_dealloc<40>();
|
||||
}
|
||||
}
|
||||
/////////////////////////////////////////////////////
|
||||
|
||||
/// Device NT GMMA + TMA specialized
|
||||
template<uint32_t Stages_, typename ClusterShape_>
|
||||
struct PipelineTest {
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
static constexpr uint32_t Stages = Stages_;
|
||||
static constexpr uint32_t kBlockSize = 128 * 3;
|
||||
using ClusterShape = ClusterShape_;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
// Ctor
|
||||
PipelineTest(){};
|
||||
|
||||
// Run CuTe GEMM kernel
|
||||
cudaError_t run(uint32_t const kNumIters,
|
||||
cudaStream_t stream = 0) {
|
||||
|
||||
float elapsed_ms = 0.0f;
|
||||
// Pipeline (multistage pipeline)
|
||||
auto num_stages = Int<Stages>{};
|
||||
auto cluster_shape = Shape<Int<ClusterShape::kM>, Int<ClusterShape::kN>, _1>{};
|
||||
|
||||
//
|
||||
// Configure and launch
|
||||
//
|
||||
int iterations = 1;
|
||||
cudaEvent_t events[2];
|
||||
cudaError_t result;
|
||||
|
||||
for (cudaEvent_t & event : events) {
|
||||
result = cudaEventCreate(&event);
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to create event.";
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
result = cudaEventRecord(events[0]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to record start event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
for (int iter = 0; iter < iterations; ++iter) {
|
||||
|
||||
using MainloopPipeline = typename cutlass::PipelineTmaAsync<Stages, decltype(cluster_shape)>;
|
||||
|
||||
int smem_size = int(sizeof(SharedStorage<Stages, decltype(cluster_shape)>));
|
||||
|
||||
result = cudaFuncSetAttribute(
|
||||
pipeline_device<decltype(cluster_shape), Stages>,
|
||||
cudaFuncAttributeMaxDynamicSharedMemorySize,
|
||||
smem_size);
|
||||
|
||||
// Launch a single Cluster, with kBlockSize threads per CTA
|
||||
dim3 dimCluster(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimGrid(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimBlock(kBlockSize,1,1);
|
||||
|
||||
const void* kernel = (const void*)pipeline_device<decltype(cluster_shape), Stages>;
|
||||
KernelParams params{kNumIters, nullptr};
|
||||
void* kernel_params[] = {reinterpret_cast<void*>(¶ms)};
|
||||
cutlass::ClusterLauncher::launch(dimGrid, dimCluster, dimBlock, smem_size, stream, kernel, kernel_params);
|
||||
|
||||
}
|
||||
|
||||
result = cudaEventRecord(events[1]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to record stop event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
result = cudaDeviceSynchronize();
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: cudaDeviceSynchronize() failed" << std::endl;
|
||||
return result;
|
||||
}
|
||||
|
||||
result = cudaEventElapsedTime(&elapsed_ms, events[0], events[1]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Failed to create event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
for (cudaEvent_t & event : events) {
|
||||
(void)cudaEventDestroy(event);
|
||||
}
|
||||
|
||||
return cudaSuccess;
|
||||
}
|
||||
};
|
||||
|
||||
#if CUDA_12_0_SM90_FEATURES_SUPPORTED
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster1x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster1x1_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster1x1_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster2x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster2x2_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster2x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster4x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster4x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster2x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster2x1_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster1x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster1x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster4x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster4x1_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster1x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster1x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster2x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster2x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster4x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS, Cluster4x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,585 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2023 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Unit test for the PipelineTmaAsync class used in a WarpSpecialized Persistent loop
|
||||
*/
|
||||
|
||||
#define KERNEL_DBG_TRACE false
|
||||
|
||||
#include "../common/cutlass_unit_test.h"
|
||||
#include <thrust/host_vector.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cute/tensor.hpp>
|
||||
#include <cute/arch/cluster_sm90.hpp>
|
||||
|
||||
#include <cutlass/util/reference/host/gemm.h>
|
||||
#include <cutlass/cluster_launch.hpp>
|
||||
|
||||
#include "cutlass/core_io.h"
|
||||
#include "cutlass/util/print_error.hpp"
|
||||
#include "cutlass/util/GPU_Clock.hpp"
|
||||
|
||||
#include "testbed.h"
|
||||
#include "cutlass/pipeline.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cute/arch/cluster_sm90.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cutlass/arch/reg_reconfig.h"
|
||||
|
||||
|
||||
using namespace cute;
|
||||
using namespace cutlass;
|
||||
|
||||
//////////////////// KERNEL /////////////////////////
|
||||
|
||||
template <uint32_t Stages, typename ClusterShape, typename PingPongBarrier>
|
||||
struct SharedStorage
|
||||
{
|
||||
typename cutlass::PipelineTmaAsync<Stages, ClusterShape>::SharedStorage pipeline_storage;
|
||||
typename PingPongBarrier::SharedStorage pingpong_storage;
|
||||
};
|
||||
|
||||
template <typename ClusterShape, uint32_t Stages>
|
||||
struct CollectiveSimulation {
|
||||
using MainloopPipeline = typename cutlass::PipelineTmaAsync<Stages, ClusterShape>;
|
||||
using PipelineState = typename cutlass::PipelineState<Stages>;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
static void
|
||||
dma_wg_simulation(MainloopPipeline pipeline, PipelineState tile_start_state_pipe,
|
||||
uint32_t const num_iterations) {
|
||||
uint32_t const per_cta_bytes = sizeof(uint32_t);
|
||||
int warp_idx_in_warpgroup = __shfl_sync(0xffffffff, (threadIdx.x / 32) % 4, 0);
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
if (warp_idx_in_warpgroup==0 && lane_predicate) {
|
||||
|
||||
int tma_k_prologue = min(Stages, num_iterations);
|
||||
|
||||
// Simulating Prologue TMA Loads
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for(int i = 0; i < tma_k_prologue; ++i) {
|
||||
pipeline.producer_acquire(tile_start_state_pipe);
|
||||
// Simulating cp.async.bulk.tensor behavior
|
||||
pipeline.producer_commit(tile_start_state_pipe.index(), per_cta_bytes);
|
||||
++tile_start_state_pipe;
|
||||
}
|
||||
int tma_k_iter = num_iterations - tma_k_prologue;
|
||||
|
||||
PipelineState wr_pipe = tile_start_state_pipe;
|
||||
// Simulating Mainloop TMA Loads
|
||||
CUTE_NO_UNROLL
|
||||
for ( ; tma_k_iter > 0; --tma_k_iter){
|
||||
|
||||
pipeline.producer_acquire(wr_pipe);
|
||||
|
||||
// Simulating cp.async.bulk.tensor behavior
|
||||
pipeline.producer_commit(wr_pipe.index(), per_cta_bytes);
|
||||
|
||||
// Advance write stage
|
||||
++wr_pipe;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
static void
|
||||
math_wg_simulation(MainloopPipeline pipeline, PipelineState tile_start_state_pipe,
|
||||
uint32_t const num_iterations, int* data_ptr) {
|
||||
PipelineState rd_pipe = tile_start_state_pipe;
|
||||
PipelineState release_pipe = rd_pipe;
|
||||
|
||||
// simulates accumulators + extra reg. pressure
|
||||
int arr[168];
|
||||
|
||||
// Init Shared Memory read stages & PhaseBit
|
||||
static constexpr uint32_t K_PIPE_MMAS = 1;
|
||||
static_assert( K_PIPE_MMAS < Stages, "ERROR : Too many MMAs in flight");
|
||||
|
||||
// Total number of gemm iterations
|
||||
auto gemm_k_iterations = num_iterations;
|
||||
|
||||
// Simulating Prologue MMAs
|
||||
int mma_k_prologue = min(K_PIPE_MMAS, gemm_k_iterations);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int iter = 0; iter < mma_k_prologue; ++iter) {
|
||||
pipeline.consumer_wait(rd_pipe);
|
||||
|
||||
warpgroup_arrive();
|
||||
// GMMA would typically happen here
|
||||
|
||||
++rd_pipe;
|
||||
}
|
||||
gemm_k_iterations -= mma_k_prologue;
|
||||
|
||||
// Simulating Mainloop MMAs
|
||||
CUTLASS_PRAGMA_NO_UNROLL
|
||||
for ( ; gemm_k_iterations > 0; --gemm_k_iterations) {
|
||||
|
||||
/// Wait on the rd_pipe stage / phase
|
||||
pipeline.consumer_wait(rd_pipe);
|
||||
|
||||
warpgroup_arrive();
|
||||
// GMMA would typically happen here
|
||||
|
||||
// Dummy op - which will never happen
|
||||
// But simulates high register usage.
|
||||
CUTE_UNROLL
|
||||
for(int i = 0; i < 168; ++i){
|
||||
if (threadIdx.x > 384){
|
||||
arr[i] += data_ptr[i];
|
||||
}
|
||||
}
|
||||
|
||||
pipeline.consumer_release(release_pipe);
|
||||
|
||||
// Advance stages
|
||||
++rd_pipe;
|
||||
++release_pipe;
|
||||
}
|
||||
|
||||
// Dummy op - which will never happen
|
||||
CUTE_UNROLL
|
||||
for(int i = 0; i < 168; ++i){
|
||||
if (threadIdx.x > 384){
|
||||
data_ptr[i] = arr[i];
|
||||
}
|
||||
}
|
||||
|
||||
// Tail Loop
|
||||
for (int i = 0; i < K_PIPE_MMAS; ++i){
|
||||
pipeline.consumer_release(release_pipe);
|
||||
++release_pipe;
|
||||
}
|
||||
|
||||
}
|
||||
};
|
||||
|
||||
struct KernelParams
|
||||
{
|
||||
uint32_t num_iterations;
|
||||
int tiles_per_cluster;
|
||||
int* data_ptr;
|
||||
};
|
||||
|
||||
// Goal of this kernel is to complete deadlock-free
|
||||
template <typename ClusterShape, uint32_t Stages>
|
||||
__launch_bounds__(384, 1)
|
||||
__global__ static
|
||||
void pipeline_device(KernelParams params)
|
||||
{
|
||||
extern __shared__ char shared_memory[];
|
||||
using DispatchPolicy = cutlass::gemm::MainloopSm90TmaGmmaWarpSpecialized<Stages,
|
||||
ClusterShape,
|
||||
cutlass::gemm::KernelTmaWarpSpecializedPersistent>;
|
||||
using MainloopPipeline = typename cutlass::PipelineTmaAsync<Stages, ClusterShape>;
|
||||
using PipelineState = typename cutlass::PipelineState<Stages>;
|
||||
|
||||
/* One for Mainloop and one for Epilogue */
|
||||
constexpr int StagesPerMathWarpGroup = 2;
|
||||
constexpr int MathWarpGroupCountPersistent = 2;
|
||||
using PingPongBarrier = typename cutlass::OrderedSequenceBarrier<StagesPerMathWarpGroup, MathWarpGroupCountPersistent>;
|
||||
|
||||
using SharedStorage = SharedStorage<Stages, ClusterShape, PingPongBarrier>;
|
||||
SharedStorage& shared_storage = *reinterpret_cast<SharedStorage*>(shared_memory);
|
||||
|
||||
auto cta_layout = Layout<ClusterShape>{}; // (m,n) -> cta_id
|
||||
int warp_group_idx = __shfl_sync(0xffffffff, threadIdx.x / NumThreadsPerWarpGroup, 0);
|
||||
int warp_group_thread_idx = threadIdx.x % NumThreadsPerWarpGroup;
|
||||
dim3 block_id_in_cluster = cute::block_id_in_cluster();
|
||||
|
||||
auto cluster_shape = ClusterShape{};
|
||||
|
||||
// #Producers = #RowsInCluster + #ColsInCluster - 1
|
||||
uint32_t const NumProducers = cute::size<0>(cluster_shape) + cute::size<1>(cluster_shape) - 1;
|
||||
uint32_t const TmaTransactionBytes = static_cast<uint32_t>(sizeof(uint32_t) * NumProducers);
|
||||
|
||||
// mbarrier.init
|
||||
typename MainloopPipeline::Params pipeline_params;
|
||||
pipeline_params.transaction_bytes = TmaTransactionBytes;
|
||||
if (warp_group_idx == 0) {
|
||||
pipeline_params.role = MainloopPipeline::ThreadCategory::Producer;
|
||||
}
|
||||
else {
|
||||
pipeline_params.role = MainloopPipeline::ThreadCategory::Consumer;
|
||||
}
|
||||
pipeline_params.is_leader = warp_group_thread_idx == 0;
|
||||
pipeline_params.num_consumers = NumThreadsPerWarpGroup;
|
||||
|
||||
MainloopPipeline pipeline(shared_storage.pipeline_storage, pipeline_params);
|
||||
PipelineState tile_start_state_pipe;
|
||||
|
||||
int tiles_per_cluster = params.tiles_per_cluster;
|
||||
|
||||
/* Offset pipeline start state for Math WG 2 */
|
||||
if (warp_group_idx == 2) {
|
||||
// Update pipeline state for next persistent tile
|
||||
tile_start_state_pipe.advance(params.num_iterations);
|
||||
tiles_per_cluster--;
|
||||
}
|
||||
|
||||
typename PingPongBarrier::Params pingpong_params;
|
||||
pingpong_params.group_id = warp_group_idx - 1; // Since DMA Warp Group Idx 0 will not participate
|
||||
pingpong_params.group_size = NumThreadsPerWarpGroup; // Number of threads / participants in a group
|
||||
PingPongBarrier math_wg_barrier(shared_storage.pingpong_storage, pingpong_params);
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Ensure All CTAs in Cluster have completed init before issuing commits
|
||||
cute::cluster_arrive_relaxed();
|
||||
cute::cluster_wait();
|
||||
|
||||
// Producer/DMA WarpGroup
|
||||
if (warp_group_idx == 0) {
|
||||
cutlass::arch::warpgroup_reg_dealloc<40>();
|
||||
// For the DMA (prologue) - we start with an opposite phase - since we skip all waits
|
||||
// i.e., we know that the buffer is indeed empty
|
||||
PipelineState tile_prologue_state_pipe = make_producer_start_state<MainloopPipeline>();
|
||||
while (tiles_per_cluster > 0) {
|
||||
CollectiveSimulation<ClusterShape,Stages>::dma_wg_simulation(pipeline, tile_prologue_state_pipe, params.num_iterations);
|
||||
// Update pipeline state for next persistent tile
|
||||
tile_prologue_state_pipe.advance(params.num_iterations);
|
||||
tiles_per_cluster--;
|
||||
}
|
||||
}
|
||||
// Math WarpGropups
|
||||
if(warp_group_idx == 1 || warp_group_idx == 2) {
|
||||
cutlass::arch::warpgroup_reg_alloc<232>();
|
||||
while (tiles_per_cluster > 0) {
|
||||
// MMA
|
||||
math_wg_barrier.wait();
|
||||
CollectiveSimulation<ClusterShape,Stages>::math_wg_simulation(pipeline, tile_start_state_pipe, params.num_iterations, params.data_ptr);
|
||||
math_wg_barrier.arrive();
|
||||
// Epilogue
|
||||
math_wg_barrier.wait();
|
||||
// Simulates long running stage
|
||||
#if defined(__CUDA_ARCH__) && (__CUDA_ARCH__ >= 700)
|
||||
__nanosleep(100000);
|
||||
#endif
|
||||
math_wg_barrier.arrive();
|
||||
// Update pipeline state for next persistent tile
|
||||
tile_start_state_pipe.advance(params.num_iterations * 2);
|
||||
tiles_per_cluster -= 2;
|
||||
}
|
||||
}
|
||||
|
||||
// Makes sure remote SMEM doesn't get destroyed
|
||||
cute::cluster_arrive_relaxed();
|
||||
cute::cluster_wait();
|
||||
}
|
||||
/////////////////////////////////////////////////////
|
||||
|
||||
/// Device NT GMMA + TMA specialized
|
||||
template<uint32_t Stages_, typename ClusterShape_>
|
||||
struct PipelineTest {
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
static constexpr uint32_t Stages = Stages_;
|
||||
static constexpr uint32_t kBlockSize = 128 * 3;
|
||||
using ClusterShape = ClusterShape_;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
// Run CuTe GEMM kernel
|
||||
cudaError_t run(uint32_t const kNumIters,
|
||||
cudaStream_t stream = 0) {
|
||||
|
||||
float elapsed_ms = 0.0f;
|
||||
// Pipeline (multistage pipeline)
|
||||
auto num_stages = Int<Stages>{};
|
||||
auto cluster_shape = Shape<Int<ClusterShape::kM>, Int<ClusterShape::kN>, _1>{};
|
||||
|
||||
//
|
||||
// Configure and launch
|
||||
//
|
||||
int iterations = 1;
|
||||
cudaEvent_t events[2];
|
||||
cudaError_t result;
|
||||
|
||||
for (cudaEvent_t & event : events) {
|
||||
result = cudaEventCreate(&event);
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to create event.";
|
||||
return result;
|
||||
}
|
||||
}
|
||||
|
||||
result = cudaEventRecord(events[0]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to record start event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
for (int iter = 0; iter < iterations; ++iter) {
|
||||
|
||||
using MainloopPipeline = typename cutlass::PipelineTmaAsync<Stages, decltype(cluster_shape)>;
|
||||
|
||||
constexpr int StagesPerMathWarpGroup = 2;
|
||||
constexpr int MathWarpGroupCountPersistent = 2;
|
||||
int smem_size = int(sizeof(SharedStorage<Stages, decltype(cluster_shape),
|
||||
typename cutlass::OrderedSequenceBarrier<StagesPerMathWarpGroup, MathWarpGroupCountPersistent>>));
|
||||
|
||||
result = cudaFuncSetAttribute(
|
||||
pipeline_device<decltype(cluster_shape), Stages>,
|
||||
cudaFuncAttributeMaxDynamicSharedMemorySize,
|
||||
smem_size);
|
||||
|
||||
// Launch a single Cluster, with kBlockSize threads per CTA
|
||||
dim3 dimCluster(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimGrid(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimBlock(kBlockSize,1,1);
|
||||
|
||||
int tiles_per_cluster = (kNumIters % 10) + 1;
|
||||
printf("Persistent version: Tiles per Cluster = %d\n", tiles_per_cluster);
|
||||
|
||||
const void* kernel = (const void*)pipeline_device<decltype(cluster_shape), Stages>;
|
||||
KernelParams params{kNumIters, tiles_per_cluster, nullptr};
|
||||
void *kernel_params[] = {¶ms};
|
||||
cutlass::ClusterLauncher::launch(dimGrid, dimCluster, dimBlock, smem_size, stream, kernel, kernel_params);
|
||||
|
||||
}
|
||||
|
||||
result = cudaEventRecord(events[1]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: Failed to record stop event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
result = cudaDeviceSynchronize();
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: cudaDeviceSynchronize() failed" << std::endl;
|
||||
return result;
|
||||
}
|
||||
|
||||
result = cudaEventElapsedTime(&elapsed_ms, events[0], events[1]);
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Failed to create event.";
|
||||
return result;
|
||||
}
|
||||
|
||||
for (cudaEvent_t & event : events) {
|
||||
(void)cudaEventDestroy(event);
|
||||
}
|
||||
|
||||
return cudaSuccess;
|
||||
}
|
||||
};
|
||||
|
||||
#if CUDA_12_0_SM90_FEATURES_SUPPORTED
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster1x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster1x1_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster1x1_Stage10) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 1, 1>;
|
||||
static constexpr uint32_t Stages = 10;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster2x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster2x2_Stage5) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 5;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster2x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster4x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster4x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster2x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster2x1_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 1, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster1x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster1x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster4x1_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster4x1_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 1, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster1x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster1x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<1, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster2x4_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster2x4_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<2, 4, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster4x2_Stage2) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_PipelineTmaAsync_WS_Persistent, Cluster4x2_Stage7) {
|
||||
Options options;
|
||||
using ClusterShape = cutlass::gemm::GemmShape<4, 2, 1>;
|
||||
static constexpr uint32_t Stages = 7;
|
||||
using Test = PipelineTest<Stages, ClusterShape>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,226 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2023 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Unit test for the OrderedSequenceBarrier class
|
||||
*/
|
||||
|
||||
#include "../common/cutlass_unit_test.h"
|
||||
#include <thrust/host_vector.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cute/tensor.hpp>
|
||||
#include <cute/arch/cluster_sm90.hpp>
|
||||
|
||||
#include <cutlass/util/reference/host/gemm.h>
|
||||
#include <cutlass/cluster_launch.hpp>
|
||||
|
||||
#include "cutlass/core_io.h"
|
||||
|
||||
#include "cutlass/util/print_error.hpp"
|
||||
#include "cutlass/util/GPU_Clock.hpp"
|
||||
|
||||
#include "testbed.h"
|
||||
#include "cutlass/pipeline.hpp"
|
||||
#include "cutlass/arch/barrier.h"
|
||||
#include "cute/arch/cluster_sm90.hpp"
|
||||
|
||||
using namespace cute;
|
||||
|
||||
//////////////////// KERNEL /////////////////////////
|
||||
|
||||
template<typename OrderedSequencer>
|
||||
struct SharedStorage
|
||||
{
|
||||
typename OrderedSequencer::SharedStorage storage;
|
||||
};
|
||||
|
||||
// Goal of this kernel is to complete deadlock-free
|
||||
template<int Stages, int GroupCount, int ThreadsPerGroup>
|
||||
__global__ static
|
||||
void ordered_sequence_device(uint32_t const num_iterations)
|
||||
{
|
||||
|
||||
extern __shared__ char shared_memory[];
|
||||
using SequenceBarrier = typename cutlass::OrderedSequenceBarrier<Stages, GroupCount>;
|
||||
using SmemStorage = SharedStorage<SequenceBarrier>;
|
||||
|
||||
SmemStorage& shared_storage = *reinterpret_cast<SmemStorage*>(shared_memory);
|
||||
|
||||
int group_idx = threadIdx.x / ThreadsPerGroup;
|
||||
|
||||
typename SequenceBarrier::Params params;
|
||||
params.group_id = group_idx; // sequence ID
|
||||
params.group_size = ThreadsPerGroup; // Number of threads / participants in a group
|
||||
|
||||
SequenceBarrier barrier(shared_storage.storage, params);
|
||||
|
||||
// Ensure All CTAs in Cluster have completed init before issuing commits
|
||||
__syncthreads();
|
||||
cute::cluster_arrive_relaxed();
|
||||
cute::cluster_wait();
|
||||
|
||||
CUTLASS_PRAGMA_NO_UNROLL
|
||||
for (int i = 0; i < num_iterations; ++i){
|
||||
|
||||
barrier.wait();
|
||||
// STAGE 1 CODE...
|
||||
#ifndef NDEBUG
|
||||
int thread_idx_in_group = threadIdx.x % ThreadsPerGroup;
|
||||
if (thread_idx_in_group == 0) {
|
||||
printf("STAGE 0 : Group_IDX : %d, id = %d, iter = %d, tidx = %d\n", group_idx, params.id, i, threadIdx.x);
|
||||
}
|
||||
#endif
|
||||
// Simulates long running stage
|
||||
#if defined(__CUDA_ARCH__) && (__CUDA_ARCH__ >= 700)
|
||||
__nanosleep(100000);
|
||||
#endif
|
||||
barrier.arrive();
|
||||
|
||||
barrier.wait();
|
||||
// STAGE 2 CODE...
|
||||
#ifndef NDEBUG
|
||||
if (thread_idx_in_group == 0) {
|
||||
printf("STAGE 1 : Group_IDX : %d, id = %d, iter = %d, tidx = %d\n", group_idx, params.id, i, threadIdx.x);
|
||||
}
|
||||
#endif
|
||||
// Simulates long running stage
|
||||
#if defined(__CUDA_ARCH__) && (__CUDA_ARCH__ >= 700)
|
||||
__nanosleep(100000);
|
||||
#endif
|
||||
barrier.arrive();
|
||||
}
|
||||
|
||||
// To make sure remote SMEM doesn't get destroyed
|
||||
cute::cluster_arrive();
|
||||
cute::cluster_wait();
|
||||
}
|
||||
/////////////////////////////////////////////////////
|
||||
|
||||
template<uint32_t Stages_, uint32_t GroupCount_>
|
||||
struct PipelineTest {
|
||||
|
||||
//
|
||||
// Data members
|
||||
//
|
||||
static constexpr uint32_t ThreadsPerGroup = 128;
|
||||
static constexpr uint32_t BlockSize = GroupCount_ * ThreadsPerGroup;
|
||||
static constexpr uint32_t Stages = Stages_;
|
||||
static constexpr uint32_t GroupCount = GroupCount_;
|
||||
using SequenceBarrier = typename cutlass::OrderedSequenceBarrier<Stages, GroupCount>;
|
||||
using SmemStorage = SharedStorage<SequenceBarrier>;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
// Run CuTe GEMM kernel
|
||||
cudaError_t run(uint32_t const kNumIters,
|
||||
cudaStream_t stream = nullptr) {
|
||||
|
||||
// Pipeline (multistage pipeline)
|
||||
auto cluster_shape = Shape<_1, _1, _1>{};
|
||||
|
||||
//
|
||||
// Configure and launch
|
||||
//
|
||||
int iterations = 1;
|
||||
cudaError_t result;
|
||||
|
||||
for (int iter = 0; iter < iterations; ++iter) {
|
||||
|
||||
int smem_size = int(sizeof(SmemStorage));
|
||||
|
||||
result = cudaFuncSetAttribute(
|
||||
ordered_sequence_device<Stages, GroupCount, ThreadsPerGroup>,
|
||||
cudaFuncAttributeMaxDynamicSharedMemorySize,
|
||||
smem_size);
|
||||
|
||||
// Launch a single Cluster, with 128 thread per CTA
|
||||
dim3 dimCluster(size<0>(cluster_shape), size<1>(cluster_shape), size<2>(cluster_shape));
|
||||
dim3 dimGrid(size<0>(cluster_shape), size<1>(cluster_shape), 1);
|
||||
dim3 dimBlock(BlockSize,1,1);
|
||||
|
||||
const void* kernel = (const void*)ordered_sequence_device<Stages, GroupCount, ThreadsPerGroup>;
|
||||
int iters = kNumIters;
|
||||
void* kernel_params[] = {reinterpret_cast<void*>(&iters)};
|
||||
cutlass::ClusterLauncher::launch(dimGrid, dimCluster, dimBlock, smem_size, stream, kernel, kernel_params);
|
||||
|
||||
} // profiling loop ends
|
||||
|
||||
result = cudaDeviceSynchronize();
|
||||
|
||||
if (result != cudaSuccess) {
|
||||
std::cerr << "Error: cudaDeviceSynchronize() failed" << std::endl;
|
||||
return result;
|
||||
}
|
||||
|
||||
return cudaSuccess;
|
||||
}
|
||||
};
|
||||
|
||||
#if CUDA_12_0_SM90_FEATURES_SUPPORTED
|
||||
TEST(SM90_Verify_OrderedSequence, Depth_2_Length_2) {
|
||||
Options options;
|
||||
static constexpr uint32_t GroupCount = 2;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, GroupCount>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_OrderedSequence, Depth_2_Length_3) {
|
||||
Options options;
|
||||
static constexpr uint32_t GroupCount = 3;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, GroupCount>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_OrderedSequence, Depth_2_Length_4) {
|
||||
Options options;
|
||||
static constexpr uint32_t GroupCount = 4;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, GroupCount>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
|
||||
TEST(SM90_Verify_OrderedSequence, Depth_2_Length_5) {
|
||||
Options options;
|
||||
static constexpr uint32_t GroupCount = 5;
|
||||
static constexpr uint32_t Stages = 2;
|
||||
using Test = PipelineTest<Stages, GroupCount>;
|
||||
Testbed<Test> testbed(options);
|
||||
EXPECT_TRUE(testbed.verification());
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,145 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2023 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
/*! \file
|
||||
\brief Common Testbed file shared by Pipeline unit tests
|
||||
*/
|
||||
|
||||
#include <cstdlib>
|
||||
#include <cstdio>
|
||||
#include <cassert>
|
||||
#include <cutlass/gemm/gemm.h>
|
||||
|
||||
#include "cutlass/util/command_line.h"
|
||||
#include "../common/cutlass_unit_test.h"
|
||||
|
||||
#if CUDA_12_0_SM90_FEATURES_SUPPORTED
|
||||
#define CUTLASS_UNIT_TEST_PIPELINE true
|
||||
#else
|
||||
#define CUTLASS_UNIT_TEST_PIPELINE false
|
||||
#endif
|
||||
|
||||
// Command line test options
|
||||
struct Options {
|
||||
//
|
||||
// Data Members
|
||||
//
|
||||
bool help;
|
||||
bool verification_enabled;
|
||||
int SM_count;
|
||||
int clock_MHz;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
Options():
|
||||
help(false),
|
||||
verification_enabled(true),
|
||||
SM_count(116),
|
||||
clock_MHz(1477)
|
||||
{ }
|
||||
|
||||
void parse(int argc, char const **args) {
|
||||
cutlass::CommandLine cmd(argc, args);
|
||||
|
||||
if (cmd.check_cmd_line_flag("help")) {
|
||||
help = true;
|
||||
}
|
||||
|
||||
cmd.get_cmd_line_argument("verification-enabled", verification_enabled, true);
|
||||
cmd.get_cmd_line_argument("sm-count", SM_count, 116);
|
||||
cmd.get_cmd_line_argument("clock", clock_MHz, 1477);
|
||||
}
|
||||
|
||||
/// Prints the usage statement.
|
||||
std::ostream & print_usage(std::ostream &out) const {
|
||||
|
||||
out << "Options:\n\n"
|
||||
<< " --help If specified, displays this usage statement.\n\n"
|
||||
<< " --verification-enabled=<bool> Enable/Disable verification\n"
|
||||
<< " --sm-count=<int> Number of SMs on the chip\n"
|
||||
<< " --clock=<int> Locked clock value in Mhz\n";
|
||||
|
||||
return out;
|
||||
}
|
||||
};
|
||||
|
||||
//
|
||||
// Testbed
|
||||
//
|
||||
|
||||
template<typename Pipeline>
|
||||
struct Testbed {
|
||||
private:
|
||||
// Commandline options
|
||||
Options options;
|
||||
|
||||
void run_test(uint32_t const kNumIters) {
|
||||
|
||||
// Run CuTe Gemm
|
||||
Pipeline pipeline;
|
||||
|
||||
cudaError_t result = pipeline.run(kNumIters);
|
||||
|
||||
CUTE_CHECK_LAST();
|
||||
}
|
||||
|
||||
|
||||
public:
|
||||
Testbed(Options const &options_) : options(options_) {
|
||||
int device_id = 0;
|
||||
cudaDeviceProp device_prop;
|
||||
CUTE_CHECK_ERROR(cudaSetDevice(device_id));
|
||||
CUTE_CHECK_ERROR(cudaGetDeviceProperties(&device_prop, device_id));
|
||||
|
||||
if (device_prop.major < 1) {
|
||||
fprintf(stderr, "Device does not support CUDA.\n");
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
/// Run verification Gemm problem sizes
|
||||
bool verification() {
|
||||
|
||||
std::array<uint32_t, 5> kNumIters;
|
||||
|
||||
for (int i = 0; i < kNumIters.size(); ++i) {
|
||||
kNumIters[i] = (rand() % 1000) + 1;
|
||||
}
|
||||
|
||||
for (int n : kNumIters) {
|
||||
std::cout << "Stages = " << Pipeline::Stages << " kNumIters = " << n << "\n";
|
||||
run_test(n);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
};
|
||||
Reference in New Issue
Block a user