v4.1 release
This commit is contained in:
@@ -0,0 +1,118 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2017 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
/*! \file
|
||||
\brief Matrix multiply
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "cutlass/arch/mma.h"
|
||||
#include "cutlass/layout/matrix.h"
|
||||
#include "cutlass/numeric_types.h"
|
||||
#include "cutlass/arch/config.h"
|
||||
#include "cute/arch/simd_sm100.hpp"
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace cutlass{
|
||||
namespace arch {
|
||||
|
||||
|
||||
/// Matrix multiply-add operation
|
||||
template <
|
||||
/// Data type of A elements
|
||||
typename ElementA,
|
||||
/// Layout of A matrix (concept: MatrixLayout)
|
||||
typename LayoutA,
|
||||
/// Data type of B elements
|
||||
typename ElementB,
|
||||
/// Layout of B matrix (concept: MatrixLayout)
|
||||
typename LayoutB,
|
||||
/// Element type of C matrix
|
||||
typename ElementC_,
|
||||
/// Layout of C matrix (concept: MatrixLayout)
|
||||
typename LayoutC
|
||||
>
|
||||
struct Mma<gemm::GemmShape<2, 1, 1>, 1, ElementA, LayoutA, ElementB, LayoutB, ElementC_, LayoutC, OpMultiplyAdd> {
|
||||
|
||||
using Shape = gemm::GemmShape<2, 1, 1>;
|
||||
using Operator = OpMultiplyAdd;
|
||||
using ElementC = ElementC_;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
void operator()(
|
||||
Array<ElementC, 2> &d,
|
||||
Array<ElementA, 2> const &a,
|
||||
Array<ElementB, 1> const &b,
|
||||
Array<ElementC, 2> const &c
|
||||
) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
d[i] = a[i] * b[0] + c[i];
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/// Matrix multiply-add operation
|
||||
template <
|
||||
/// Layout of A matrix
|
||||
typename LayoutA,
|
||||
/// Layout of B matrix
|
||||
typename LayoutB,
|
||||
/// Layout of C matrix
|
||||
typename LayoutC
|
||||
>
|
||||
struct Mma<gemm::GemmShape<2, 1, 1>, 1, float, LayoutA, float, LayoutB, float, LayoutC, OpMultiplyAdd> {
|
||||
|
||||
using Shape = gemm::GemmShape<2, 1, 1>;
|
||||
using Operator = OpMultiplyAdd;
|
||||
using ElementC = float;
|
||||
|
||||
CUTLASS_DEVICE
|
||||
void operator()(
|
||||
Array<float, 2> &d,
|
||||
Array<float, 2> const &a,
|
||||
Array<float, 1> const &b,
|
||||
Array<float, 2> const &c
|
||||
) {
|
||||
float2 result;
|
||||
cute::fma(result, make_float2(a[0], a[1]), make_float2(b[0], b[0]), make_float2(c[0], c[1]));
|
||||
d[0] = result.x;
|
||||
d[1] = result.y;
|
||||
}
|
||||
};
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace arch
|
||||
} // namespace cutlass
|
||||
@@ -88,7 +88,7 @@ struct LayoutAwareConvertImpl<
|
||||
static void convert(
|
||||
cute::Tensor<EngineIn,
|
||||
cute::Layout<cute::Shape<_2,_4>, cute::Stride<_4,_1>>
|
||||
> const& src,
|
||||
> const& src,
|
||||
cute::Tensor<EngineOut,
|
||||
cute::Layout<_8>
|
||||
>& dst) {
|
||||
@@ -136,7 +136,7 @@ struct LayoutAwareConvertImpl<
|
||||
static void convert(
|
||||
cute::Tensor<EngineIn,
|
||||
cute::Layout<cute::Shape<_2,_4>, cute::Stride<_4,_1>>
|
||||
> const& src,
|
||||
> const& src,
|
||||
cute::Tensor<EngineOut,
|
||||
cute::Layout<_8>
|
||||
>& dst) {
|
||||
@@ -184,7 +184,7 @@ struct LayoutAwareConvertImpl<
|
||||
static void convert(
|
||||
cute::Tensor<EngineIn,
|
||||
cute::Layout<cute::Shape<_2,_4>, cute::Stride<_4,_1>>
|
||||
> const& src,
|
||||
> const& src,
|
||||
cute::Tensor<EngineOut,
|
||||
cute::Layout<_8>
|
||||
>& dst) {
|
||||
@@ -250,7 +250,7 @@ struct LayoutAwareConvertImpl<
|
||||
static void convert(
|
||||
cute::Tensor<EngineIn,
|
||||
cute::Layout<cute::Shape<_2,_4>, cute::Stride<_4,_1>>
|
||||
> const& src,
|
||||
> const& src,
|
||||
cute::Tensor<EngineOut,
|
||||
cute::Layout<_8>
|
||||
>& dst) {
|
||||
@@ -477,7 +477,7 @@ void LayoutAwareConvert(
|
||||
Tensor dst_vm = coalesce(dst);
|
||||
Layout src_layout = src_vm.layout();
|
||||
Layout dst_layout = dst_vm.layout();
|
||||
LayoutAwareConvertImpl<SrcType,
|
||||
LayoutAwareConvertImpl<SrcType,
|
||||
DstType,
|
||||
decltype(src_layout),
|
||||
decltype(dst_layout)>::convert(src_vm, dst_vm);
|
||||
@@ -487,18 +487,25 @@ void LayoutAwareConvert(
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace cutlass {
|
||||
namespace detail {
|
||||
enum class ConversionMode {
|
||||
DirectConvert, // A * B
|
||||
ConvertAndScale, // (scale * A) * B
|
||||
ConvertAndScaleWithZero // (scale * A + zeros) * B
|
||||
};
|
||||
} // namespace detail
|
||||
} //namespace cutlass
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace cutlass::gemm::collective::detail {
|
||||
|
||||
template <class PointerType>
|
||||
static constexpr
|
||||
CUTLASS_HOST_DEVICE
|
||||
auto get_logical_ptr(PointerType const* ptr) {
|
||||
if constexpr (cute::sizeof_bits_v<PointerType> < 8) {
|
||||
return subbyte_iterator<PointerType const>(ptr);
|
||||
}
|
||||
else {
|
||||
return ptr;
|
||||
}
|
||||
return cute::recast_ptr<PointerType const>(ptr);
|
||||
}
|
||||
template<int Stages, class LayoutAtom, class TileShape, class Stride>
|
||||
static constexpr
|
||||
@@ -530,8 +537,8 @@ auto get_gmem_layout(Shape const& shape, Stride const& stride) {
|
||||
template<class Collective>
|
||||
struct MixedInputUtils {
|
||||
private:
|
||||
using ConversionMode = cutlass::detail::ConversionMode;
|
||||
using KernelSchedule = typename Collective::KernelSchedule;
|
||||
using ConversionMode = typename Collective::ConversionMode;
|
||||
using SmemLayoutA = typename Collective::SmemLayoutA;
|
||||
using SmemLayoutB = typename Collective::SmemLayoutB;
|
||||
using SmemLayoutScale = typename Collective::SmemLayoutScale;
|
||||
@@ -551,10 +558,10 @@ public:
|
||||
elements_per_smem_scale() {
|
||||
if constexpr (KernelConversionMode == ConversionMode::DirectConvert) {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
else if constexpr (ModeHasScales) {
|
||||
return cute::cosize_v<SmemLayoutScale>;
|
||||
}
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Type not handled in scale smem allocation.");
|
||||
}
|
||||
@@ -565,10 +572,10 @@ public:
|
||||
if constexpr (KernelConversionMode == ConversionMode::DirectConvert ||
|
||||
KernelConversionMode == ConversionMode::ConvertAndScale ) {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
else if constexpr (KernelConversionMode == ConversionMode::ConvertAndScaleWithZero) {
|
||||
return cute::cosize_v<SmemLayoutScale>;
|
||||
}
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Type not handled in scale smem allocation.");
|
||||
}
|
||||
@@ -634,7 +641,7 @@ public:
|
||||
// We are starting a new k-tile so copy the scale
|
||||
if constexpr (KernelConversionMode == ConversionMode::DirectConvert) {
|
||||
// nothing to do
|
||||
}
|
||||
}
|
||||
else if constexpr (ModeHasScales) {
|
||||
auto smem_tiled_copy_S = cute::get<0>(tiled_copy_and_views);
|
||||
auto tCrS_copy_view = cute::get<1>(tiled_copy_and_views);
|
||||
@@ -649,13 +656,23 @@ public:
|
||||
} else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled in A -> RF path.");
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled in A -> RF path.");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Helper functions to select packing for conversion
|
||||
template <class SrcType,
|
||||
class DstType,
|
||||
int Cosize>
|
||||
struct select_packing { // Naive packing policy
|
||||
static constexpr auto value() {
|
||||
return Int<cute::gcd(Cosize, 32 / cute::min(sizeof_bits_v<SrcType>, sizeof_bits_v<DstType>))>{};
|
||||
}
|
||||
};
|
||||
|
||||
// The core converter uses a lookup table to converts i4 -> 8 bit value.
|
||||
template <class EngineIn,
|
||||
class LayoutIn,
|
||||
@@ -669,7 +686,7 @@ public:
|
||||
Tensor<EngineOut, LayoutOut> && dst,
|
||||
Tensor<EngineScale, LayoutScale> const& scales_neg,
|
||||
Tensor<EngineScale, LayoutScale> const& scales_pos) {
|
||||
|
||||
|
||||
lookup_table_convert(src, dst, scales_neg, scales_pos);
|
||||
}
|
||||
template <class EngineIn,
|
||||
@@ -687,7 +704,7 @@ public:
|
||||
|
||||
constexpr int N = cute::cosize(LayoutIn{});
|
||||
static_assert(N == 4 || N == 8);
|
||||
static_assert(cosize(LayoutScale{}) <= N / 4,
|
||||
static_assert(cosize(LayoutScale{}) <= N / 4,
|
||||
"at least 4 consecutive weights must share the same scale.");
|
||||
using SrcArray = cutlass::Array<cutlass::int4b_t, 8>;
|
||||
using DstArray = cutlass::Array<RealSwappedElementB, 8>;
|
||||
@@ -699,7 +716,7 @@ public:
|
||||
|
||||
// Determines if to get from the signed or unsigned candidates
|
||||
static constexpr uint32_t immLut = (0xf0 & 0xcc) | 0xaa;
|
||||
uint32_t sign; // ((reg & 0x88888888) | 0x64206420) >> 1
|
||||
uint32_t sign; // ((reg & 0x88888888) | 0x64206420) >> 1
|
||||
asm volatile(
|
||||
"{\n"
|
||||
" lop3.b32 %0, %1, %2, %3, %4;\n" \
|
||||
@@ -743,13 +760,13 @@ public:
|
||||
static_check_scale(flatten(Layout{}));
|
||||
}
|
||||
template <class EngineIn,
|
||||
class EngineOut,
|
||||
class EngineOut,
|
||||
class LayoutIn,
|
||||
class LayoutOut,
|
||||
class... Ts>
|
||||
CUTLASS_DEVICE
|
||||
static void dequantize_A_kblock(
|
||||
Tensor<EngineIn, LayoutIn> const& tCrA_load,
|
||||
Tensor<EngineIn, LayoutIn> const& tCrA_load,
|
||||
Tensor<EngineOut, LayoutOut>& tCrA_mma,
|
||||
cute::tuple<Ts...>& partitioned_extra_info,
|
||||
int const k_block) {
|
||||
@@ -764,7 +781,7 @@ public:
|
||||
|
||||
Tensor src = tCrA_load(_, _, k_block);
|
||||
Tensor dst = tCrA_mma(_, _, k_block);
|
||||
|
||||
|
||||
CUTE_STATIC_ASSERT_V(size(src(_, 0)) == cosize(src(_, 0).layout()),
|
||||
"The first mode of tensor src must be contiguous in memory");
|
||||
// try to make the size of the first mode equal to 32bit
|
||||
@@ -778,7 +795,7 @@ public:
|
||||
for (int i = 0; i < size<1>(dst_vm); ++i) {
|
||||
LayoutAwareConvert(src_vm(_, i), dst_vm(_, i));
|
||||
}
|
||||
}
|
||||
}
|
||||
else if constexpr (UseScaleLookupTable) {
|
||||
constexpr int num_elements = decltype(size(src))::value;
|
||||
static_assert(is_same_v<RealSwappedElementA, cutlass::int4b_t>, "Lookup table only supports int4 being the quant type now.");
|
||||
@@ -856,7 +873,7 @@ public:
|
||||
CUTE_STATIC_ASSERT_V(size(src) == size(zeros));
|
||||
Tensor scales_vm = cute::group_modes<1,-1>(cute::zipped_divide(scales, Int<NumValPerSrcReg>{}));
|
||||
Tensor zeros_vm = cute::group_modes<1,-1>(cute::zipped_divide(zeros, Int<NumValPerSrcReg>{}));
|
||||
|
||||
|
||||
if constexpr (is_same_v<DstType, ElementScale>) {
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < size<1>(dst_vm); ++i) {
|
||||
@@ -885,6 +902,7 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/// Utilities for any additional inputs inside of the TMA load
|
||||
template <
|
||||
class Params,
|
||||
@@ -897,39 +915,39 @@ public:
|
||||
cute::tuple<Ts...> const& load_inputs,
|
||||
TensorStorage& shared_tensors,
|
||||
uint2 const& cluster_local_block_id,
|
||||
int const m_coord,
|
||||
int const m_coord,
|
||||
int const l_coord) {
|
||||
|
||||
if constexpr (KernelConversionMode == ConversionMode::DirectConvert) {
|
||||
return cute::make_tuple();
|
||||
}
|
||||
}
|
||||
else if constexpr (ModeHasScales) {
|
||||
Tensor sS = make_tensor(make_smem_ptr(shared_tensors.smem_scale.begin()), SmemLayoutScale{}); // (BLK_M,BLK_K,PIPE)
|
||||
Tensor gS_mkl = get<2>(load_inputs);
|
||||
auto block_tma_s = mainloop_params.tma_load_scale.get_slice(cluster_local_block_id.y);
|
||||
Tensor gS = gS_mkl(_,_,m_coord,_,l_coord); // (BLK_M,BLK_K,k)
|
||||
|
||||
Tensor tSgS = block_tma_s.partition_S(gS); // (TMA,TMA_M,TMA_K,k)
|
||||
Tensor tSgS = block_tma_s.partition_S(gS);
|
||||
Tensor tSsS = block_tma_s.partition_D(sS); // (TMA,TMA_M,TMA_K,PIPE)
|
||||
if constexpr (KernelConversionMode == ConversionMode::ConvertAndScale) {
|
||||
return cute::make_tuple(tSgS, tSsS);
|
||||
}
|
||||
}
|
||||
else if constexpr (KernelConversionMode == ConversionMode::ConvertAndScaleWithZero) {
|
||||
Tensor sZ = make_tensor(make_smem_ptr(shared_tensors.smem_zero.begin()), SmemLayoutScale{}); // (BLK_M,BLK_K,PIPE)
|
||||
Tensor gZ_mkl = get<3>(load_inputs);
|
||||
auto block_tma_z = mainloop_params.tma_load_zero.get_slice(cluster_local_block_id.y);
|
||||
Tensor gZ = gZ_mkl(_,_,m_coord,_,l_coord); // (BLK_M,BLK_K,k)
|
||||
|
||||
Tensor tZgZ = block_tma_z.partition_S(gZ); // (TMA,TMA_M,TMA_K,k)
|
||||
Tensor tZgZ = block_tma_z.partition_S(gZ);
|
||||
Tensor tZsZ = block_tma_z.partition_D(sZ); // (TMA,TMA_M,TMA_K,PIPE)
|
||||
return cute::make_tuple(tSgS, tSsS, tZgZ, tZsZ);
|
||||
return cute::make_tuple(tSgS, tSsS, tZgZ, tZsZ);
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled for input partitioning.");
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled for input partitioning.");
|
||||
}
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled for input partitioning.");
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled for input partitioning.");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -938,7 +956,7 @@ public:
|
||||
class ThreadMma,
|
||||
class TensorStorage
|
||||
>
|
||||
CUTLASS_DEVICE
|
||||
CUTLASS_DEVICE
|
||||
static auto partition_extra_mma_info(
|
||||
ThreadMma const& mma_thread_slice,
|
||||
TensorStorage& shared_tensors) {
|
||||
@@ -950,8 +968,8 @@ public:
|
||||
else if constexpr (UseScaleLookupTable) {
|
||||
Tensor sS = make_tensor(make_smem_ptr(shared_tensors.smem_scale.begin()), SmemLayoutScale{});// (BLK_M,BLK_SCALE_K,PIPE)
|
||||
Tensor tCsS = mma_thread_slice.partition_A(sS);
|
||||
Tensor tCrS_neg = make_tensor<ElementScale>(mma_thread_slice.partition_fragment_A(sS(_,_,Int<0>{})).layout());
|
||||
Tensor tCrS_pos = make_tensor<ElementScale>(mma_thread_slice.partition_fragment_A(sS(_,_,Int<0>{})).layout());
|
||||
Tensor tCrS_neg = make_tensor<ElementScale>(mma_thread_slice.partition_fragment_A(sS(_,_,Int<0>{})).layout());
|
||||
Tensor tCrS_pos = make_tensor<ElementScale>(mma_thread_slice.partition_fragment_A(sS(_,_,Int<0>{})).layout());
|
||||
|
||||
if constexpr (KernelConversionMode == ConversionMode::ConvertAndScale) {
|
||||
return cute::make_tuple(tCsS, tCrS_neg, tCrS_pos);
|
||||
@@ -960,7 +978,7 @@ public:
|
||||
else if constexpr (ModeHasScales) {
|
||||
Tensor sS = make_tensor(make_smem_ptr(shared_tensors.smem_scale.begin()), SmemLayoutScale{});// (BLK_M,BLK_SCALE_K,PIPE)
|
||||
Tensor tCsS = mma_thread_slice.partition_A(sS);
|
||||
Tensor tCrS = make_tensor<ElementScale>(mma_thread_slice.partition_fragment_A(sS(_,_,Int<0>{})).layout());
|
||||
Tensor tCrS = make_tensor<ElementScale>(mma_thread_slice.partition_fragment_A(sS(_,_,Int<0>{})).layout());
|
||||
|
||||
if constexpr (KernelConversionMode == ConversionMode::ConvertAndScale) {
|
||||
return cute::make_tuple(tCsS, tCrS);
|
||||
@@ -968,13 +986,13 @@ public:
|
||||
else if constexpr (KernelConversionMode == ConversionMode::ConvertAndScaleWithZero) {
|
||||
Tensor sZ = make_tensor(make_smem_ptr(shared_tensors.smem_zero.begin()), SmemLayoutScale{});// (BLK_M,BLK_SCALE_K,PIPE)
|
||||
Tensor tCsZ = mma_thread_slice.partition_A(sZ);
|
||||
Tensor tCrZ = make_tensor<ElementZero>(mma_thread_slice.partition_fragment_A(sZ(_,_,Int<0>{})).layout());
|
||||
Tensor tCrZ = make_tensor<ElementZero>(mma_thread_slice.partition_fragment_A(sZ(_,_,Int<0>{})).layout());
|
||||
return cute::make_tuple(tCsS, tCrS, tCsZ, tCrZ);
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled in A -> RF path.");
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled in A -> RF path.");
|
||||
}
|
||||
@@ -996,18 +1014,18 @@ public:
|
||||
auto smem_tiled_copy_S = make_tiled_copy_A(SmemCopyAtomScale{}, tiled_mma);
|
||||
auto smem_thr_copy_S = smem_tiled_copy_S.get_thread_slice(warp_group_thread_idx);
|
||||
Tensor tCrS_copy_view = smem_thr_copy_S.retile_D(cute::get<1>(partitioned_extra_info)); // (CPY,CPY_M,CPY_K)
|
||||
|
||||
|
||||
if constexpr (KernelConversionMode == ConversionMode::ConvertAndScale) {
|
||||
return cute::make_tuple(smem_tiled_copy_S, tCrS_copy_view);
|
||||
}
|
||||
}
|
||||
else if constexpr (KernelConversionMode == ConversionMode::ConvertAndScaleWithZero) {
|
||||
Tensor tCrZ_copy_view = smem_thr_copy_S.retile_D(cute::get<3>(partitioned_extra_info)); // (CPY,CPY_M,CPY_K)
|
||||
return cute::make_tuple(smem_tiled_copy_S, tCrS_copy_view, tCrZ_copy_view);
|
||||
}
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled in A -> RF path.");
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
static_assert(cutlass::detail::dependent_false<KernelSchedule>, "Conversion mode not handled in A -> RF path.");
|
||||
}
|
||||
|
||||
@@ -1519,6 +1519,105 @@ public:
|
||||
>::CollectiveOp;
|
||||
};
|
||||
|
||||
template <
|
||||
class MmaTileShape_MNK,
|
||||
class ClusterShape_MNK,
|
||||
class ElementAccumulator,
|
||||
class ElementCompute,
|
||||
class ElementC_,
|
||||
class GmemLayoutTagC_,
|
||||
int AlignmentC,
|
||||
class ElementD,
|
||||
class GmemLayoutTagD,
|
||||
int AlignmentD,
|
||||
class EpilogueScheduleType,
|
||||
class FusionOp
|
||||
>
|
||||
struct CollectiveBuilder<
|
||||
arch::Sm100,
|
||||
arch::OpClassSimt,
|
||||
MmaTileShape_MNK,
|
||||
ClusterShape_MNK,
|
||||
cutlass::epilogue::collective::EpilogueTileAuto,
|
||||
ElementAccumulator,
|
||||
ElementCompute,
|
||||
ElementC_,
|
||||
GmemLayoutTagC_,
|
||||
AlignmentC,
|
||||
ElementD,
|
||||
GmemLayoutTagD,
|
||||
AlignmentD,
|
||||
EpilogueScheduleType,
|
||||
FusionOp,
|
||||
cute::enable_if_t<
|
||||
cute::is_same_v<EpilogueScheduleType, EpilogueSimtVectorized> ||
|
||||
cute::is_same_v<EpilogueScheduleType, EpiloguePtrArraySimtVectorized> ||
|
||||
cute::is_same_v<EpilogueScheduleType, EpilogueScheduleAuto> >> {
|
||||
using CtaTileShape_MNK = MmaTileShape_MNK; // cluster MMA not supported
|
||||
|
||||
// Passing void C disables source load
|
||||
using ElementC = cute::conditional_t<cute::is_void_v<ElementC_>,
|
||||
ElementD, ElementC_>; // prevents void ref breakages
|
||||
using GmemLayoutTagC = cute::conditional_t<cute::is_void_v<ElementC_>,
|
||||
GmemLayoutTagD, GmemLayoutTagC_>;
|
||||
static constexpr thread::ScaleType::Kind ScaleType = cute::is_void_v<ElementC_> ?
|
||||
thread::ScaleType::OnlyAlphaScaling : thread::ScaleType::Default;
|
||||
|
||||
using GmemStrideTypeC = cutlass::detail::TagToStrideC_t<GmemLayoutTagC>;
|
||||
using GmemStrideTypeD = cutlass::detail::TagToStrideC_t<GmemLayoutTagD>;
|
||||
|
||||
using ThreadOp = cute::conditional_t<
|
||||
IsDefaultFusionOp<FusionOp>::value,
|
||||
thread::LinearCombination<
|
||||
ElementD, AlignmentD, ElementAccumulator, ElementCompute,
|
||||
ScaleType, FloatRoundStyle::round_to_nearest, ElementC>
|
||||
,
|
||||
thread::LinearCombinationBiasElementwise<
|
||||
ElementC, ElementAccumulator, ElementCompute, ElementD, ElementD, AlignmentD,
|
||||
typename FusionOp::ActivationFn, cutlass::plus<ElementCompute>,
|
||||
false, typename FusionOp::ElementBias>
|
||||
>;
|
||||
static_assert(not (cute::is_same_v<EpilogueScheduleType, EpiloguePtrArraySimtVectorized> && not IsDefaultFusionOp<FusionOp>::value), "unsupported schedule + fusion");
|
||||
|
||||
using WarpShape_MNK = decltype(cutlass::gemm::collective::detail::sm100_simt_f32_warp_shape_mnk_selector<CtaTileShape_MNK>());
|
||||
static constexpr int ThreadCount = cute::size(WarpShape_MNK{}) * NumThreadsPerWarp;
|
||||
static constexpr int WarpShape_M = cute::size<0>(WarpShape_MNK{});
|
||||
static constexpr int WarpShape_N = cute::size<1>(WarpShape_MNK{});
|
||||
|
||||
// For 32 threads in 1 warp, we use [8 x 4] thread layouts and each thread will hold [4 x 4] accumulator value layouts.
|
||||
// Then totally each warp will hold [32 x 16] accumulator value layouts.
|
||||
// We separate the whole epilogue calculation to multi steps,
|
||||
// each step will calculate 1x [32 x 16] for each warp to reduce register pressure (mainly for C register allocation for beta 1!= 0 case).
|
||||
// So EpiTileM = WarpShape_M * 32 and EpiTileN = WarpShape_N * 16.
|
||||
using EpiTileM = Int<WarpShape_M * 32>;
|
||||
using EpiTileN = Int<WarpShape_N * 16>;
|
||||
|
||||
using SmemLayout = cute::conditional_t<cutlass::detail::is_major<0>(GmemStrideTypeD{}),
|
||||
cute::Layout<cute::Shape<EpiTileM, EpiTileN>, cute::Stride<_1, EpiTileM>>,
|
||||
cute::Layout<cute::Shape<EpiTileM, EpiTileN>, cute::Stride<EpiTileN, _1>>>;
|
||||
|
||||
using CopyAtomR2S = Copy_Atom<cute::AutoVectorizingCopyWithAssumedAlignment<128>, ElementAccumulator>;
|
||||
|
||||
using CopyAtomS2R = Copy_Atom<cute::AutoVectorizingCopyWithAssumedAlignment<AlignmentD * sizeof_bits_v<ElementAccumulator>>, ElementAccumulator>;
|
||||
|
||||
using TiledCopyS2R = decltype(
|
||||
cutlass::gemm::collective::detail::make_simt_gmem_tiled_copy<
|
||||
CopyAtomS2R, ThreadCount, AlignmentD, GmemStrideTypeD, EpiTileM, EpiTileN>());
|
||||
|
||||
using Schedule = cute::conditional_t<is_same_v<EpilogueScheduleType, EpilogueScheduleAuto>,
|
||||
EpilogueSimtVectorized,
|
||||
EpilogueScheduleType>;
|
||||
using CopyAtomR2G = Copy_Atom<cute::AutoVectorizingCopyWithAssumedAlignment<AlignmentD * sizeof_bits_v<ElementD>>, ElementD>;
|
||||
using CollectiveOp = cutlass::epilogue::collective::Epilogue<
|
||||
GmemStrideTypeC,
|
||||
GmemStrideTypeD,
|
||||
ThreadOp,
|
||||
SmemLayout,
|
||||
CopyAtomR2S,
|
||||
TiledCopyS2R,
|
||||
CopyAtomR2G,
|
||||
Schedule>;
|
||||
};
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace cutlass::epilogue::collective
|
||||
|
||||
@@ -205,6 +205,16 @@ struct IsThreadEpilogueOpWithPerChannelScaling <ThreadEpilogueOp, cute::enable_i
|
||||
static constexpr bool value = true;
|
||||
};
|
||||
|
||||
template <typename ThreadEpilogueOp, typename = void>
|
||||
struct IsThreadEpilogueOpWithResidualAdd {
|
||||
static constexpr bool value = false;
|
||||
};
|
||||
|
||||
template <typename ThreadEpilogueOp>
|
||||
struct IsThreadEpilogueOpWithResidualAdd <ThreadEpilogueOp, cute::void_t<decltype(ThreadEpilogueOp::IsResidualSupported)>> {
|
||||
static constexpr bool value = ThreadEpilogueOp::IsResidualSupported;
|
||||
};
|
||||
|
||||
template <typename ThreadEpilogueOp, typename = void>
|
||||
struct IsThreadEpilogueOpWithActivation {
|
||||
static constexpr bool value = false;
|
||||
|
||||
@@ -39,6 +39,8 @@
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/epilogue/collective/detail.hpp"
|
||||
#include "cutlass/detail/helper_macros.hpp"
|
||||
#include "cutlass/conv/convnd_problem_shape.hpp"
|
||||
#include "cutlass/conv/detail.hpp"
|
||||
|
||||
#include "cute/tensor.hpp"
|
||||
#include "cute/numeric/numeric_types.hpp"
|
||||
@@ -133,6 +135,7 @@ public:
|
||||
constexpr static int ThreadCount = 128;
|
||||
constexpr static int kOutputAlignment = ThreadEpilogueOp::kCount;
|
||||
constexpr static bool isEpilogueBiasSupported = detail::IsThreadEpilogueOpWithBias<ThreadEpilogueOp>::value;
|
||||
constexpr static bool isSourceNeeded = not cute::is_void_v<ElementC>;
|
||||
|
||||
using AlignmentType = typename cute::uint_bit<sizeof_bits<ElementOutput>::value * kOutputAlignment>::type;
|
||||
constexpr static uint32_t TmaTransactionBytes = 0;
|
||||
@@ -173,12 +176,27 @@ public:
|
||||
return cutlass::Status::kSuccess;
|
||||
}
|
||||
|
||||
template <conv::Operator ConvOp, int NumDims>
|
||||
static bool
|
||||
can_implement(cutlass::conv::ConvProblemShape<ConvOp,NumDims> const& problem_shape, Arguments const& args) {
|
||||
return can_implement(cutlass::conv::detail::get_transformed_problem_shape_MNKL(problem_shape), args);
|
||||
}
|
||||
|
||||
template <class ProblemShape>
|
||||
static bool
|
||||
can_implement(
|
||||
[[maybe_unused]] ProblemShape const& problem_shape,
|
||||
[[maybe_unused]] Arguments const& args) {
|
||||
return true;
|
||||
auto problem_shape_MNKL = append<4>(problem_shape, 1);
|
||||
auto [M,N,K,L] = problem_shape_MNKL;
|
||||
auto shape = cute::make_shape(M,N,L);
|
||||
|
||||
bool implementable = true;
|
||||
implementable = implementable && cutlass::detail::check_alignment<AlignmentD{}>(shape, StrideD{});
|
||||
if constexpr (isSourceNeeded) {
|
||||
implementable = implementable && cutlass::detail::check_alignment<AlignmentC{}>(shape, StrideC{});
|
||||
}
|
||||
return implementable;
|
||||
}
|
||||
|
||||
//
|
||||
|
||||
@@ -57,6 +57,7 @@ struct FusionOperation {
|
||||
|
||||
using ElementSource = void;
|
||||
static constexpr bool IsSourceSupported = false;
|
||||
static constexpr bool IsResidualSupported = false; // Source is added after activation
|
||||
|
||||
using ElementScalar = void;
|
||||
static constexpr int AlignmentScalar = 0;
|
||||
@@ -317,6 +318,24 @@ struct PerColLinCombPerColBiasEltAct
|
||||
static constexpr bool IsPerColScaleSupported = true;
|
||||
};
|
||||
|
||||
// D = activation(per-col alpha * acc + per-column bias) + per-col beta * C
|
||||
template<
|
||||
template <class> class ActivationFn_,
|
||||
class ElementOutput_,
|
||||
class ElementCompute_,
|
||||
class ElementBias_ = ElementOutput_,
|
||||
class ElementSource_ = ElementOutput_,
|
||||
class ElementScalar_ = ElementCompute_, // per-row alpha/beta
|
||||
int AlignmentBias_ = 128 / cute::sizeof_bits_v<ElementBias_>,
|
||||
int AlignmentScalar_ = 128 / cute::sizeof_bits_v<ElementScalar_>,
|
||||
FloatRoundStyle RoundStyle_ = FloatRoundStyle::round_to_nearest
|
||||
>
|
||||
struct PerColResAddPerColBiasEltAct
|
||||
: PerColLinCombPerColBiasEltAct<ActivationFn_, ElementOutput_, ElementCompute_,
|
||||
ElementBias_, ElementSource_, ElementScalar_, AlignmentBias_, AlignmentScalar_, RoundStyle_> {
|
||||
static constexpr bool IsResidualSupported = true;
|
||||
};
|
||||
|
||||
// Z = scale_a * scale_b * alpha * acc + beta * scale_c * C + per-row bias
|
||||
// if D is fp8
|
||||
// D = scale_d * activation(Z)
|
||||
|
||||
@@ -1306,6 +1306,114 @@ struct FusionCallbacks<
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
// D = activation(per-col alpha * acc + per-column bias) + per-col beta * C
|
||||
template<
|
||||
class CtaTileShapeMNK,
|
||||
class EpilogueTile,
|
||||
template <class> class ActivationFn,
|
||||
class ElementOutput,
|
||||
class ElementCompute,
|
||||
class ElementBias = ElementOutput,
|
||||
class ElementSource = ElementOutput,
|
||||
class ElementScalar = ElementCompute,
|
||||
int AlignmentBias = 128 / sizeof_bits_v<ElementBias>,
|
||||
int AlignmentScalar = 128 / sizeof_bits_v<ElementScalar>,
|
||||
FloatRoundStyle RoundStyle = FloatRoundStyle::round_to_nearest
|
||||
>
|
||||
using Sm90PerColResAddPerColBiasEltAct =
|
||||
Sm90EVT<Sm90Compute<homogeneous_multiply_add, ElementOutput, ElementCompute, RoundStyle>, // beta * C + activation(alpha * acc + bias)
|
||||
Sm90RowBroadcast<0, CtaTileShapeMNK, ElementScalar, ElementCompute, Stride<_0,bool,int64_t>, AlignmentScalar>, // beta, dynamic scalar/vector broadcast
|
||||
Sm90SrcFetch<ElementSource>, // C
|
||||
Sm90EVT<Sm90Compute<ActivationFn, ElementCompute, ElementCompute, RoundStyle>, // activation(alpha * acc + bias)
|
||||
Sm90EVT<Sm90Compute<homogeneous_multiply_add, ElementCompute, ElementCompute, RoundStyle>, // alpha * acc + bias
|
||||
Sm90RowBroadcast<0, CtaTileShapeMNK, ElementScalar, ElementCompute, Stride<_0,bool,int64_t>, AlignmentScalar>, // alpha, dynamic scalar/vector broadcast
|
||||
Sm90AccFetch, // acc
|
||||
Sm90RowBroadcast<0, CtaTileShapeMNK, ElementBias, ElementCompute, Stride<_0,_1,int64_t>, AlignmentBias> // bias
|
||||
>
|
||||
>
|
||||
>;
|
||||
|
||||
template <
|
||||
int StagesC,
|
||||
int StagesD,
|
||||
int FragmentSize,
|
||||
bool ReuseSmemC,
|
||||
bool DelayTmaStore,
|
||||
template <class> class ActivationFn,
|
||||
class ElementOutput,
|
||||
class ElementCompute,
|
||||
class ElementBias,
|
||||
class ElementSource,
|
||||
class ElementScalar,
|
||||
int AlignmentBias,
|
||||
int AlignmentScalar,
|
||||
FloatRoundStyle RoundStyle,
|
||||
class CtaTileShapeMNK,
|
||||
class EpilogueTile
|
||||
>
|
||||
struct FusionCallbacks<
|
||||
epilogue::Sm90TmaWarpSpecialized<StagesC, StagesD, FragmentSize, ReuseSmemC, DelayTmaStore>,
|
||||
fusion::PerColResAddPerColBiasEltAct<
|
||||
ActivationFn, ElementOutput, ElementCompute, ElementBias, ElementSource, ElementScalar, AlignmentBias, AlignmentScalar, RoundStyle
|
||||
>,
|
||||
CtaTileShapeMNK,
|
||||
EpilogueTile
|
||||
> : Sm90PerColResAddPerColBiasEltAct<
|
||||
CtaTileShapeMNK, EpilogueTile, ActivationFn, ElementOutput, ElementCompute, ElementBias, ElementSource, ElementScalar, AlignmentBias, AlignmentScalar, RoundStyle
|
||||
> {
|
||||
|
||||
using Impl =
|
||||
Sm90PerColResAddPerColBiasEltAct<
|
||||
CtaTileShapeMNK, EpilogueTile, ActivationFn, ElementOutput, ElementCompute, ElementBias, ElementSource, ElementScalar, AlignmentBias, AlignmentScalar, RoundStyle
|
||||
>;
|
||||
using Operation =
|
||||
fusion::PerColResAddPerColBiasEltAct<
|
||||
ActivationFn, ElementOutput, ElementCompute, ElementBias, ElementSource, ElementScalar, AlignmentBias, AlignmentScalar, RoundStyle
|
||||
>;
|
||||
|
||||
struct Arguments {
|
||||
ElementScalar alpha = ElementScalar(1);
|
||||
ElementScalar beta = ElementScalar(0);
|
||||
ElementScalar const* alpha_ptr = nullptr;
|
||||
ElementScalar const* beta_ptr = nullptr;
|
||||
|
||||
using StrideAlpha = Stride<_0,bool,int64_t>;
|
||||
using StrideBeta = Stride<_0,bool,int64_t>;
|
||||
StrideAlpha dAlpha = {_0{}, bool(1), 0};
|
||||
StrideBeta dBeta = {_0{}, bool(1), 0};
|
||||
|
||||
using StrideBias = Stride<_0,_1,int64_t>;
|
||||
ElementBias const* bias_ptr = nullptr;
|
||||
StrideBias dBias = {};
|
||||
|
||||
using ActivationArguments = typename Sm90Compute<ActivationFn, ElementOutput, ElementCompute, RoundStyle>::Arguments;
|
||||
ActivationArguments activation = ActivationArguments();
|
||||
|
||||
operator typename Impl::Arguments() const {
|
||||
return
|
||||
{ // ternary op : beta * C + activation(alpha * acc + bias)
|
||||
{beta_ptr, beta, dBeta}, // leaf args : beta
|
||||
{}, // leaf args : C
|
||||
{ // unary op : activation(alpha * acc + bias)
|
||||
{ // ternary op : alpha * acc + bias
|
||||
{alpha_ptr, alpha, dAlpha}, // leaf args : alpha
|
||||
{}, // leaf args : acc
|
||||
{bias_ptr, ElementBias(0), dBias}, // leaf args : bias
|
||||
{} // ternary args : multiply_add
|
||||
}, // end ternary op
|
||||
activation // unary args : activation
|
||||
}, // end unary op
|
||||
{} // ternary args : multiply_add
|
||||
}; // end ternary op
|
||||
}
|
||||
};
|
||||
|
||||
// Ctor inheritance
|
||||
using Impl::Impl;
|
||||
};
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace detail {
|
||||
|
||||
template <typename T>
|
||||
|
||||
@@ -591,7 +591,7 @@ struct Sm90TreeVisitor<
|
||||
|
||||
auto [M, N, K, L] = args.problem_shape_mnkl;
|
||||
auto [m, n, k, l] = args.tile_coord_mnkl;
|
||||
gmem_ptr ptr_aux = make_gmem_ptr(subbyte_iterator<cutlass::uint1b_t>(params_aux.ptr_aux));
|
||||
gmem_ptr ptr_aux = make_gmem_ptr<cutlass::uint1b_t>(params_aux.ptr_aux);
|
||||
Tensor mAux = make_tensor(ptr_aux, make_layout(make_shape(M,N,L), params_aux.dAux)); // (M,N,L)
|
||||
Tensor gAux = local_tile(mAux, take<0,2>(args.tile_shape_mnk), make_coord(m,n,l)); // (CTA_M,CTA_N)
|
||||
|
||||
@@ -765,7 +765,7 @@ struct Sm90AuxLoad<
|
||||
|
||||
auto [M, N, K, L] = args.problem_shape_mnkl;
|
||||
auto [m, n, k, l] = args.tile_coord_mnkl;
|
||||
gmem_ptr ptr_aux = make_gmem_ptr(subbyte_iterator<cutlass::uint1b_t const>(params.ptr_aux));
|
||||
gmem_ptr ptr_aux = make_gmem_ptr<cutlass::uint1b_t const>(params.ptr_aux);
|
||||
Tensor mAux = make_tensor(ptr_aux, make_layout(make_shape(M,N,L), params.dAux)); // (M,N,L)
|
||||
Tensor gAux = local_tile(mAux, take<0,2>(args.tile_shape_mnk), make_coord(m,n,l)); // (CTA_M,CTA_N)
|
||||
|
||||
|
||||
@@ -1173,8 +1173,9 @@ public:
|
||||
CUTLASS_DEVICE auto
|
||||
get_consumer_store_callbacks(ConsumerStoreArgs<Args...> const& args) {
|
||||
Layout ref_layout_MN = [&] () {
|
||||
if constexpr (ReferenceSrc) { return get<0>(args.tiled_copy.get_layoutS_MN()); }
|
||||
else { return get<0>(args.tiled_copy.get_layoutD_MN()); }
|
||||
auto mn_shape = shape(typename decltype(args.tiled_copy)::Tiler_MN{});
|
||||
if constexpr (ReferenceSrc) { return right_inverse(args.tiled_copy.get_layoutS_TV()).with_shape(mn_shape); }
|
||||
else { return right_inverse(args.tiled_copy.get_layoutD_TV()).with_shape(mn_shape); }
|
||||
}(); // tile_mn -> tv_idx
|
||||
|
||||
// Get the MN layout + coord of lanes to determine shuffle reduction iterations
|
||||
@@ -1650,8 +1651,9 @@ public:
|
||||
CUTLASS_DEVICE auto
|
||||
get_consumer_store_callbacks(ConsumerStoreArgs<Args...> const& args) {
|
||||
Layout ref_layout_MN = [&] () {
|
||||
if constexpr (ReferenceSrc) { return get<0>(args.tiled_copy.get_layoutS_MN()); }
|
||||
else { return get<0>(args.tiled_copy.get_layoutD_MN()); }
|
||||
auto mn_shape = shape(typename decltype(args.tiled_copy)::Tiler_MN{});
|
||||
if constexpr (ReferenceSrc) { return right_inverse(args.tiled_copy.get_layoutS_TV()).with_shape(mn_shape); }
|
||||
else { return right_inverse(args.tiled_copy.get_layoutD_TV()).with_shape(mn_shape); }
|
||||
}(); // tile_mn -> tv_idx
|
||||
|
||||
// Get the MN layout + coord of lanes to determine shuffle reduction iterations
|
||||
|
||||
@@ -93,7 +93,7 @@ Array<float, 2> top_2_reduce(Array<float, 2> a, Array<float, 2> b) {
|
||||
" setp.gtu.f32 p, %2, %4;\n" // a0 > b0
|
||||
" selp.f32 %1, mx.x, mx.y, p;\n" // a0 > b0 ? max(a1, b0) : max(a0, b1)
|
||||
" selp.f32 %0, %2, %4, p;\n" // a0 > b0 ? a0 : b0
|
||||
"}\n" : "=f"(out[0]), "=f"(out[1]) :
|
||||
"}\n" : "=f"(out[0]), "=f"(out[1]) :
|
||||
"f"(a[0]), "f"(a[1]), "f"(b[0]), "f"(b[1]));
|
||||
return out;
|
||||
}
|
||||
@@ -117,8 +117,8 @@ Array<float, 4> top_4_reduce_scalar(Array<float, 4> a, float scalar) {
|
||||
" selp.f32 %1, %5, %8, p1;\n" // a0 = a1 > b ? a1 : b
|
||||
" selp.f32 %1, %1, %4, p0;\n" // a0 > b ? max(a1, b) : a0 == a0 > b ? a0 : old_a0
|
||||
" selp.f32 %0, %4, %8, p0;\n" // a0 = a0 > b ? a0 : b
|
||||
"}\n" :
|
||||
"=f"(out[0]), "=f"(out[1]), "=f"(out[2]), "=f"(out[3]) :
|
||||
"}\n" :
|
||||
"=f"(out[0]), "=f"(out[1]), "=f"(out[2]), "=f"(out[3]) :
|
||||
"f"(a[0]), "f"(a[1]), "f"(a[2]), "f"(a[3]), "f"(scalar));
|
||||
return out;
|
||||
}
|
||||
@@ -187,8 +187,8 @@ Array<float, 4> top_4_reduce(Array<float, 4> a, Array<float, 4> b) {
|
||||
" selp.f32 %3, mxa2b1, %3, pa1b1;\n" // a3 = a1 > b1 ? max(a2, b1) ** second most likely case
|
||||
" selp.f32 %3, mxa3b0, %3, pa2b0;\n" // a0 > a1 > a2 > b0
|
||||
" selp.f32 %3, mxa0b3, %3, pb2a0;\n" // b0 > b1 > b2 > a0
|
||||
"}\n" :
|
||||
"=f"(out[0]), "=f"(out[1]), "=f"(out[2]), "=f"(out[3]) :
|
||||
"}\n" :
|
||||
"=f"(out[0]), "=f"(out[1]), "=f"(out[2]), "=f"(out[3]) :
|
||||
"f"(a[0]), "f"(a[1]), "f"(a[2]), "f"(a[3]),
|
||||
"f"(b[0]), "f"(b[1]), "f"(b[2]), "f"(b[3]));
|
||||
return out;
|
||||
@@ -351,7 +351,7 @@ private:
|
||||
// we can track logsumexp instead of tracking two variables (sum of exps and the max).
|
||||
// In addition, subtracting logsumexp from any element and taking its exp is equivalent to
|
||||
// computing its softmax.
|
||||
//
|
||||
//
|
||||
// The overlap between softmax and top-K is that we don't need to reduce logsumexp along the
|
||||
// way at all, because any element not in the top-K is going to be masked out and set to 0.
|
||||
// Therefore, we only reduce the top-K elements, and when done, compute their logsumexp and
|
||||
@@ -370,7 +370,7 @@ private:
|
||||
ReductionResult() { }
|
||||
|
||||
CUTLASS_DEVICE
|
||||
ReductionResult(ElementCompute min, ElementCompute logsumexp):
|
||||
ReductionResult(ElementCompute min, ElementCompute logsumexp):
|
||||
logsumexp_(logsumexp), min_(min) { }
|
||||
|
||||
// Warp shuffle broadcast
|
||||
@@ -541,7 +541,7 @@ public:
|
||||
visit(Array<ElementAccumulator, FragmentSize> const& frg_acc, int epi_v, int epi_m, int epi_n,
|
||||
Array<ElementInput, FragmentSize> const& frg_input) {
|
||||
|
||||
auto& [tCrTopK, tCrSoftmax, tCcCol, cCol,
|
||||
auto& [tCrTopK, tCrSoftmax, tCcCol, cCol,
|
||||
lane_layout_MN, lane_mn,
|
||||
residue_cCol, residue_tCcCol] = args_tuple;
|
||||
Tensor tCcCol_mn = tCcCol(_,_,_,epi_m,epi_n);
|
||||
@@ -566,7 +566,7 @@ public:
|
||||
CUTLASS_DEVICE void
|
||||
reduce(STensor&& smem_buffer, SyncFn const& sync_fn, int epi_m, int epi_n, bool is_last_iteration, VTensor visit_results) {
|
||||
|
||||
auto& [tCrTopK, tCrSoftmax, tCcCol, cCol,
|
||||
auto& [tCrTopK, tCrSoftmax, tCcCol, cCol,
|
||||
lane_layout_MN, lane_mn,
|
||||
residue_cCol, residue_tCcCol] = args_tuple;
|
||||
|
||||
@@ -668,7 +668,7 @@ public:
|
||||
|
||||
CUTLASS_DEVICE void
|
||||
end_loop(int epi_m, int epi_n) {
|
||||
auto& [tCrTopK, tCrSoftmax, tCcCol, cCol,
|
||||
auto& [tCrTopK, tCrSoftmax, tCcCol, cCol,
|
||||
lane_layout_MN, lane_mn,
|
||||
residue_cCol, residue_tCcCol] = args_tuple;
|
||||
|
||||
@@ -690,8 +690,9 @@ public:
|
||||
CUTLASS_DEVICE auto
|
||||
get_consumer_store_callbacks(ConsumerStoreArgs<Args...> const& args) {
|
||||
Layout ref_layout_MN = [&] () {
|
||||
if constexpr (ReferenceSrc) { return get<0>(args.tiled_copy.get_layoutS_MN()); }
|
||||
else { return get<0>(args.tiled_copy.get_layoutD_MN()); }
|
||||
auto mn_shape = shape(typename decltype(args.tiled_copy)::Tiler_MN{});
|
||||
if constexpr (ReferenceSrc) { return right_inverse(args.tiled_copy.get_layoutS_TV()).with_shape(mn_shape); }
|
||||
else { return right_inverse(args.tiled_copy.get_layoutD_TV()).with_shape(mn_shape); }
|
||||
}(); // tile_mn -> tv_idx
|
||||
|
||||
// Get the MN layout + coord of lanes to determine shuffle reduction iterations
|
||||
@@ -739,7 +740,7 @@ public:
|
||||
Tensor tRS_rSoftmax = thread_r2s.retile_S(tCrColReduce); // ((R2S,R2S_V),MMA_M,MMA_N)
|
||||
auto tCrC_layout = args.tCrC.layout(); // (R2S,R2S_M,R2S_N)
|
||||
|
||||
// Compose the new accumulator R2S layout with the expected tCrC layout to get final
|
||||
// Compose the new accumulator R2S layout with the expected tCrC layout to get final
|
||||
// reduction tensor layout.
|
||||
auto tCrSoftmax_layout = take<0, 3>(tRS_rSoftmax.layout()).compose(tCrC_layout); // (R2S,R2S_V) o (R2S,R2S_M,R2S_N)
|
||||
|
||||
|
||||
@@ -569,6 +569,47 @@ sm100_make_trivial_fastFP32_tiled_mma() {
|
||||
}
|
||||
}
|
||||
|
||||
template<
|
||||
class CtaShape_MNK
|
||||
>
|
||||
constexpr auto
|
||||
sm100_simt_f32_warp_shape_mnk_selector() {
|
||||
using namespace cute;
|
||||
|
||||
constexpr int CtaShape_M = cute::size<0>(CtaShape_MNK{});
|
||||
constexpr int CtaShape_N = cute::size<1>(CtaShape_MNK{});
|
||||
constexpr int CtaShape_K = cute::size<2>(CtaShape_MNK{});
|
||||
|
||||
// CTA tile shape M and N are supposed to be divisible by 32.
|
||||
static_assert(CtaShape_M % 32 == 0, "CtaShape_M needs to be divisible by 32.");
|
||||
static_assert(CtaShape_N % 32 == 0, "CtaShape_N needs to be divisible by 32.");
|
||||
|
||||
// WarpShape_MNK configuration
|
||||
// We assume WarpShape_K is always 1 in our SM100 SIMT SGEMM implementation.
|
||||
if constexpr (CtaShape_M >= CtaShape_N) {
|
||||
if constexpr (CtaShape_M == 256 && CtaShape_N == 128) {
|
||||
return cute::Shape<_4, _2, _1>{};
|
||||
}
|
||||
else if constexpr ((CtaShape_M == 64 || CtaShape_M == 32) && CtaShape_N == 32) {
|
||||
return cute::Shape<_1, _2, _1>{};
|
||||
}
|
||||
else {
|
||||
return cute::Shape<_2, _2, _1>{};
|
||||
}
|
||||
}
|
||||
else {
|
||||
if constexpr (CtaShape_M == 128 && CtaShape_N == 256) {
|
||||
return cute::Shape<_2, _4, _1>{};
|
||||
}
|
||||
else if constexpr (CtaShape_M == 32 && CtaShape_N == 64) {
|
||||
return cute::Shape<_1, _2, _1>{};
|
||||
}
|
||||
else {
|
||||
return cute::Shape<_1, _4, _1>{};
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
template <
|
||||
class ElementPairA,
|
||||
|
||||
@@ -0,0 +1,216 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2023 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/gemm/collective/builders/sm100_common.inl"
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace cutlass::gemm::collective {
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace detail {
|
||||
|
||||
template<
|
||||
class LayoutA,
|
||||
int AlignmentA,
|
||||
class LayoutB,
|
||||
int AlignmentB,
|
||||
class CtaShape_MNK,
|
||||
class WarpShape_MNK
|
||||
>
|
||||
constexpr auto
|
||||
sm100_make_simt_f32_tiled_mma() {
|
||||
using namespace cute;
|
||||
|
||||
constexpr int CtaShape_M = cute::size<0>(CtaShape_MNK{});
|
||||
constexpr int CtaShape_N = cute::size<1>(CtaShape_MNK{});
|
||||
constexpr int CtaShape_K = cute::size<2>(CtaShape_MNK{});
|
||||
|
||||
constexpr int WarpShape_M = cute::size<0>(WarpShape_MNK{});
|
||||
constexpr int WarpShape_N = cute::size<1>(WarpShape_MNK{});
|
||||
constexpr int WarpShape_K = cute::size<2>(WarpShape_MNK{});
|
||||
|
||||
// Use Permutation to achieve a [4 x 4] value layout for each thread.
|
||||
// Ideally, we want the tiled mma to be such that loads from shared memory are 128 bit wide.
|
||||
// While as we are using CtaShape_K = 16, when A and B are K-major, we use tranpose + 8 byte padding to avoid smem bank conflict,
|
||||
// so we could only use 64 bit smem load.
|
||||
// When A and B are MN-major, we use 128 bit smem load.
|
||||
using PermutationA = Layout<Shape<_2, Int<WarpShape_M * 8>, _2>, Stride< _1, _4, _2>>;
|
||||
using PermutationB = Layout<Shape<Int<WarpShape_N * 4>, _4>, Stride< _4, _1>>;
|
||||
|
||||
// For 32 threads in 1 warp, we use [8 x 4] thread layouts and each thread will hold [4 x 4] value layouts.
|
||||
// Then totally each warp will hold [32 x 16] value layouts.
|
||||
// So WarpShape_M needs to be equal or smaller than CtaShape_M / 32 and WarpShape_N needs to be equal or smaller than CtaShape_N / 16.
|
||||
static_assert(WarpShape_M <= CtaShape_M / 32, "WarpShape_M is too large, it needs to be equal or smaller than CtaShape_M / 32.");
|
||||
static_assert(WarpShape_N <= CtaShape_N / 16, "WarpShape_N is too large, it needs to be equal or smaller than CtaShape_N / 16.");
|
||||
|
||||
constexpr int WarpStride_M = (WarpShape_M != 1) * NumThreadsPerWarp;
|
||||
constexpr int WarpStride_N = WarpShape_M * NumThreadsPerWarp;
|
||||
|
||||
// We first introduce a [8 x 4] thread layouts in 1 warp.
|
||||
// And inside this [8 x 4] thread layouts, each 4 threads will be arranged as [2 x 2].
|
||||
// Then we could set different WarpShape to finalize how many warps we use in our tiled mma.
|
||||
// For example :
|
||||
// With 128 threads in the tiled mma, we could set the WarpShapeMNK as [2 x 2 x 1], [1 x 4 x 1] and [4 x 1 x 1].
|
||||
// With 64 threads in the tiled mma, we could set the WarpShapeMNK as [1 x 2 x 1] and [2 x 1 x 1].
|
||||
return make_tiled_mma(
|
||||
MMA_Atom<SM100_2x1x1_F32F32F32F32>{},
|
||||
Layout<Shape < Shape <_2, _4, Int<WarpShape_M>>, Shape <_2, _2, Int<WarpShape_N>>, _1>,
|
||||
Stride< Stride<_1, _8, Int<WarpStride_M>>, Stride<_2, _4, Int<WarpStride_N>>, _1>>{},
|
||||
Tile<
|
||||
PermutationA,
|
||||
PermutationB,
|
||||
Underscore>{});
|
||||
}
|
||||
|
||||
} // namespace detail
|
||||
|
||||
template <
|
||||
class GmemLayoutATag,
|
||||
int AlignmentA,
|
||||
class GmemLayoutBTag,
|
||||
int AlignmentB,
|
||||
class CtaShape_MNK,
|
||||
class ClusterShape_MNK,
|
||||
int stages,
|
||||
class BuilderScheduleTag>
|
||||
struct CollectiveBuilder<
|
||||
arch::Sm100,
|
||||
arch::OpClassSimt,
|
||||
float,
|
||||
GmemLayoutATag,
|
||||
AlignmentA,
|
||||
float,
|
||||
GmemLayoutBTag,
|
||||
AlignmentB,
|
||||
float,
|
||||
CtaShape_MNK,
|
||||
ClusterShape_MNK,
|
||||
StageCount<stages>,
|
||||
BuilderScheduleTag,
|
||||
cute::enable_if_t<
|
||||
(cute::is_same_v<BuilderScheduleTag, KernelMultistage> ||
|
||||
cute::is_same_v<BuilderScheduleTag, KernelPtrArrayMultistage> ||
|
||||
cute::is_same_v<BuilderScheduleTag, KernelScheduleAuto>) &&
|
||||
((sizeof(float) * AlignmentA) % detail::cp_async_min_alignment_bytes == 0) &&
|
||||
((sizeof(float) * AlignmentB) % detail::cp_async_min_alignment_bytes == 0) >> {
|
||||
static_assert(cute::size<2>(CtaShape_MNK{}) == 16, "SM100 SIMT SGEMM Kernels only support TileShape_K = 16.");
|
||||
|
||||
// This kernel is specialized for F32 data type.
|
||||
using ElementA = float;
|
||||
using ElementB = float;
|
||||
|
||||
using M = decltype(cute::size<0>(CtaShape_MNK{}));
|
||||
using N = decltype(cute::size<1>(CtaShape_MNK{}));
|
||||
using K = decltype(cute::size<2>(CtaShape_MNK{}));
|
||||
|
||||
using WarpShape_MNK = decltype(detail::sm100_simt_f32_warp_shape_mnk_selector<CtaShape_MNK>());
|
||||
|
||||
static constexpr int ThreadCount = cute::size(WarpShape_MNK{}) * NumThreadsPerWarp;
|
||||
|
||||
using TiledMma = decltype(
|
||||
detail::sm100_make_simt_f32_tiled_mma<
|
||||
GmemLayoutATag,
|
||||
AlignmentA,
|
||||
GmemLayoutBTag,
|
||||
AlignmentB,
|
||||
CtaShape_MNK,
|
||||
WarpShape_MNK>());
|
||||
|
||||
// for K major layouts, add a smem alignment offset to avoid bank conflicts
|
||||
static constexpr int SmemAlignmentOffsetA = cutlass::gemm::detail::is_mn_major_A<GmemLayoutATag>() ? 0 : 2;
|
||||
static constexpr int SmemAlignmentOffsetB = cutlass::gemm::detail::is_mn_major_B<GmemLayoutBTag>() ? 0 : 2;
|
||||
static constexpr int CtaShape_M = cute::size<0>(CtaShape_MNK{});
|
||||
static constexpr int CtaShape_N = cute::size<1>(CtaShape_MNK{});
|
||||
|
||||
// Shared memory layout is [M x K] in M-major
|
||||
using SmemLayoutAtomA = cute::Layout<cute::Shape< M, K>,
|
||||
cute::Stride<_1, Int<CtaShape_M + SmemAlignmentOffsetA>>>;
|
||||
// A M-major use 128bit smem load.
|
||||
// A K-major needs to do tranpose and 8 byte padding to make smem bank conflict free, then we can only use 64bit smem load.
|
||||
using SmemCopyAtomA = std::conditional_t<cutlass::gemm::detail::is_mn_major_A<GmemLayoutATag>(),
|
||||
cute::Copy_Atom<cute::AutoVectorizingCopyWithAssumedAlignment<128>, ElementA>,
|
||||
cute::Copy_Atom<cute::AutoVectorizingCopyWithAssumedAlignment<64>, ElementA>>;
|
||||
|
||||
using AlignmentTypeA = cute::uint_byte_t<static_cast<int>(sizeof(ElementA)) * AlignmentA>;
|
||||
using GmemCopyAtomA = cute::Copy_Atom<SM80_CP_ASYNC_CACHEALWAYS_ZFILL<AlignmentTypeA>, ElementA>;
|
||||
using GmemTiledCopyA = decltype(
|
||||
detail::make_simt_gmem_tiled_copy<
|
||||
GmemCopyAtomA, ThreadCount, AlignmentA, TagToStrideA_t<GmemLayoutATag>, M, K>());
|
||||
|
||||
// Shared memory layout is [N x K] in N-major
|
||||
using SmemLayoutAtomB = cute::Layout<cute::Shape< N, K>,
|
||||
cute::Stride<_1, Int<CtaShape_N + SmemAlignmentOffsetB>>>;
|
||||
// B N-major use 128bit smem load.
|
||||
// B K-major needs to do tranpose and 8 byte padding to make smem bank conflict free, then we can only use 64bit smem load.
|
||||
using SmemCopyAtomB = std::conditional_t<cutlass::gemm::detail::is_mn_major_B<GmemLayoutBTag>(),
|
||||
cute::Copy_Atom<cute::AutoVectorizingCopyWithAssumedAlignment<128>, ElementB>,
|
||||
cute::Copy_Atom<cute::AutoVectorizingCopyWithAssumedAlignment<64>, ElementB>>;
|
||||
|
||||
using AlignmentTypeB = cute::uint_byte_t<static_cast<int>(sizeof(ElementB)) * AlignmentB>;
|
||||
using GmemCopyAtomB = cute::Copy_Atom<SM80_CP_ASYNC_CACHEALWAYS_ZFILL<AlignmentTypeB>, ElementB>;
|
||||
using GmemTiledCopyB = decltype(
|
||||
detail::make_simt_gmem_tiled_copy<
|
||||
GmemCopyAtomB, ThreadCount, AlignmentB, TagToStrideB_t<GmemLayoutBTag>, N, K>());
|
||||
|
||||
static constexpr bool IsArrayOfPointersGemm = cute::is_same_v<BuilderScheduleTag, KernelPtrArrayMultistage>;
|
||||
using DispatchPolicy = cute::conditional_t<IsArrayOfPointersGemm,
|
||||
cutlass::gemm::MainloopSm80ArrayCpAsync<stages,
|
||||
ClusterShape_MNK>,
|
||||
cutlass::gemm::MainloopSm80CpAsync<stages,
|
||||
ClusterShape_MNK>
|
||||
>;
|
||||
|
||||
using CollectiveOp = cutlass::gemm::collective::CollectiveMma<
|
||||
DispatchPolicy,
|
||||
CtaShape_MNK,
|
||||
ElementA,
|
||||
TagToStrideA_t<GmemLayoutATag>,
|
||||
ElementB,
|
||||
TagToStrideB_t<GmemLayoutBTag>,
|
||||
TiledMma,
|
||||
GmemTiledCopyA,
|
||||
SmemLayoutAtomA,
|
||||
SmemCopyAtomA,
|
||||
cute::identity,
|
||||
GmemTiledCopyB,
|
||||
SmemLayoutAtomB,
|
||||
SmemCopyAtomB,
|
||||
cute::identity
|
||||
>;
|
||||
};
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace cutlass::gemm::collective
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
@@ -46,6 +46,7 @@
|
||||
#include "cutlass/gemm/collective/builders/sm100_blockscaled_umma_builder.inl"
|
||||
#include "cutlass/gemm/collective/builders/sm100_blockwise_umma_builder.inl"
|
||||
#include "cutlass/gemm/collective/builders/sm100_blockscaled_sparse_umma_builder.inl"
|
||||
#include "cutlass/gemm/collective/builders/sm100_simt_builder.inl"
|
||||
#include "cutlass/gemm/collective/builders/sm120_mma_builder.inl"
|
||||
#include "cutlass/gemm/collective/builders/sm120_blockscaled_mma_builder.inl"
|
||||
#include "cutlass/gemm/collective/builders/sm120_sparse_mma_builder.inl"
|
||||
|
||||
@@ -37,6 +37,7 @@
|
||||
|
||||
#include "cutlass/gemm/collective/sm70_mma_twostage.hpp"
|
||||
#include "cutlass/gemm/collective/sm80_mma_multistage.hpp"
|
||||
#include "cutlass/gemm/collective/sm80_mma_array_multistage.hpp"
|
||||
#include "cutlass/gemm/collective/sm90_mma_multistage_gmma_ss_warpspecialized.hpp"
|
||||
#include "cutlass/gemm/collective/sm90_mma_multistage_gmma_rs_warpspecialized.hpp"
|
||||
#include "cutlass/gemm/collective/sm90_mma_tma_gmma_ss.hpp"
|
||||
|
||||
@@ -0,0 +1,412 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2023 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/gemm/dispatch_policy.hpp"
|
||||
|
||||
#include "cute/algorithm/functional.hpp"
|
||||
#include "cute/atom/mma_atom.hpp"
|
||||
#include "cute/algorithm/gemm.hpp"
|
||||
#include "cute/numeric/arithmetic_tuple.hpp"
|
||||
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
namespace cutlass::gemm::collective {
|
||||
using namespace cute;
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template <
|
||||
int Stages,
|
||||
class ClusterShape_,
|
||||
class TileShape_,
|
||||
class ElementA_,
|
||||
class StrideA_,
|
||||
class ElementB_,
|
||||
class StrideB_,
|
||||
class TiledMma_,
|
||||
class GmemTiledCopyA_,
|
||||
class SmemLayoutAtomA_,
|
||||
class SmemCopyAtomA_,
|
||||
class TransformA_,
|
||||
class GmemTiledCopyB_,
|
||||
class SmemLayoutAtomB_,
|
||||
class SmemCopyAtomB_,
|
||||
class TransformB_
|
||||
>
|
||||
struct CollectiveMma<
|
||||
MainloopSm80ArrayCpAsync<
|
||||
Stages,
|
||||
ClusterShape_>,
|
||||
TileShape_,
|
||||
ElementA_,
|
||||
StrideA_,
|
||||
ElementB_,
|
||||
StrideB_,
|
||||
TiledMma_,
|
||||
GmemTiledCopyA_,
|
||||
SmemLayoutAtomA_,
|
||||
SmemCopyAtomA_,
|
||||
TransformA_,
|
||||
GmemTiledCopyB_,
|
||||
SmemLayoutAtomB_,
|
||||
SmemCopyAtomB_,
|
||||
TransformB_
|
||||
>
|
||||
{
|
||||
//
|
||||
// Type Aliases
|
||||
//
|
||||
using DispatchPolicy = MainloopSm80ArrayCpAsync<
|
||||
Stages,
|
||||
ClusterShape_>;
|
||||
using TileShape = TileShape_;
|
||||
// Follow the change in TestSmall: TileShape switch to CtaShape
|
||||
// In legacy arch, it should be same
|
||||
using CtaShape_MNK = TileShape;
|
||||
using ElementA = ElementA_;
|
||||
using StrideA = StrideA_;
|
||||
using InternalStrideA = cute::remove_pointer_t<StrideA>;
|
||||
using ElementB = ElementB_;
|
||||
using StrideB = StrideB_;
|
||||
using InternalStrideB = cute::remove_pointer_t<StrideB>;
|
||||
using TiledMma = TiledMma_;
|
||||
using ElementAccumulator = typename TiledMma::ValTypeC; using GmemTiledCopyA = GmemTiledCopyA_;
|
||||
using GmemTiledCopyB = GmemTiledCopyB_;
|
||||
using SmemLayoutAtomA = SmemLayoutAtomA_;
|
||||
using SmemLayoutAtomB = SmemLayoutAtomB_;
|
||||
using SmemCopyAtomA = SmemCopyAtomA_;
|
||||
using SmemCopyAtomB = SmemCopyAtomB_;
|
||||
using TransformA = TransformA_;
|
||||
using TransformB = TransformB_;
|
||||
using ArchTag = typename DispatchPolicy::ArchTag;
|
||||
using ArrayElementA = ElementA;
|
||||
using ArrayElementB = ElementB;
|
||||
static_assert(cute::rank(SmemLayoutAtomA{}) == 2, "SmemLayoutAtom must be rank 2 (M/N, K)");
|
||||
static_assert((size<0>(TileShape{}) % size<0>(SmemLayoutAtomA{})) == 0, "SmemLayoutAtom must evenly divide tile shape.");
|
||||
static_assert((size<2>(TileShape{}) % size<1>(SmemLayoutAtomA{})) == 0, "SmemLayoutAtom must evenly divide tile shape.");
|
||||
|
||||
static_assert(cute::rank(SmemLayoutAtomB{}) == 2, "SmemLayoutAtom must be rank 2 (M/N, K)");
|
||||
static_assert((size<1>(TileShape{}) % size<0>(SmemLayoutAtomB{})) == 0, "SmemLayoutAtom must evenly divide tile shape.");
|
||||
static_assert((size<2>(TileShape{}) % size<1>(SmemLayoutAtomB{})) == 0, "SmemLayoutAtom must evenly divide tile shape.");
|
||||
|
||||
using SmemLayoutA = decltype(tile_to_shape(
|
||||
SmemLayoutAtomA{},
|
||||
make_shape(shape<0>(TileShape{}), shape<2>(TileShape{}), Int<DispatchPolicy::Stages>{})));
|
||||
using SmemLayoutB = decltype(tile_to_shape(
|
||||
SmemLayoutAtomB{},
|
||||
make_shape(shape<1>(TileShape{}), shape<2>(TileShape{}), Int<DispatchPolicy::Stages>{})));
|
||||
|
||||
static_assert(DispatchPolicy::Stages >= 2, "CpAsync mainloop must have at least 2 stages in the pipeline.");
|
||||
|
||||
struct SharedStorage
|
||||
{
|
||||
cute::array_aligned<ElementA, cute::cosize_v<SmemLayoutA>> smem_a;
|
||||
cute::array_aligned<ElementB, cute::cosize_v<SmemLayoutB>> smem_b;
|
||||
};
|
||||
|
||||
// Host side kernel arguments
|
||||
struct Arguments {
|
||||
ElementA const** ptr_A{nullptr};
|
||||
StrideA dA{};
|
||||
ElementB const** ptr_B{nullptr};
|
||||
StrideB dB{};
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
using Params = Arguments;
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
CollectiveMma() = default;
|
||||
|
||||
template <class ProblemShape>
|
||||
static constexpr Params
|
||||
to_underlying_arguments(ProblemShape const& _, Arguments const& args, void* workspace) {
|
||||
(void) workspace;
|
||||
return args;
|
||||
}
|
||||
|
||||
/// Perform a collective-scoped matrix multiply-accumulate
|
||||
template <
|
||||
class FrgTensorD,
|
||||
class TensorA,
|
||||
class TensorB,
|
||||
class FrgTensorC,
|
||||
class KTileIterator,
|
||||
class ResidueMNK
|
||||
>
|
||||
CUTLASS_DEVICE void
|
||||
operator() (
|
||||
FrgTensorD &accum,
|
||||
TensorA gA, // (BLK_M, BLK_K, K_TILES)
|
||||
TensorB gB, // (BLK_N, BLK_K, K_TILES)
|
||||
FrgTensorC const &src_accum,
|
||||
KTileIterator k_tile_iter, int k_tile_count,
|
||||
ResidueMNK residue_mnk,
|
||||
int thread_idx,
|
||||
char *smem_buf)
|
||||
{
|
||||
using namespace cute;
|
||||
|
||||
static_assert(is_rmem<FrgTensorD>::value, "D tensor must be rmem resident.");
|
||||
static_assert(is_gmem<TensorA>::value, "A tensor must be gmem resident.");
|
||||
static_assert(is_gmem<TensorB>::value, "B tensor must be gmem resident.");
|
||||
static_assert(is_rmem<FrgTensorC>::value, "C tensor must be rmem resident.");
|
||||
static_assert(cute::rank(SmemLayoutA{}) == 3, "Smem layout must be rank 3.");
|
||||
static_assert(cute::rank(SmemLayoutB{}) == 3, "Smem layout must be rank 3.");
|
||||
|
||||
// Construct shared memory tiles
|
||||
SharedStorage& storage = *reinterpret_cast<SharedStorage*>(smem_buf);
|
||||
Tensor sA = make_tensor(make_smem_ptr(storage.smem_a.data()), SmemLayoutA{}); // (BLK_M,BLK_K,PIPE)
|
||||
Tensor sB = make_tensor(make_smem_ptr(storage.smem_b.data()), SmemLayoutB{}); // (BLK_N,BLK_K,PIPE)
|
||||
|
||||
CUTE_STATIC_ASSERT_V(size<0>(gA) == size<0>(sA)); // BLK_M
|
||||
CUTE_STATIC_ASSERT_V(size<1>(gA) == size<1>(sA)); // BLK_K
|
||||
CUTE_STATIC_ASSERT_V(size<0>(gB) == size<0>(sB)); // BLK_N
|
||||
CUTE_STATIC_ASSERT_V(size<1>(gB) == size<1>(sB)); // BLK_K
|
||||
CUTE_STATIC_ASSERT_V(size<1>(sA) == size<1>(sB)); // BLK_K
|
||||
CUTE_STATIC_ASSERT_V(Int<DispatchPolicy::Stages>{} == size<2>(sA)); // PIPE
|
||||
CUTE_STATIC_ASSERT_V(Int<DispatchPolicy::Stages>{} == size<2>(sB)); // PIPE
|
||||
|
||||
// Shift tensor so residue_k is at origin (Can't read any k_coord < residue_k)
|
||||
// This aligns the tensor with BLK_K for all but the 0th k_tile
|
||||
gA = cute::domain_offset(make_coord(0, get<2>(residue_mnk), 0), gA);
|
||||
gB = cute::domain_offset(make_coord(0, get<2>(residue_mnk), 0), gB);
|
||||
|
||||
// Partition the copying of A and B tiles across the threads
|
||||
GmemTiledCopyA gmem_tiled_copy_A;
|
||||
GmemTiledCopyB gmem_tiled_copy_B;
|
||||
auto gmem_thr_copy_A = gmem_tiled_copy_A.get_slice(thread_idx);
|
||||
auto gmem_thr_copy_B = gmem_tiled_copy_B.get_slice(thread_idx);
|
||||
|
||||
Tensor tAgA = gmem_thr_copy_A.partition_S(gA); // (ACPY,ACPY_M,ACPY_K,k)
|
||||
Tensor tAsA = gmem_thr_copy_A.partition_D(sA); // (ACPY,ACPY_M,ACPY_K,PIPE)
|
||||
Tensor tBgB = gmem_thr_copy_B.partition_S(gB); // (BCPY,BCPY_N,BCPY_K,k)
|
||||
Tensor tBsB = gmem_thr_copy_B.partition_D(sB); // (BCPY,BCPY_N,BCPY_K,PIPE)
|
||||
|
||||
//
|
||||
// PREDICATES
|
||||
//
|
||||
|
||||
// Allocate predicate tensors for m and n
|
||||
Tensor tApA = make_tensor<bool>(make_shape(size<1>(tAsA), size<2>(tAsA)), Stride<_1,_0>{});
|
||||
Tensor tBpB = make_tensor<bool>(make_shape(size<1>(tBsB), size<2>(tBsB)), Stride<_1,_0>{});
|
||||
|
||||
// Construct identity layout for sA and sB
|
||||
Tensor cA = make_identity_tensor(make_shape(size<0>(sA), size<1>(sA))); // (BLK_M,BLK_K) -> (blk_m,blk_k)
|
||||
Tensor cB = make_identity_tensor(make_shape(size<0>(sB), size<1>(sB))); // (BLK_N,BLK_K) -> (blk_n,blk_k)
|
||||
|
||||
// Repeat the partitioning with identity layouts
|
||||
Tensor tAcA = gmem_thr_copy_A.partition_S(cA); // (ACPY,ACPY_M,ACPY_K) -> (blk_m,blk_k)
|
||||
Tensor tBcB = gmem_thr_copy_B.partition_S(cB); // (BCPY,BCPY_N,BCPY_K) -> (blk_n,blk_k)
|
||||
|
||||
// Set predicates for m bounds
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int m = 0; m < size<0>(tApA); ++m) {
|
||||
tApA(m,0) = get<0>(tAcA(0,m,0)) < get<0>(residue_mnk); // blk_m coord < residue_m
|
||||
}
|
||||
// Set predicates for n bounds
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int n = 0; n < size<0>(tBpB); ++n) {
|
||||
tBpB(n,0) = get<0>(tBcB(0,n,0)) < get<1>(residue_mnk); // blk_n coord < residue_n
|
||||
}
|
||||
|
||||
//
|
||||
// PREFETCH
|
||||
//
|
||||
|
||||
// Clear the smem tiles to account for predicated off loads
|
||||
clear(tAsA);
|
||||
clear(tBsB);
|
||||
|
||||
// Start async loads for 0th k-tile, where we take care of the k residue
|
||||
{
|
||||
constexpr int k_pipe = 0;
|
||||
|
||||
Tensor tAgAk = tAgA(_,_,_,*k_tile_iter);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k = 0; k < size<2>(tAsA); ++k) {
|
||||
if (get<1>(tAcA(0,0,k)) >= -get<2>(residue_mnk)) { // blk_k coord < residue_k (gA shifted)
|
||||
copy_if(gmem_tiled_copy_A, tApA(_,k), tAgAk(_,_,k), tAsA(_,_,k,k_pipe));
|
||||
}
|
||||
}
|
||||
Tensor tBgBk = tBgB(_,_,_,*k_tile_iter);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k = 0; k < size<2>(tBsB); ++k) {
|
||||
if (get<1>(tBcB(0,0,k)) >= -get<2>(residue_mnk)) { // blk_k coord < residue_k (gB shifted)
|
||||
copy_if(gmem_tiled_copy_B, tBpB(_,k), tBgBk(_,_,k), tBsB(_,_,k,k_pipe));
|
||||
}
|
||||
}
|
||||
cp_async_fence();
|
||||
++k_tile_iter;
|
||||
--k_tile_count;
|
||||
}
|
||||
|
||||
// Start async loads for 1st k-tile onwards, no k-residue handling needed
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int k_pipe = 1; k_pipe < DispatchPolicy::Stages-1; ++k_pipe) {
|
||||
if (k_tile_count <= 0) {
|
||||
clear(tApA);
|
||||
clear(tBpB);
|
||||
}
|
||||
copy_if(gmem_tiled_copy_A, tApA, tAgA(_,_,_,*k_tile_iter), tAsA(_,_,_,k_pipe)); // CpAsync
|
||||
copy_if(gmem_tiled_copy_B, tBpB, tBgB(_,_,_,*k_tile_iter), tBsB(_,_,_,k_pipe)); // CpAsync
|
||||
cp_async_fence();
|
||||
++k_tile_iter;
|
||||
--k_tile_count;
|
||||
}
|
||||
|
||||
//
|
||||
// MMA Atom partitioning
|
||||
//
|
||||
|
||||
// Tile MMA compute thread partitions and allocate accumulators
|
||||
TiledMma tiled_mma;
|
||||
auto thr_mma = tiled_mma.get_thread_slice(thread_idx);
|
||||
Tensor tCrA = thr_mma.partition_fragment_A(sA(_,_,0)); // (MMA,MMA_M,MMA_K)
|
||||
Tensor tCrB = thr_mma.partition_fragment_B(sB(_,_,0)); // (MMA,MMA_N,MMA_K)
|
||||
|
||||
CUTE_STATIC_ASSERT_V(size<1>(tCrA) == size<1>(accum)); // MMA_M
|
||||
CUTE_STATIC_ASSERT_V(size<1>(tCrA) == size<1>(src_accum)); // MMA_M
|
||||
CUTE_STATIC_ASSERT_V(size<1>(tCrB) == size<2>(accum)); // MMA_N
|
||||
CUTE_STATIC_ASSERT_V(size<1>(tCrB) == size<2>(src_accum)); // MMA_N
|
||||
CUTE_STATIC_ASSERT_V(size<2>(tCrA) == size<2>(tCrB)); // MMA_K
|
||||
|
||||
//
|
||||
// Copy Atom retiling
|
||||
//
|
||||
|
||||
auto smem_tiled_copy_A = make_tiled_copy_A(SmemCopyAtomA{}, tiled_mma);
|
||||
auto smem_thr_copy_A = smem_tiled_copy_A.get_thread_slice(thread_idx);
|
||||
Tensor tCsA = smem_thr_copy_A.partition_S(sA); // (CPY,CPY_M,CPY_K,PIPE)
|
||||
Tensor tCrA_copy_view = smem_thr_copy_A.retile_D(tCrA); // (CPY,CPY_M,CPY_K)
|
||||
CUTE_STATIC_ASSERT_V(size<1>(tCsA) == size<1>(tCrA_copy_view)); // CPY_M
|
||||
CUTE_STATIC_ASSERT_V(size<2>(tCsA) == size<2>(tCrA_copy_view)); // CPY_K
|
||||
|
||||
auto smem_tiled_copy_B = make_tiled_copy_B(SmemCopyAtomB{}, tiled_mma);
|
||||
auto smem_thr_copy_B = smem_tiled_copy_B.get_thread_slice(thread_idx);
|
||||
Tensor tCsB = smem_thr_copy_B.partition_S(sB); // (CPY,CPY_N,CPY_K,PIPE)
|
||||
Tensor tCrB_copy_view = smem_thr_copy_B.retile_D(tCrB); // (CPY,CPY_N,CPY_K)
|
||||
CUTE_STATIC_ASSERT_V(size<1>(tCsB) == size<1>(tCrB_copy_view)); // CPY_N
|
||||
CUTE_STATIC_ASSERT_V(size<2>(tCsB) == size<2>(tCrB_copy_view)); // CPY_K
|
||||
|
||||
//
|
||||
// PIPELINED MAIN LOOP
|
||||
//
|
||||
|
||||
// Current pipe index in smem to read from
|
||||
int smem_pipe_read = 0;
|
||||
// Current pipe index in smem to write to
|
||||
int smem_pipe_write = DispatchPolicy::Stages-1;
|
||||
|
||||
Tensor tCsA_p = tCsA(_,_,_,smem_pipe_read);
|
||||
Tensor tCsB_p = tCsB(_,_,_,smem_pipe_read);
|
||||
|
||||
// Size of the register pipeline
|
||||
auto K_BLOCK_MAX = size<2>(tCrA);
|
||||
|
||||
// PREFETCH register pipeline
|
||||
if (K_BLOCK_MAX > 1) {
|
||||
// Wait until our first prefetched tile is loaded in
|
||||
cp_async_wait<DispatchPolicy::Stages-2>();
|
||||
__syncthreads();
|
||||
|
||||
// Prefetch the first rmem from the first k-tile
|
||||
copy(smem_tiled_copy_A, tCsA_p(_,_,Int<0>{}), tCrA_copy_view(_,_,Int<0>{}));
|
||||
copy(smem_tiled_copy_B, tCsB_p(_,_,Int<0>{}), tCrB_copy_view(_,_,Int<0>{}));
|
||||
}
|
||||
|
||||
CUTLASS_PRAGMA_NO_UNROLL
|
||||
for ( ; k_tile_count > -(DispatchPolicy::Stages-1); --k_tile_count)
|
||||
{
|
||||
// Pipeline the outer products with a static for loop.
|
||||
//
|
||||
// Note, the for_each() function is required here to ensure `k_block` is of type Int<N>.
|
||||
for_each(make_int_sequence<K_BLOCK_MAX>{}, [&] (auto k_block)
|
||||
{
|
||||
if (k_block == K_BLOCK_MAX - 1)
|
||||
{
|
||||
// Slice the smem_pipe_read smem
|
||||
tCsA_p = tCsA(_,_,_,smem_pipe_read);
|
||||
tCsB_p = tCsB(_,_,_,smem_pipe_read);
|
||||
|
||||
// Commit the smem for smem_pipe_read
|
||||
cp_async_wait<DispatchPolicy::Stages-2>();
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
// Load A, B shmem->regs for k_block+1
|
||||
auto k_block_next = (k_block + Int<1>{}) % K_BLOCK_MAX; // static
|
||||
copy(smem_tiled_copy_A, tCsA_p(_,_,k_block_next), tCrA_copy_view(_,_,k_block_next));
|
||||
copy(smem_tiled_copy_B, tCsB_p(_,_,k_block_next), tCrB_copy_view(_,_,k_block_next));
|
||||
// Copy gmem to smem before computing gemm on each k-pipe
|
||||
if (k_block == 0)
|
||||
{
|
||||
// Set all predicates to false if we are going to overshoot bounds
|
||||
if (k_tile_count <= 0) {
|
||||
clear(tApA);
|
||||
clear(tBpB);
|
||||
}
|
||||
copy_if(gmem_tiled_copy_A, tApA, tAgA(_,_,_,*k_tile_iter), tAsA(_,_,_,smem_pipe_write));
|
||||
copy_if(gmem_tiled_copy_B, tBpB, tBgB(_,_,_,*k_tile_iter), tBsB(_,_,_,smem_pipe_write));
|
||||
cp_async_fence();
|
||||
++k_tile_iter;
|
||||
|
||||
// Advance the pipe -- Doing it here accounts for K_BLOCK_MAX = 1 (no rmem pipe)
|
||||
smem_pipe_write = smem_pipe_read;
|
||||
++smem_pipe_read;
|
||||
smem_pipe_read = (smem_pipe_read == DispatchPolicy::Stages) ? 0 : smem_pipe_read;
|
||||
}
|
||||
|
||||
// Transform before compute
|
||||
cute::transform(tCrA(_,_,k_block), TransformA{});
|
||||
cute::transform(tCrB(_,_,k_block), TransformB{});
|
||||
// Thread-level register gemm for k_block
|
||||
cute::gemm(tiled_mma, accum, tCrA(_,_,k_block), tCrB(_,_,k_block), src_accum);
|
||||
});
|
||||
|
||||
}
|
||||
|
||||
cp_async_wait<0>();
|
||||
__syncthreads();
|
||||
}
|
||||
};
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace cutlass::gemm::collective
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
+1
-6
@@ -89,15 +89,10 @@ struct CollectiveMma<
|
||||
TransformB_>
|
||||
{
|
||||
public:
|
||||
enum class ConversionMode {
|
||||
DirectConvert,
|
||||
ConvertAndScale,
|
||||
ConvertAndScaleWithZero
|
||||
};
|
||||
|
||||
//
|
||||
// Type Aliases
|
||||
//
|
||||
using ConversionMode = cutlass::detail::ConversionMode;
|
||||
using DispatchPolicy = MainloopSm90ArrayTmaGmmaWarpSpecializedMixedInput<Stages, ClusterShape, KernelSchedule_>;
|
||||
using TileShape = TileShape_;
|
||||
using KernelSchedule = KernelSchedule_;
|
||||
|
||||
+1
-5
@@ -96,15 +96,11 @@ struct CollectiveMma<
|
||||
TransformB_>
|
||||
{
|
||||
public:
|
||||
enum class ConversionMode {
|
||||
DirectConvert,
|
||||
ConvertAndScale,
|
||||
ConvertAndScaleWithZero
|
||||
};
|
||||
|
||||
//
|
||||
// Type Aliases
|
||||
//
|
||||
using ConversionMode = cutlass::detail::ConversionMode;
|
||||
using DispatchPolicy = MainloopSm90TmaGmmaRmemAWarpSpecializedMixedInput<Stages, ClusterShape, KernelSchedule_>;
|
||||
using TileShape = TileShape_;
|
||||
using KernelSchedule = KernelSchedule_;
|
||||
|
||||
@@ -109,6 +109,7 @@ static constexpr bool HasAuxiliaryLoad_v = HasAuxiliaryLoad<T>::value;
|
||||
// Kernel schedule policies (the base class tags, one for each kernel layer file)
|
||||
//
|
||||
struct KernelMultistage { };
|
||||
struct KernelPtrArrayMultistage { };
|
||||
struct KernelCpAsyncWarpSpecialized { };
|
||||
struct KernelCpAsyncWarpSpecializedPingpong { };
|
||||
struct KernelCpAsyncWarpSpecializedCooperative { };
|
||||
@@ -198,6 +199,17 @@ struct MainloopSm80CpAsync {
|
||||
using ClusterShape = ClusterShape_;
|
||||
};
|
||||
|
||||
// n-buffer in smem (cp.async), pipelined with registers, with predicated gmem loads for SM100 Simt Ptr-Array
|
||||
template<int Stages_,
|
||||
class ClusterShape_ = Shape<_1,_1,_1>
|
||||
>
|
||||
struct MainloopSm80ArrayCpAsync {
|
||||
constexpr static int Stages = Stages_;
|
||||
using ArchTag = cute::conditional_t<(size(ClusterShape_{}) > 1), arch::Sm90, arch::Sm80>;
|
||||
using Schedule = KernelPtrArrayMultistage;
|
||||
using ClusterShape = ClusterShape_;
|
||||
};
|
||||
|
||||
// n-buffer in smem (cp.async), pipelined with Hopper GMMA, with predicated gmem loads, warp specialized dynamic schedule
|
||||
template<
|
||||
int Stages_,
|
||||
@@ -479,6 +491,16 @@ struct KernelTmaWarpSpecializedInputTransformSm100 final {
|
||||
static constexpr int AccumulatorPipelineStageCount = AccumulatorPipelineStageCount_;
|
||||
};
|
||||
|
||||
// InputTransform GEMM
|
||||
template<
|
||||
int SchedulerPipelineStageCount_,
|
||||
int AccumulatorPipelineStageCount_
|
||||
>
|
||||
struct KernelTmaWarpSpecializedMixedInputTransformSm100 final {
|
||||
static constexpr int SchedulerPipelineStageCount = SchedulerPipelineStageCount_;
|
||||
static constexpr int AccumulatorPipelineStageCount = AccumulatorPipelineStageCount_;
|
||||
};
|
||||
|
||||
// Ptr-Array Dense GEMM: SM100 tensor op policy that applies to both 1SM and 2SM MMA atoms
|
||||
template<
|
||||
int SchedulerPipelineStageCount_,
|
||||
|
||||
@@ -54,6 +54,7 @@ struct IsCutlass3ArrayKernel<ProblemShape, cute::void_t<typename ProblemShape::U
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include "cutlass/gemm/kernel/sm70_gemm.hpp"
|
||||
#include "cutlass/gemm/kernel/sm70_gemm_array.hpp"
|
||||
#include "cutlass/gemm/kernel/sm90_gemm_tma.hpp"
|
||||
#include "cutlass/gemm/kernel/sm90_gemm_warpspecialized.hpp"
|
||||
#include "cutlass/gemm/kernel/sm90_gemm_warpspecialized_pingpong.hpp"
|
||||
|
||||
@@ -1008,39 +1008,6 @@ public:
|
||||
// Advance the mm2accum pipe
|
||||
mma2accum_pipeline_consumer_state = mma2accum_pipeline_consumer_state_next;
|
||||
}
|
||||
else if constexpr (InputTransformType == cutlass::gemm::detail::KernelInputTransformType::MixedInput) {
|
||||
|
||||
mma2accum_pipeline.consumer_wait(mma2accum_pipeline_consumer_state);
|
||||
|
||||
// Accumulators
|
||||
Tensor accumulators = bulk_tmem(_,_,_,mma2accum_pipeline_consumer_state.index()); // ((MMA_TILE_M,MMA_TILE_N),MMA_M,MMA_N)
|
||||
|
||||
mma2accum_pipeline_consumer_state = scheduler.template fixup<IsComplex>(
|
||||
TiledMma{},
|
||||
work_tile_info,
|
||||
accumulators,
|
||||
mma2accum_pipeline,
|
||||
mma2accum_pipeline_consumer_state,
|
||||
typename CollectiveEpilogue::CopyOpT2R{}
|
||||
);
|
||||
|
||||
//
|
||||
// Epilogue and write to gD
|
||||
//
|
||||
if (scheduler.compute_epilogue(work_tile_info)) {
|
||||
auto [mma2accum_pipeline_state_next] = collective_epilogue(
|
||||
mma2accum_pipeline,
|
||||
mma2accum_pipeline_consumer_state,
|
||||
problem_shape_MNKL,
|
||||
CtaShape_MNK{},
|
||||
cta_coord_mnkl,
|
||||
accumulators,
|
||||
shared_storage.tensors.epilogue
|
||||
);
|
||||
// Advance the mma2accum pipe
|
||||
mma2accum_pipeline_consumer_state = mma2accum_pipeline_state_next;
|
||||
}
|
||||
}
|
||||
// Complex kernels use a collective epilogue
|
||||
else {
|
||||
mma2accum_pipeline.consumer_wait(mma2accum_pipeline_consumer_state);
|
||||
|
||||
@@ -0,0 +1,279 @@
|
||||
/***************************************************************************************************
|
||||
* Copyright (c) 2023 - 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
* SPDX-License-Identifier: BSD-3-Clause
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions are met:
|
||||
*
|
||||
* 1. Redistributions of source code must retain the above copyright notice, this
|
||||
* list of conditions and the following disclaimer.
|
||||
*
|
||||
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||
* this list of conditions and the following disclaimer in the documentation
|
||||
* and/or other materials provided with the distribution.
|
||||
*
|
||||
* 3. Neither the name of the copyright holder nor the names of its
|
||||
* contributors may be used to endorse or promote products derived from
|
||||
* this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
||||
* DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
||||
* SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
||||
* CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
||||
* OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*
|
||||
**************************************************************************************************/
|
||||
#pragma once
|
||||
|
||||
#include "cutlass/cutlass.h"
|
||||
#include "cutlass/kernel_hardware_info.hpp"
|
||||
#include "cutlass/gemm/gemm.h"
|
||||
#include "cutlass/gemm/dispatch_policy.hpp"
|
||||
|
||||
#include "cute/tensor.hpp"
|
||||
|
||||
namespace cutlass::gemm::kernel {
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template <
|
||||
class ProblemShape_,
|
||||
class CollectiveMainloop_,
|
||||
class CollectiveEpilogue_,
|
||||
class TileScheduler_
|
||||
>
|
||||
class GemmUniversal<
|
||||
ProblemShape_,
|
||||
CollectiveMainloop_,
|
||||
CollectiveEpilogue_,
|
||||
TileScheduler_,
|
||||
cute::enable_if_t<cute::is_base_of_v<KernelPtrArrayMultistage, typename CollectiveMainloop_::DispatchPolicy::Schedule>>>
|
||||
{
|
||||
public:
|
||||
//
|
||||
// Type Aliases
|
||||
//
|
||||
using ProblemShape = ProblemShape_;
|
||||
static_assert(rank(typename ProblemShape::UnderlyingProblemShape{}) == 4,
|
||||
"ProblemShape{} should be <M,N,K> or <M,N,K,L>");
|
||||
|
||||
// Mainloop derived types
|
||||
using CollectiveMainloop = CollectiveMainloop_;
|
||||
using TileShape = typename CollectiveMainloop::TileShape;
|
||||
using TiledMma = typename CollectiveMainloop::TiledMma;
|
||||
using ArchTag = typename CollectiveMainloop::ArchTag;
|
||||
using ElementA = typename CollectiveMainloop::ElementA;
|
||||
using StrideA = typename CollectiveMainloop::StrideA;
|
||||
using InternalStrideA = typename CollectiveMainloop::InternalStrideA;
|
||||
using ElementB = typename CollectiveMainloop::ElementB;
|
||||
using StrideB = typename CollectiveMainloop::StrideB;
|
||||
using InternalStrideB = typename CollectiveMainloop::InternalStrideB;
|
||||
using DispatchPolicy = typename CollectiveMainloop::DispatchPolicy;
|
||||
using ElementAccumulator = typename CollectiveMainloop::ElementAccumulator;
|
||||
using MainloopArguments = typename CollectiveMainloop::Arguments;
|
||||
using MainloopParams = typename CollectiveMainloop::Params;
|
||||
|
||||
using TileSchedulerTag = TileScheduler_;
|
||||
using TileScheduler = typename detail::TileSchedulerSelector<
|
||||
TileScheduler_, ArchTag, TileShape,
|
||||
cute::Shape<cute::Int<1>, cute::Int<1>, cute::Int<1>>>::Scheduler;
|
||||
using TileSchedulerArguments = typename TileScheduler::Arguments;
|
||||
static constexpr bool IsGdcEnabled = false;
|
||||
|
||||
static constexpr bool is_valid_tile_scheduler =
|
||||
cute::is_void_v<TileScheduler_> or cute::is_same_v<TileScheduler_, PersistentScheduler>;
|
||||
static_assert(is_valid_tile_scheduler, "SM70 kernel does not support specializing the tile scheduler.");
|
||||
|
||||
// Epilogue derived types
|
||||
using CollectiveEpilogue = CollectiveEpilogue_;
|
||||
using ElementC = typename CollectiveEpilogue::ElementC;
|
||||
using StrideC = typename CollectiveEpilogue::StrideC;
|
||||
using InternalStrideC = typename CollectiveEpilogue::InternalStrideC;
|
||||
using ElementD = typename CollectiveEpilogue::ElementD;
|
||||
using StrideD = typename CollectiveEpilogue::StrideD;
|
||||
using InternalStrideD = typename CollectiveEpilogue::InternalStrideD;
|
||||
using EpilogueArguments = typename CollectiveEpilogue::Arguments;
|
||||
using EpilogueParams = typename CollectiveEpilogue::Params;
|
||||
static_assert(cute::is_same_v<ElementAccumulator, typename CollectiveEpilogue::ElementAccumulator>,
|
||||
"Mainloop and epilogue do not agree on accumulator value type.");
|
||||
|
||||
// MSVC requires the cast to fix a warning-as-error.
|
||||
static constexpr int SharedStorageSize = static_cast<int>(cute::max(
|
||||
sizeof(typename CollectiveMainloop::SharedStorage),
|
||||
sizeof(typename CollectiveEpilogue::SharedStorage)));
|
||||
|
||||
static constexpr uint32_t MaxThreadsPerBlock = CUTE_STATIC_V(cute::size(TiledMma{}));
|
||||
static constexpr uint32_t MinBlocksPerMultiprocessor = 1;
|
||||
|
||||
// Device side arguments
|
||||
struct Arguments {
|
||||
GemmUniversalMode mode{};
|
||||
ProblemShape problem_shape{};
|
||||
MainloopArguments mainloop{};
|
||||
EpilogueArguments epilogue{};
|
||||
KernelHardwareInfo hw_info{};
|
||||
TileSchedulerArguments scheduler{};
|
||||
};
|
||||
|
||||
// Kernel entry point API
|
||||
struct Params {
|
||||
GemmUniversalMode mode{};
|
||||
typename ProblemShape::UnderlyingProblemShape problem_shape{};
|
||||
MainloopParams mainloop{};
|
||||
EpilogueParams epilogue{};
|
||||
};
|
||||
|
||||
//
|
||||
// Methods
|
||||
//
|
||||
|
||||
// Convert to underlying arguments. In this case, a simple copy for the aliased type.
|
||||
static
|
||||
Params
|
||||
to_underlying_arguments(Arguments const& args, void* workspace) {
|
||||
(void) workspace;
|
||||
typename ProblemShape::UnderlyingProblemShape problem_shape = args.problem_shape.get_host_problem_shape();
|
||||
|
||||
KernelHardwareInfo hw_info{args.hw_info.device_id, args.hw_info.sm_count};
|
||||
auto problem_shape_MNKL = append<4>(args.problem_shape, Int<1>{});
|
||||
|
||||
return {
|
||||
args.mode,
|
||||
problem_shape,
|
||||
CollectiveMainloop::to_underlying_arguments(problem_shape, args.mainloop, workspace),
|
||||
CollectiveEpilogue::to_underlying_arguments(problem_shape, args.epilogue, workspace)
|
||||
};
|
||||
}
|
||||
|
||||
static bool
|
||||
can_implement(Arguments const& args) {
|
||||
|
||||
bool implementable = (args.mode == GemmUniversalMode::kArray && rank(typename ProblemShape::UnderlyingProblemShape{}) == 4);
|
||||
if (!implementable) {
|
||||
CUTLASS_TRACE_HOST(" CAN IMPLEMENT: Arguments or Problem Shape don't meet the requirements.\n");
|
||||
return implementable;
|
||||
}
|
||||
typename ProblemShape::UnderlyingProblemShape problem_shape = args.problem_shape.get_host_problem_shape();
|
||||
implementable &= TileScheduler::can_implement(args.scheduler);
|
||||
return implementable;
|
||||
}
|
||||
|
||||
static size_t
|
||||
get_workspace_size(Arguments const& args) {
|
||||
size_t workspace_size = 0;
|
||||
return workspace_size;
|
||||
}
|
||||
|
||||
static
|
||||
cutlass::Status
|
||||
initialize_workspace(Arguments const& args, void* workspace = nullptr, cudaStream_t stream = nullptr,
|
||||
CudaHostAdapter* cuda_adapter = nullptr) {
|
||||
cutlass::Status status = Status::kSuccess;
|
||||
|
||||
return status;
|
||||
}
|
||||
|
||||
static dim3
|
||||
get_grid_shape(Params const& params) {
|
||||
int batch_count = cute::size<3>(params.problem_shape);
|
||||
return dim3(
|
||||
cute::size(cute::ceil_div(cute::shape<0>(params.problem_shape), cute::shape<0>(TileShape{}))),
|
||||
cute::size(cute::ceil_div(cute::shape<1>(params.problem_shape), cute::shape<1>(TileShape{}))),
|
||||
batch_count
|
||||
);
|
||||
}
|
||||
|
||||
static dim3
|
||||
get_block_shape() {
|
||||
return dim3(MaxThreadsPerBlock, 1, 1);
|
||||
}
|
||||
|
||||
CUTLASS_DEVICE
|
||||
void
|
||||
operator()(Params const& params, char* smem_buf) {
|
||||
using namespace cute;
|
||||
using X = Underscore;
|
||||
|
||||
// Preconditions
|
||||
CUTE_STATIC_ASSERT(is_static<TileShape>::value);
|
||||
|
||||
// Separate out problem shape for convenience
|
||||
// Optionally append 1s until problem shape is rank-4 in case its is only rank-3 (MNK)
|
||||
auto problem_shape_MNKL = append<4>(params.problem_shape, Int<1>{});
|
||||
auto [M,N,K,L] = problem_shape_MNKL;
|
||||
|
||||
// Preconditions
|
||||
static_assert(cute::rank(StrideA{}) == 3, "StrideA must be rank-3: [M, K, L]. If batch mode is not needed, set L stride to Int<0>.");
|
||||
static_assert(cute::rank(StrideB{}) == 3, "StrideB must be rank-3: [N, K, L]. If batch mode is not needed, set L stride to Int<0>.");
|
||||
static_assert(cute::rank(StrideC{}) == 3, "StrideC must be rank-3: [M, N, L]. If batch mode is not needed, set L stride to Int<0>.");
|
||||
static_assert(cute::rank(StrideD{}) == 3, "StrideD must be rank-3: [M, N, L]. If batch mode is not needed, set L stride to Int<0>.");
|
||||
|
||||
// Get the appropriate blocks for this thread block -- potential for thread block locality
|
||||
int thread_idx = int(threadIdx.x);
|
||||
auto blk_shape = TileShape{}; // (BLK_M,BLK_N,BLK_K)
|
||||
auto [m_coord, n_coord, l_coord] = static_cast<uint3>(blockIdx);
|
||||
auto blk_coord_mnkl = make_coord(int(m_coord), int(n_coord), _, int(l_coord)); // (m,n,k,l)
|
||||
|
||||
// Represent the full tensors
|
||||
Tensor mA_mkl = make_tensor(make_gmem_ptr(params.mainloop.ptr_A[l_coord]), make_shape(M,K,1), params.mainloop.dA); //(m,k,l)
|
||||
Tensor mB_nkl = make_tensor(make_gmem_ptr(params.mainloop.ptr_B[l_coord]), make_shape(N,K,1), params.mainloop.dB); //(n,k,l)
|
||||
|
||||
// Get batch slice
|
||||
Tensor mA_mk = mA_mkl(_,_,0); // (m,k)
|
||||
Tensor mB_nk = mB_nkl(_,_,0); // (n,k)
|
||||
|
||||
// Slice to get the tiles this thread block is responsible for
|
||||
Tensor gA = local_tile(mA_mk, blk_shape, take<0,3>(blk_coord_mnkl), Step<_1, X,_1>{}); // (BLK_M,BLK_K,k)
|
||||
Tensor gB = local_tile(mB_nk, blk_shape, take<0,3>(blk_coord_mnkl), Step< X,_1,_1>{}); // (BLK_N,BLK_K,k)
|
||||
|
||||
// Compute tile residues for predication
|
||||
auto m_max_coord = M - size<0>(gA) * get<0>(blk_coord_mnkl); // M - BLK_M * m_coord
|
||||
auto n_max_coord = N - size<0>(gB) * get<1>(blk_coord_mnkl); // N - BLK_N * n_coord
|
||||
auto k_residue = K - size<1>(gA) * size<2>(gA); // K - BLK_K * k_coord_max
|
||||
auto residue_mnk = make_tuple(m_max_coord, n_max_coord, k_residue);
|
||||
|
||||
// Allocate the tiled_mma and the accumulators for the (M,N) blk_shape
|
||||
TiledMma tiled_mma;
|
||||
Tensor accumulators = partition_fragment_C(tiled_mma, take<0,2>(blk_shape)); // (MMA,MMA_M,MMA_N)
|
||||
clear(accumulators);
|
||||
|
||||
auto k_tile_iter = cute::make_coord_iterator(shape<2>(gA));
|
||||
int k_tile_count = size<2>(gA);
|
||||
|
||||
|
||||
// Perform the collective scoped MMA
|
||||
CollectiveMainloop collective_mma;
|
||||
collective_mma(
|
||||
accumulators,
|
||||
gA,
|
||||
gB,
|
||||
accumulators,
|
||||
k_tile_iter, k_tile_count,
|
||||
residue_mnk,
|
||||
thread_idx,
|
||||
smem_buf
|
||||
);
|
||||
|
||||
// Epilogue and write to gD
|
||||
CollectiveEpilogue epilogue{params.epilogue};
|
||||
epilogue(
|
||||
problem_shape_MNKL,
|
||||
blk_shape,
|
||||
blk_coord_mnkl,
|
||||
accumulators,
|
||||
tiled_mma,
|
||||
residue_mnk,
|
||||
thread_idx,
|
||||
smem_buf
|
||||
);
|
||||
}
|
||||
};
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
} // namespace cutlass::gemm::kernel
|
||||
@@ -194,6 +194,9 @@ struct integer_subbyte {
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/// 1-bit binary type
|
||||
using bin1_t = bool;
|
||||
|
||||
/// 1-bit Unsigned integer type
|
||||
using uint1b_t = integer_subbyte<1, false>;
|
||||
|
||||
@@ -209,14 +212,12 @@ using int4b_t = integer_subbyte<4, true>;
|
||||
/// 4-bit Unsigned integer type
|
||||
using uint4b_t = integer_subbyte<4, false>;
|
||||
|
||||
/// 6-bit integer type
|
||||
using int6b_t = integer_subbyte<6, true>;
|
||||
|
||||
/// 6-bit unsigned integer type
|
||||
using uint6b_t = integer_subbyte<6, false>;
|
||||
|
||||
|
||||
/// 1-bit binary type
|
||||
using bin1_t = bool;
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
template <int Bits, bool Signed>
|
||||
|
||||
@@ -50,7 +50,13 @@ struct sizeof_bits {
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
struct sizeof_bits<T const>: sizeof_bits<T> {};
|
||||
struct sizeof_bits<T const> : sizeof_bits<T> {};
|
||||
|
||||
template <typename T>
|
||||
struct sizeof_bits<T volatile> : sizeof_bits<T> {};
|
||||
|
||||
template <typename T>
|
||||
struct sizeof_bits<T const volatile> : sizeof_bits<T> {};
|
||||
|
||||
template <>
|
||||
struct sizeof_bits<void> {
|
||||
|
||||
Reference in New Issue
Block a user