v4.1 release update v2. (#2481)
This commit is contained in:
@@ -31,7 +31,7 @@
|
||||
|
||||
/*! \file
|
||||
\brief Cutlass provides helper template functions to figure out the right
|
||||
datastructures to instanciate to run a GEMM with various parameters (see
|
||||
datastructures to instantiate to run a GEMM with various parameters (see
|
||||
`cutlass/gemm/threadblock/default_mma.h`). However, due to template
|
||||
instantiation priority rules, it will only create an MmaMultiStage with
|
||||
kStages=3 (otherwise creates an MmePipelined - which is not compatible with
|
||||
@@ -83,7 +83,7 @@ template <
|
||||
typename InstructionShape,
|
||||
/// Number of stages used in the pipelined mainloop
|
||||
int Stages,
|
||||
/// Operation perfomed by GEMM
|
||||
/// Operation performed by GEMM
|
||||
typename Operator,
|
||||
typename Enable_ = void>
|
||||
struct FindDefaultMma {
|
||||
|
||||
@@ -522,7 +522,7 @@ class MmaPipelinedFromSharedMemory : public MmaBaseFromSharedMemory<
|
||||
|
||||
// For API compatibility with MmaMultistageFromSharedMemory
|
||||
// but not supported as it worsens perf: older gpus < sm80 don't
|
||||
// support async tranfers and have to waste registers
|
||||
// support async transfers and have to waste registers
|
||||
CUTLASS_DEVICE
|
||||
void set_prologue_done(bool value) {}
|
||||
CUTLASS_DEVICE
|
||||
|
||||
Reference in New Issue
Block a user