CUTLASS 3.1 (#915)

Co-authored-by: Aniket Shivam <ashivam@nvidia.com>
This commit is contained in:
ANIKET SHIVAM
2023-04-14 23:19:34 -04:00
committed by GitHub
co-authored by Aniket Shivam
parent 9b8166e3f0
commit d572cc1aab
482 changed files with 37175 additions and 16410 deletions
@@ -118,6 +118,6 @@ struct DefaultMmaTensorOp {
/////////////////////////////////////////////////////////////////////////////////////////////////
#include "default_mma_tensor_op_sm80.h"
#include "cutlass/gemm/warp/default_mma_tensor_op_sm80.h"
/////////////////////////////////////////////////////////////////////////////////////////////////
@@ -819,8 +819,13 @@ public:
// Define conversions from source type to instruction operands' type
//
#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 900
FloatRoundStyle const kRoundA = FloatRoundStyle::round_to_nearest;
FloatRoundStyle const kRoundB = FloatRoundStyle::round_to_nearest;
#else
FloatRoundStyle const kRoundA = FloatRoundStyle::round_half_ulp_trunc_dntz;
FloatRoundStyle const kRoundB = FloatRoundStyle::round_half_ulp_trunc_dntz;
#endif
detail::UnpackComplexConvertAndPackForMma <
RealElementA,