* v3.8 update x

* fix blackwell gg

* doc change

* doc change

* doc change

---------

Co-authored-by: yuzhai <yuzhai@nvidia.com>
Co-authored-by: Haicheng Wu <haichengw@nvidia.com>
Co-authored-by: Haicheng Wu <57973641+hwu36@users.noreply.github.com>
This commit is contained in:
Yujia Zhai
2025-03-21 01:52:23 -04:00
committed by GitHub
co-authored by yuzhai Haicheng Wu Haicheng Wu
parent 8c4d1dc47d
commit 62750a2b75
334 changed files with 91517 additions and 2656 deletions
@@ -127,9 +127,13 @@ template <typename OperatorClass> struct ArchMap<arch::Sm100, OperatorClass> {
template <> struct ArchMap<arch::Sm100, arch::OpClassTensorOp> {
static int const kMin = 100;
static int const kMax = 100;
static int const kMax = 101;
};
template <typename OperatorClass> struct ArchMap<arch::Sm120, OperatorClass> {
static int const kMin = 120;
static int const kMax = 120;
};
/////////////////////////////////////////////////////////////////////////////////////////////////
@@ -323,12 +323,11 @@ struct GemmUniversalArguments {
int swizzle_size{1};
int split_k_slices{1};
// For mixed input dtype kernels
bool is_mixed_dtype{false};
// For SM90 mixed input dtype kernels
bool is_sm90_mixed_dtype{false};
Sm90MixedInputWiderOperand wider_operand{Sm90MixedInputWiderOperand::B};
bool generate_scale_and_zero{false};
bool generate_dequantized_AB{false};
bool *dequantized_AB_ready{nullptr}; // Carry the info back to gemm_operation_profiler.cu
void *Scale{nullptr}; // Scale tensor
void *Zero{nullptr}; // Zero tensor
void *dequantized_AB{nullptr}; // Dequantized A or B tensor for verification