* v3.8 update x

* fix blackwell gg

* doc change

* doc change

* doc change

---------

Co-authored-by: yuzhai <yuzhai@nvidia.com>
Co-authored-by: Haicheng Wu <haichengw@nvidia.com>
Co-authored-by: Haicheng Wu <57973641+hwu36@users.noreply.github.com>
This commit is contained in:
Yujia Zhai
2025-03-21 01:52:23 -04:00
committed by GitHub
co-authored by yuzhai Haicheng Wu Haicheng Wu
parent 8c4d1dc47d
commit 62750a2b75
334 changed files with 91517 additions and 2656 deletions
@@ -37,6 +37,7 @@
#pragma once
#include <vector>
#include <array>
#include <string>
#include <memory>
#include <algorithm>
@@ -71,6 +72,14 @@ public:
cutlass::library::GemmUniversalMode mode{library::GemmUniversalMode::kGemm};
/// For profiling purposes
std::vector<gemm::GemmCoord> problem_sizes;
std::vector<std::array<int64_t, 3>> leading_dims;
std::vector<std::array<int64_t, 3>> preferred_clusters;
std::vector<std::array<int64_t, 3>> fallback_clusters;
std::vector<cutlass::library::RasterOrder> raster_orders;
std::vector<int> swizzle_sizes;
int64_t m{16};
int64_t n{16};
int64_t k{16};
@@ -119,6 +128,14 @@ public:
ProblemSpace const &problem_space,
ProblemSpace::Problem const &problem);
int64_t bytes_with_problem_shape(
library::BlockScaledGemmDescription const &operation_desc,
gemm::GemmCoord const &problem_shape) const;
int64_t flops_with_problem_shape(
library::BlockScaledGemmDescription const &operation_desc,
gemm::GemmCoord const &problem_shape) const;
/// Total number of bytes loaded
int64_t bytes(library::BlockScaledGemmDescription const &operation_desc) const;
@@ -239,6 +256,27 @@ public:
protected:
/// Update workspace configuration according to flexible user setups
void update_workspace_(
GemmWorkspace &gemm_workspace,
gemm::GemmCoord const &problem_shape,
std::array<int64_t, 3> const &leading_dim,
std::array<int64_t, 3> const &preferred_cluster,
std::array<int64_t, 3> const &fallback_cluster,
cutlass::library::RasterOrder const &raster_order,
int swizzle_size);
/// Update performance result configuration according to flexible user setups
void update_result_(
PerformanceResult &result,
library::BlockScaledGemmDescription const &operation_desc,
ProblemSpace const &problem_space,
gemm::GemmCoord const &problem_shape,
cutlass::library::RasterOrder const &raster_order,
std::array<int64_t, 3> const &preferred_cluster,
std::array<int64_t, 3> const &fallback_cluster,
int swizzle_size);
/// Initializes the performance result
void initialize_result_(
PerformanceResult &result,
@@ -35,6 +35,7 @@
#pragma once
#include <vector>
#include <array>
#include <string>
#include <memory>
#include <algorithm>
@@ -69,6 +70,14 @@ public:
cutlass::library::GemmUniversalMode mode{library::GemmUniversalMode::kGemm};
/// For profiling purposes
std::vector<gemm::GemmCoord> problem_sizes;
std::vector<std::array<int64_t, 3>> leading_dims;
std::vector<std::array<int64_t, 3>> preferred_clusters;
std::vector<std::array<int64_t, 3>> fallback_clusters;
std::vector<cutlass::library::RasterOrder> raster_orders;
std::vector<int> swizzle_sizes;
int64_t m{16};
int64_t n{16};
int64_t k{16};
@@ -120,6 +129,14 @@ public:
ProblemSpace const &problem_space,
ProblemSpace::Problem const &problem);
int64_t bytes_with_problem_shape(
library::GemmDescription const &operation_desc,
gemm::GemmCoord const &problem_shape) const;
int64_t flops_with_problem_shape(
library::GemmDescription const &operation_desc,
gemm::GemmCoord const &problem_shape) const;
/// Total number of bytes loaded
int64_t bytes(library::GemmDescription const &operation_desc) const;
@@ -243,6 +260,26 @@ public:
ProblemSpace::Problem const &problem);
protected:
/// Update workspace configuration according to flexible user setups
void update_workspace_(
GemmWorkspace &gemm_workspace,
gemm::GemmCoord const &problem_shape,
std::array<int64_t, 3> const &leading_dim,
std::array<int64_t, 3> const &preferred_cluster,
std::array<int64_t, 3> const &fallback_cluster,
cutlass::library::RasterOrder const &raster_order,
int swizzle_size);
/// Update performance result configuration according to flexible user setups
void update_result_(
PerformanceResult &result,
library::GemmDescription const &operation_desc,
ProblemSpace const &problem_space,
gemm::GemmCoord const &problem_shape,
cutlass::library::RasterOrder const &raster_order,
std::array<int64_t, 3> const &preferred_cluster,
std::array<int64_t, 3> const &fallback_cluster,
int swizzle_size);
/// Initializes the performance result
void initialize_result_(
@@ -208,8 +208,23 @@ public:
/// Minimum number of iterations to profile
int min_iterations{10};
/// If true, profiling with cuda graph enabled.
bool use_cuda_graphs{false};
/// If enabled, the CUTLASS profiler searches for the best-performing kernel
/// within the subset of kernels matching a kernel filter regex. The best
/// performance is determined by screening over a set of predefined M/N/K
/// sizes and performance-related parameters, including cluster shapes,
/// swizzle sizes, and rasterization orders.
/// For now, it only supports legacy GEMM and blockscaled GEMM.
bool enable_kernel_performance_search{false};
/// If enabled, the CUTLASS profiler searches for the best-performing kernel
/// for a given M/N/K problem size by evaluating various performance-related
/// parameters such as cluster shapes, swizzle sizes, and rasterization orders.
/// For now, it only supports legacy GEMM and blockscaled GEMM.
bool enable_best_kernel_for_fixed_shape{false};
/// Number of ms to sleep between profiling periods (ms)
int sleep_duration{50};
@@ -264,8 +279,11 @@ public:
/// Prints human-readable text to stdout. If false, nothing is written to stdout
bool verbose;
/// Sort results by (currently by flops-per-byte)
bool sort_results;
/// Sort results by flops-per-byte
bool sort_flops_per_byte;
/// Sort results by flops-per-second
bool sort_flops_per_sec;
/// Prints the name of the kernel being profiled before running the kernel.
/// This is useful for determining which kernel is causing a run of the profiler to hang
@@ -92,7 +92,8 @@ public:
void next_problem();
void append_result(PerformanceResult result);
void sort_results(PerformanceResultVector &results);
void sort_flops_per_byte(PerformanceResultVector &results);
void sort_flops_per_sec(PerformanceResultVector &results);
void append_results(PerformanceResultVector const &results);
public:
@@ -105,6 +105,12 @@ struct PerformanceResult {
runtime(0)
{ }
// Copy constructor for deep copy
PerformanceResult(const PerformanceResult& other) = default;
// Explicitly define copy assignment operator
PerformanceResult& operator=(const PerformanceResult& other) = default;
/// Returns true if the runtime is valid
bool good() const {
return runtime > 0;