v3.9 (#2185)
* v3.8 update x * fix blackwell gg * doc change * doc change * doc change --------- Co-authored-by: yuzhai <yuzhai@nvidia.com> Co-authored-by: Haicheng Wu <haichengw@nvidia.com> Co-authored-by: Haicheng Wu <57973641+hwu36@users.noreply.github.com>
This commit is contained in:
co-authored by
yuzhai
Haicheng Wu
Haicheng Wu
parent
8c4d1dc47d
commit
62750a2b75
@@ -37,6 +37,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <vector>
|
||||
#include <array>
|
||||
#include <string>
|
||||
#include <memory>
|
||||
#include <algorithm>
|
||||
@@ -71,6 +72,14 @@ public:
|
||||
|
||||
cutlass::library::GemmUniversalMode mode{library::GemmUniversalMode::kGemm};
|
||||
|
||||
/// For profiling purposes
|
||||
std::vector<gemm::GemmCoord> problem_sizes;
|
||||
std::vector<std::array<int64_t, 3>> leading_dims;
|
||||
std::vector<std::array<int64_t, 3>> preferred_clusters;
|
||||
std::vector<std::array<int64_t, 3>> fallback_clusters;
|
||||
std::vector<cutlass::library::RasterOrder> raster_orders;
|
||||
std::vector<int> swizzle_sizes;
|
||||
|
||||
int64_t m{16};
|
||||
int64_t n{16};
|
||||
int64_t k{16};
|
||||
@@ -119,6 +128,14 @@ public:
|
||||
ProblemSpace const &problem_space,
|
||||
ProblemSpace::Problem const &problem);
|
||||
|
||||
int64_t bytes_with_problem_shape(
|
||||
library::BlockScaledGemmDescription const &operation_desc,
|
||||
gemm::GemmCoord const &problem_shape) const;
|
||||
|
||||
int64_t flops_with_problem_shape(
|
||||
library::BlockScaledGemmDescription const &operation_desc,
|
||||
gemm::GemmCoord const &problem_shape) const;
|
||||
|
||||
/// Total number of bytes loaded
|
||||
int64_t bytes(library::BlockScaledGemmDescription const &operation_desc) const;
|
||||
|
||||
@@ -239,6 +256,27 @@ public:
|
||||
|
||||
protected:
|
||||
|
||||
/// Update workspace configuration according to flexible user setups
|
||||
void update_workspace_(
|
||||
GemmWorkspace &gemm_workspace,
|
||||
gemm::GemmCoord const &problem_shape,
|
||||
std::array<int64_t, 3> const &leading_dim,
|
||||
std::array<int64_t, 3> const &preferred_cluster,
|
||||
std::array<int64_t, 3> const &fallback_cluster,
|
||||
cutlass::library::RasterOrder const &raster_order,
|
||||
int swizzle_size);
|
||||
|
||||
/// Update performance result configuration according to flexible user setups
|
||||
void update_result_(
|
||||
PerformanceResult &result,
|
||||
library::BlockScaledGemmDescription const &operation_desc,
|
||||
ProblemSpace const &problem_space,
|
||||
gemm::GemmCoord const &problem_shape,
|
||||
cutlass::library::RasterOrder const &raster_order,
|
||||
std::array<int64_t, 3> const &preferred_cluster,
|
||||
std::array<int64_t, 3> const &fallback_cluster,
|
||||
int swizzle_size);
|
||||
|
||||
/// Initializes the performance result
|
||||
void initialize_result_(
|
||||
PerformanceResult &result,
|
||||
|
||||
@@ -35,6 +35,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <vector>
|
||||
#include <array>
|
||||
#include <string>
|
||||
#include <memory>
|
||||
#include <algorithm>
|
||||
@@ -69,6 +70,14 @@ public:
|
||||
|
||||
cutlass::library::GemmUniversalMode mode{library::GemmUniversalMode::kGemm};
|
||||
|
||||
/// For profiling purposes
|
||||
std::vector<gemm::GemmCoord> problem_sizes;
|
||||
std::vector<std::array<int64_t, 3>> leading_dims;
|
||||
std::vector<std::array<int64_t, 3>> preferred_clusters;
|
||||
std::vector<std::array<int64_t, 3>> fallback_clusters;
|
||||
std::vector<cutlass::library::RasterOrder> raster_orders;
|
||||
std::vector<int> swizzle_sizes;
|
||||
|
||||
int64_t m{16};
|
||||
int64_t n{16};
|
||||
int64_t k{16};
|
||||
@@ -120,6 +129,14 @@ public:
|
||||
ProblemSpace const &problem_space,
|
||||
ProblemSpace::Problem const &problem);
|
||||
|
||||
int64_t bytes_with_problem_shape(
|
||||
library::GemmDescription const &operation_desc,
|
||||
gemm::GemmCoord const &problem_shape) const;
|
||||
|
||||
int64_t flops_with_problem_shape(
|
||||
library::GemmDescription const &operation_desc,
|
||||
gemm::GemmCoord const &problem_shape) const;
|
||||
|
||||
/// Total number of bytes loaded
|
||||
int64_t bytes(library::GemmDescription const &operation_desc) const;
|
||||
|
||||
@@ -243,6 +260,26 @@ public:
|
||||
ProblemSpace::Problem const &problem);
|
||||
|
||||
protected:
|
||||
/// Update workspace configuration according to flexible user setups
|
||||
void update_workspace_(
|
||||
GemmWorkspace &gemm_workspace,
|
||||
gemm::GemmCoord const &problem_shape,
|
||||
std::array<int64_t, 3> const &leading_dim,
|
||||
std::array<int64_t, 3> const &preferred_cluster,
|
||||
std::array<int64_t, 3> const &fallback_cluster,
|
||||
cutlass::library::RasterOrder const &raster_order,
|
||||
int swizzle_size);
|
||||
|
||||
/// Update performance result configuration according to flexible user setups
|
||||
void update_result_(
|
||||
PerformanceResult &result,
|
||||
library::GemmDescription const &operation_desc,
|
||||
ProblemSpace const &problem_space,
|
||||
gemm::GemmCoord const &problem_shape,
|
||||
cutlass::library::RasterOrder const &raster_order,
|
||||
std::array<int64_t, 3> const &preferred_cluster,
|
||||
std::array<int64_t, 3> const &fallback_cluster,
|
||||
int swizzle_size);
|
||||
|
||||
/// Initializes the performance result
|
||||
void initialize_result_(
|
||||
|
||||
@@ -208,8 +208,23 @@ public:
|
||||
/// Minimum number of iterations to profile
|
||||
int min_iterations{10};
|
||||
|
||||
/// If true, profiling with cuda graph enabled.
|
||||
bool use_cuda_graphs{false};
|
||||
|
||||
/// If enabled, the CUTLASS profiler searches for the best-performing kernel
|
||||
/// within the subset of kernels matching a kernel filter regex. The best
|
||||
/// performance is determined by screening over a set of predefined M/N/K
|
||||
/// sizes and performance-related parameters, including cluster shapes,
|
||||
/// swizzle sizes, and rasterization orders.
|
||||
/// For now, it only supports legacy GEMM and blockscaled GEMM.
|
||||
bool enable_kernel_performance_search{false};
|
||||
|
||||
/// If enabled, the CUTLASS profiler searches for the best-performing kernel
|
||||
/// for a given M/N/K problem size by evaluating various performance-related
|
||||
/// parameters such as cluster shapes, swizzle sizes, and rasterization orders.
|
||||
/// For now, it only supports legacy GEMM and blockscaled GEMM.
|
||||
bool enable_best_kernel_for_fixed_shape{false};
|
||||
|
||||
/// Number of ms to sleep between profiling periods (ms)
|
||||
int sleep_duration{50};
|
||||
|
||||
@@ -264,8 +279,11 @@ public:
|
||||
/// Prints human-readable text to stdout. If false, nothing is written to stdout
|
||||
bool verbose;
|
||||
|
||||
/// Sort results by (currently by flops-per-byte)
|
||||
bool sort_results;
|
||||
/// Sort results by flops-per-byte
|
||||
bool sort_flops_per_byte;
|
||||
|
||||
/// Sort results by flops-per-second
|
||||
bool sort_flops_per_sec;
|
||||
|
||||
/// Prints the name of the kernel being profiled before running the kernel.
|
||||
/// This is useful for determining which kernel is causing a run of the profiler to hang
|
||||
|
||||
@@ -92,7 +92,8 @@ public:
|
||||
|
||||
void next_problem();
|
||||
void append_result(PerformanceResult result);
|
||||
void sort_results(PerformanceResultVector &results);
|
||||
void sort_flops_per_byte(PerformanceResultVector &results);
|
||||
void sort_flops_per_sec(PerformanceResultVector &results);
|
||||
void append_results(PerformanceResultVector const &results);
|
||||
|
||||
public:
|
||||
|
||||
@@ -105,6 +105,12 @@ struct PerformanceResult {
|
||||
runtime(0)
|
||||
{ }
|
||||
|
||||
// Copy constructor for deep copy
|
||||
PerformanceResult(const PerformanceResult& other) = default;
|
||||
|
||||
// Explicitly define copy assignment operator
|
||||
PerformanceResult& operator=(const PerformanceResult& other) = default;
|
||||
|
||||
/// Returns true if the runtime is valid
|
||||
bool good() const {
|
||||
return runtime > 0;
|
||||
|
||||
Reference in New Issue
Block a user