CUTLASS 3.3.0 (#1167)

* Release 3.3.0

Adds support for mixed precision GEMMs On Hopper and Ampere
Adds support for < 16B aligned GEMMs on Hopper
Enhancements to EVT
Enhancements to Python interface
Enhancements to Sub-byte type handling in CuTe
Several other bug-fixes and performance improvements.

* minor doc update
This commit is contained in:
Pradeep Ramani
2023-11-02 08:09:05 -07:00
committed by GitHub
parent 922fb5108b
commit c008b4aea8
263 changed files with 16214 additions and 5008 deletions

View File

@@ -45,6 +45,7 @@
#pragma once
#if !defined(__CUDACC_RTC__)
#include "cuda.h"
#include "cuda_runtime.h"
#include "cutlass/trace.h"
@@ -55,16 +56,19 @@
namespace cutlass {
/////////////////////////////////////////////////////////////////////////////////////////////////
static constexpr int MinWorkspaceAlignment = 16;
#if !defined(__CUDACC_RTC__)
static Status
zero_workspace(void* workspace, int workspace_size, cudaStream_t stream = nullptr) {
zero_workspace(void* workspace, size_t workspace_size, cudaStream_t stream = nullptr) {
if (workspace_size > 0) {
if (workspace == nullptr) {
CUTLASS_TRACE_HOST(" error: device workspace must not be null");
return Status::kErrorWorkspaceNull;
}
CUTLASS_TRACE_HOST(" clearing barrier workspace");
CUTLASS_TRACE_HOST(" clearing workspace");
cudaError_t result = cudaMemsetAsync(workspace, 0, workspace_size, stream);
if (cudaSuccess != result) {
result = cudaGetLastError(); // to clear the error bit
@@ -77,6 +81,47 @@ zero_workspace(void* workspace, int workspace_size, cudaStream_t stream = nullpt
}
#endif
#if !defined(__CUDACC_RTC__)
template <typename T>
Status
fill_workspace(void* workspace, T fill_value, size_t fill_count, cudaStream_t stream = nullptr) {
static_assert(sizeof(T) == 4 || sizeof(T) == 2 || sizeof(T) == 1, "Unsupported fill type");
if (fill_count > 0) {
if (workspace == nullptr) {
CUTLASS_TRACE_HOST(" error: device workspace must not be null");
return Status::kErrorWorkspaceNull;
}
CUTLASS_TRACE_HOST(" filling workspace");
CUdeviceptr d_workspace = reinterpret_cast<CUdeviceptr>(workspace);
CUresult result = CUDA_SUCCESS;
if (sizeof(T) == 4) {
result = cuMemsetD32Async(d_workspace, reinterpret_cast<uint32_t&>(fill_value), fill_count, stream);
}
else if (sizeof(T) == 2) {
result = cuMemsetD16Async(d_workspace, reinterpret_cast<uint16_t&>(fill_value), fill_count, stream);
}
else if (sizeof(T) == 1) {
result = cuMemsetD8Async(d_workspace, reinterpret_cast<uint8_t&>(fill_value), fill_count, stream);
}
if (CUDA_SUCCESS != result) {
const char** error_string_ptr = nullptr;
(void) cuGetErrorString(result, error_string_ptr);
if (error_string_ptr != nullptr) {
CUTLASS_TRACE_HOST(" cuMemsetD" << sizeof(T) * 8 << "Async() returned error " << *error_string_ptr);
}
else {
CUTLASS_TRACE_HOST(" cuMemsetD" << sizeof(T) * 8 << "Async() returned unrecognized error");
}
return Status::kErrorInternal;
}
}
return Status::kSuccess;
}
#endif
/////////////////////////////////////////////////////////////////////////////////////////////////
} // namespace cutlass