minor chagnes (#730)

Co-authored-by: Haicheng Wu <haichengw@nvidia.com>
This commit is contained in:
Haicheng Wu
2022-12-10 14:44:53 -05:00
committed by GitHub
co-authored by Haicheng Wu
parent 38193d76e3
commit 3f2bb17722
3 changed files with 8 additions and 13 deletions
@@ -126,7 +126,7 @@ struct AttentionKernel {
struct Params {
// Input tensors
scalar_t* query_ptr; // [num_queries, num_heads, head_dim]
scalar_t* key_ptr; // [num_keys, num_heads, head_dim]
scalar_t* key_ptr; // [num_keys, num_heads, head_dim]
scalar_t* value_ptr; // [num_keys, num_heads, head_dim_value]
int32_t* cu_seqlens_q_ptr = nullptr;
int32_t* cu_seqlens_k_ptr = nullptr;
@@ -165,6 +165,7 @@ struct AttentionKernel {
CUTLASS_HOST_DEVICE int32_t o_strideM() const {
return head_dim_value;
}
// Moves pointers to what we should process
// Returns "false" if there is no work to do
CUTLASS_DEVICE bool advance_to_block() {