InfiniOps
Operator Library for Accelerators
Loading...
Searching...
No Matches
infini::ops::FlashAttention Class Referenceabstract

#include <flash_attention.h>

Inheritance diagram for infini::ops::FlashAttention:
infini::ops::Operator< FlashAttention > infini::ops::OperatorBase infini::ops::OperatorBase

Public Member Functions

 FlashAttention (const Tensor query, const Tensor key, const Tensor value, std::optional< Tensor > cu_seqlens_q, std::optional< Tensor > cu_seqlens_kv, std::optional< Tensor > block_table, int64_t num_heads, int64_t num_kv_heads, int64_t head_size, double scale, bool causal, int64_t window_left, int64_t window_right, int64_t block_size, Tensor output)
 
virtual void operator() (const Tensor query, const Tensor key, const Tensor value, std::optional< Tensor > cu_seqlens_q, std::optional< Tensor > cu_seqlens_kv, std::optional< Tensor > block_table, int64_t num_heads, int64_t num_kv_heads, int64_t head_size, double scale, bool causal, int64_t window_left, int64_t window_right, int64_t block_size, Tensor output) const =0
 
- Public Member Functions inherited from infini::ops::Operator< FlashAttention >
void operator() (const Handle &handle, const Args &... args)
 
void operator() (const Args &... args) const
 
void operator() (const Handle &handle, const Args &... args)
 
void operator() (const Args &... args) const
 
- Public Member Functions inherited from infini::ops::OperatorBase
virtual ~OperatorBase ()=default
 
virtual std::size_t workspace_size_in_bytes () const
 
void set_handle (const Handle &handle)
 
void set_config (const Config &config)
 
void set_stream (void *stream)
 
void set_workspace (void *workspace)
 
void set_workspace_size_in_bytes (std::size_t workspace_size_in_bytes)
 
virtual ~OperatorBase ()=default
 
virtual std::size_t workspace_size_in_bytes () const
 
void set_handle (const Handle &handle)
 
void set_config (const Config &config)
 
void set_stream (void *stream)
 
void set_workspace (void *workspace)
 
void set_workspace_size_in_bytes (std::size_t workspace_size_in_bytes)
 

Protected Attributes

Tensor::Size num_tokens_ {0}
 
int64_t num_heads_ {0}
 
int64_t num_kv_heads_ {0}
 
int64_t head_size_ {0}
 
double scale_ {0.0}
 
bool causal_ {false}
 
int64_t window_left_ {-1}
 
int64_t window_right_ {-1}
 
int64_t block_size_ {0}
 
const DataType dtype_
 
Tensor::Shape query_shape_
 
Tensor::Shape key_shape_
 
Tensor::Shape value_shape_
 
Tensor::Shape output_shape_
 
Tensor::Strides query_strides_
 
Tensor::Strides key_strides_
 
Tensor::Strides value_strides_
 
Tensor::Strides output_strides_
 
bool has_cu_seqlens_q_ {false}
 
bool has_cu_seqlens_kv_ {false}
 
bool has_block_table_ {false}
 
- Protected Attributes inherited from infini::ops::OperatorBase
std::unique_ptr< Handle > handle_ptr_
 
std::unique_ptr< Config > config_ptr_
 
void * stream_ {nullptr}
 
void * workspace_ {nullptr}
 
std::size_t workspace_size_in_bytes_ {0}
 

Additional Inherited Members

- Static Public Member Functions inherited from infini::ops::Operator< FlashAttention >
static void clear_cache ()
 
static void clear_cache ()
 
static std::unique_ptr< Operator > Make (const Config &config, const Tensor tensor, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Tensor tensor, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Config &config, const std::vector< Tensor > tensors, Args &&... args)
 
static std::unique_ptr< Operator > Make (const std::vector< Tensor > tensors, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Config &config, const Tensor tensor, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Tensor tensor, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Config &config, const std::vector< Tensor > tensors, Args &&... args)
 
static std::unique_ptr< Operator > Make (const std::vector< Tensor > tensors, Args &&... args)
 
static void Call (const Handle &handle, const Config &config, const Args &... args)
 
static void Call (const Tensor tensor, const Args &... args)
 
static auto Call (const TensorLike &tensor, const Args &... args)
 
static void Call (const Handle &handle, const Config &config, const Args &... args)
 
static void Call (const Tensor tensor, const Args &... args)
 
static auto Call (const TensorLike &tensor, const Args &... args)
 
static std::vector< std::size_t > active_implementation_indices (Device::Type dev_type)
 
static std::vector< std::size_t > active_implementation_indices (Device::Type dev_type)
 
- Static Protected Attributes inherited from infini::ops::Operator< FlashAttention >
static constexpr Device::Type device_type_
 
static constexpr std::size_t implementation_index_
 

Detailed Description

Deprecated:
Use ScaledDotProductAttention for standard attention semantics. This interface will be removed in a future release.

Constructor & Destructor Documentation

◆ FlashAttention()

infini::ops::FlashAttention::FlashAttention ( const Tensor  query,
const Tensor  key,
const Tensor  value,
std::optional< Tensor >  cu_seqlens_q,
std::optional< Tensor >  cu_seqlens_kv,
std::optional< Tensor >  block_table,
int64_t  num_heads,
int64_t  num_kv_heads,
int64_t  head_size,
double  scale,
bool  causal,
int64_t  window_left,
int64_t  window_right,
int64_t  block_size,
Tensor  output 
)
inline

Member Function Documentation

◆ operator()()

virtual void infini::ops::FlashAttention::operator() ( const Tensor  query,
const Tensor  key,
const Tensor  value,
std::optional< Tensor >  cu_seqlens_q,
std::optional< Tensor >  cu_seqlens_kv,
std::optional< Tensor >  block_table,
int64_t  num_heads,
int64_t  num_kv_heads,
int64_t  head_size,
double  scale,
bool  causal,
int64_t  window_left,
int64_t  window_right,
int64_t  block_size,
Tensor  output 
) const
pure virtual

Member Data Documentation

◆ block_size_

int64_t infini::ops::FlashAttention::block_size_ {0}
protected

◆ causal_

bool infini::ops::FlashAttention::causal_ {false}
protected

◆ dtype_

const DataType infini::ops::FlashAttention::dtype_
protected

◆ has_block_table_

bool infini::ops::FlashAttention::has_block_table_ {false}
protected

◆ has_cu_seqlens_kv_

bool infini::ops::FlashAttention::has_cu_seqlens_kv_ {false}
protected

◆ has_cu_seqlens_q_

bool infini::ops::FlashAttention::has_cu_seqlens_q_ {false}
protected

◆ head_size_

int64_t infini::ops::FlashAttention::head_size_ {0}
protected

◆ key_shape_

Tensor::Shape infini::ops::FlashAttention::key_shape_
protected

◆ key_strides_

Tensor::Strides infini::ops::FlashAttention::key_strides_
protected

◆ num_heads_

int64_t infini::ops::FlashAttention::num_heads_ {0}
protected

◆ num_kv_heads_

int64_t infini::ops::FlashAttention::num_kv_heads_ {0}
protected

◆ num_tokens_

Tensor::Size infini::ops::FlashAttention::num_tokens_ {0}
protected

◆ output_shape_

Tensor::Shape infini::ops::FlashAttention::output_shape_
protected

◆ output_strides_

Tensor::Strides infini::ops::FlashAttention::output_strides_
protected

◆ query_shape_

Tensor::Shape infini::ops::FlashAttention::query_shape_
protected

◆ query_strides_

Tensor::Strides infini::ops::FlashAttention::query_strides_
protected

◆ scale_

double infini::ops::FlashAttention::scale_ {0.0}
protected

◆ value_shape_

Tensor::Shape infini::ops::FlashAttention::value_shape_
protected

◆ value_strides_

Tensor::Strides infini::ops::FlashAttention::value_strides_
protected

◆ window_left_

int64_t infini::ops::FlashAttention::window_left_ {-1}
protected

◆ window_right_

int64_t infini::ops::FlashAttention::window_right_ {-1}
protected

The documentation for this class was generated from the following file: