InfiniOps
Operator Library for Accelerators
Loading...
Searching...
No Matches
infini::ops::PagedAttentionInfinilm Class Referenceabstract

#include <paged_attention_infinilm.h>

Inheritance diagram for infini::ops::PagedAttentionInfinilm:
infini::ops::Operator< PagedAttentionInfinilm > infini::ops::OperatorBase infini::ops::OperatorBase

Public Member Functions

 PagedAttentionInfinilm (const Tensor q, const Tensor k_cache, const Tensor v_cache, const Tensor block_tables, const Tensor seq_lens, std::optional< Tensor > alibi_slopes, float scale, Tensor out)
 
virtual void operator() (const Tensor q, const Tensor k_cache, const Tensor v_cache, const Tensor block_tables, const Tensor seq_lens, std::optional< Tensor > alibi_slopes, float scale, Tensor out) const =0
 
- Public Member Functions inherited from infini::ops::Operator< PagedAttentionInfinilm >
void operator() (const Handle &handle, const Args &... args)
 
void operator() (const Args &... args) const
 
void operator() (const Handle &handle, const Args &... args)
 
void operator() (const Args &... args) const
 
- Public Member Functions inherited from infini::ops::OperatorBase
virtual ~OperatorBase ()=default
 
virtual std::size_t workspace_size_in_bytes () const
 
void set_handle (const Handle &handle)
 
void set_config (const Config &config)
 
void set_stream (void *stream)
 
void set_workspace (void *workspace)
 
void set_workspace_size_in_bytes (std::size_t workspace_size_in_bytes)
 
virtual ~OperatorBase ()=default
 
virtual std::size_t workspace_size_in_bytes () const
 
void set_handle (const Handle &handle)
 
void set_config (const Config &config)
 
void set_stream (void *stream)
 
void set_workspace (void *workspace)
 
void set_workspace_size_in_bytes (std::size_t workspace_size_in_bytes)
 

Static Protected Member Functions

static bool IsIndexDtype (DataType dtype)
 

Protected Attributes

DataType dtype_
 
DataType index_dtype_
 
float scale_ {1.0f}
 
std::size_t num_seqs_ {0}
 
std::size_t num_heads_ {0}
 
std::size_t num_kv_heads_ {0}
 
std::size_t head_size_ {0}
 
std::size_t block_size_ {0}
 
std::size_t max_num_blocks_per_seq_ {0}
 
Tensor::Stride q_stride_ {0}
 
Tensor::Stride q_head_stride_ {0}
 
Tensor::Stride k_cache_block_stride_ {0}
 
Tensor::Stride k_cache_head_stride_ {0}
 
Tensor::Stride k_cache_slot_stride_ {0}
 
Tensor::Stride v_cache_block_stride_ {0}
 
Tensor::Stride v_cache_head_stride_ {0}
 
Tensor::Stride v_cache_slot_stride_ {0}
 
Tensor::Stride out_stride_ {0}
 
Tensor::Stride out_head_stride_ {0}
 
Tensor::Stride block_table_batch_stride_ {0}
 
Tensor::Stride seq_lens_stride_ {0}
 
- Protected Attributes inherited from infini::ops::OperatorBase
std::unique_ptr< Handle > handle_ptr_
 
std::unique_ptr< Config > config_ptr_
 
void * stream_ {nullptr}
 
void * workspace_ {nullptr}
 
std::size_t workspace_size_in_bytes_ {0}
 

Additional Inherited Members

- Static Public Member Functions inherited from infini::ops::Operator< PagedAttentionInfinilm >
static void clear_cache ()
 
static void clear_cache ()
 
static std::unique_ptr< Operator > Make (const Config &config, const Tensor tensor, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Tensor tensor, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Config &config, const std::vector< Tensor > tensors, Args &&... args)
 
static std::unique_ptr< Operator > Make (const std::vector< Tensor > tensors, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Config &config, const Tensor tensor, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Tensor tensor, Args &&... args)
 
static std::unique_ptr< Operator > Make (const Config &config, const std::vector< Tensor > tensors, Args &&... args)
 
static std::unique_ptr< Operator > Make (const std::vector< Tensor > tensors, Args &&... args)
 
static void Call (const Handle &handle, const Config &config, const Args &... args)
 
static void Call (const Tensor tensor, const Args &... args)
 
static auto Call (const TensorLike &tensor, const Args &... args)
 
static void Call (const Handle &handle, const Config &config, const Args &... args)
 
static void Call (const Tensor tensor, const Args &... args)
 
static auto Call (const TensorLike &tensor, const Args &... args)
 
static std::vector< std::size_t > active_implementation_indices (Device::Type dev_type)
 
static std::vector< std::size_t > active_implementation_indices (Device::Type dev_type)
 
- Static Protected Attributes inherited from infini::ops::Operator< PagedAttentionInfinilm >
static constexpr Device::Type device_type_
 
static constexpr std::size_t implementation_index_
 

Detailed Description

Deprecated:
Migrate to an open-source-aligned operator when available. This interface will be removed in a future release.

Constructor & Destructor Documentation

◆ PagedAttentionInfinilm()

infini::ops::PagedAttentionInfinilm::PagedAttentionInfinilm ( const Tensor  q,
const Tensor  k_cache,
const Tensor  v_cache,
const Tensor  block_tables,
const Tensor  seq_lens,
std::optional< Tensor >  alibi_slopes,
float  scale,
Tensor  out 
)
inline

Member Function Documentation

◆ IsIndexDtype()

static bool infini::ops::PagedAttentionInfinilm::IsIndexDtype ( DataType  dtype)
inlinestaticprotected

◆ operator()()

virtual void infini::ops::PagedAttentionInfinilm::operator() ( const Tensor  q,
const Tensor  k_cache,
const Tensor  v_cache,
const Tensor  block_tables,
const Tensor  seq_lens,
std::optional< Tensor >  alibi_slopes,
float  scale,
Tensor  out 
) const
pure virtual

Member Data Documentation

◆ block_size_

std::size_t infini::ops::PagedAttentionInfinilm::block_size_ {0}
protected

◆ block_table_batch_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::block_table_batch_stride_ {0}
protected

◆ dtype_

DataType infini::ops::PagedAttentionInfinilm::dtype_
protected

◆ head_size_

std::size_t infini::ops::PagedAttentionInfinilm::head_size_ {0}
protected

◆ index_dtype_

DataType infini::ops::PagedAttentionInfinilm::index_dtype_
protected

◆ k_cache_block_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::k_cache_block_stride_ {0}
protected

◆ k_cache_head_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::k_cache_head_stride_ {0}
protected

◆ k_cache_slot_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::k_cache_slot_stride_ {0}
protected

◆ max_num_blocks_per_seq_

std::size_t infini::ops::PagedAttentionInfinilm::max_num_blocks_per_seq_ {0}
protected

◆ num_heads_

std::size_t infini::ops::PagedAttentionInfinilm::num_heads_ {0}
protected

◆ num_kv_heads_

std::size_t infini::ops::PagedAttentionInfinilm::num_kv_heads_ {0}
protected

◆ num_seqs_

std::size_t infini::ops::PagedAttentionInfinilm::num_seqs_ {0}
protected

◆ out_head_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::out_head_stride_ {0}
protected

◆ out_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::out_stride_ {0}
protected

◆ q_head_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::q_head_stride_ {0}
protected

◆ q_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::q_stride_ {0}
protected

◆ scale_

float infini::ops::PagedAttentionInfinilm::scale_ {1.0f}
protected

◆ seq_lens_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::seq_lens_stride_ {0}
protected

◆ v_cache_block_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::v_cache_block_stride_ {0}
protected

◆ v_cache_head_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::v_cache_head_stride_ {0}
protected

◆ v_cache_slot_stride_

Tensor::Stride infini::ops::PagedAttentionInfinilm::v_cache_slot_stride_ {0}
protected

The documentation for this class was generated from the following file: