|
| | FlashAttnWithKvcache (const Tensor q, Tensor k_cache, Tensor v_cache, Tensor out) |
| |
| | FlashAttnWithKvcache (const Tensor q, Tensor k_cache, Tensor v_cache, const std::optional< Tensor > k, const std::optional< Tensor > v, const std::optional< Tensor > rotary_cos, const std::optional< Tensor > rotary_sin, const int64_t cache_seqlens, const std::optional< Tensor > cache_batch_idx, const std::optional< Tensor > cache_leftpad, const std::optional< Tensor > block_table, const std::optional< Tensor > alibi_slopes, const std::optional< double > softmax_scale, const bool causal, const std::vector< int64_t > window_size, const double softcap, const bool rotary_interleaved, const int64_t num_splits, const bool return_softmax_lse, Tensor out, std::optional< Tensor > softmax_lse) |
| |
| | FlashAttnWithKvcache (const Tensor q, Tensor k_cache, Tensor v_cache, const std::optional< Tensor > k, const std::optional< Tensor > v, const std::optional< Tensor > rotary_cos, const std::optional< Tensor > rotary_sin, const std::optional< Tensor > cache_seqlens, const std::optional< Tensor > cache_batch_idx, const std::optional< Tensor > cache_leftpad, const std::optional< Tensor > block_table, const std::optional< Tensor > alibi_slopes, const std::optional< double > softmax_scale, const bool causal, const std::vector< int64_t > window_size, const double softcap, const bool rotary_interleaved, const int64_t num_splits, const bool return_softmax_lse, Tensor out, std::optional< Tensor > softmax_lse) |
| |
| void | operator() (const Tensor q, Tensor k_cache, Tensor v_cache, Tensor out) const |
| |
| virtual void | operator() (const Tensor q, Tensor k_cache, Tensor v_cache, const std::optional< Tensor > k, const std::optional< Tensor > v, const std::optional< Tensor > rotary_cos, const std::optional< Tensor > rotary_sin, const int64_t cache_seqlens, const std::optional< Tensor > cache_batch_idx, const std::optional< Tensor > cache_leftpad, const std::optional< Tensor > block_table, const std::optional< Tensor > alibi_slopes, const std::optional< double > softmax_scale, const bool causal, const std::vector< int64_t > window_size, const double softcap, const bool rotary_interleaved, const int64_t num_splits, const bool return_softmax_lse, Tensor out, std::optional< Tensor > softmax_lse) const =0 |
| |
| virtual void | operator() (const Tensor q, Tensor k_cache, Tensor v_cache, const std::optional< Tensor > k, const std::optional< Tensor > v, const std::optional< Tensor > rotary_cos, const std::optional< Tensor > rotary_sin, const std::optional< Tensor > cache_seqlens, const std::optional< Tensor > cache_batch_idx, const std::optional< Tensor > cache_leftpad, const std::optional< Tensor > block_table, const std::optional< Tensor > alibi_slopes, const std::optional< double > softmax_scale, const bool causal, const std::vector< int64_t > window_size, const double softcap, const bool rotary_interleaved, const int64_t num_splits, const bool return_softmax_lse, Tensor out, std::optional< Tensor > softmax_lse) const =0 |
| |
| void | operator() (const Handle &handle, const Args &... args) |
| |
| void | operator() (const Args &... args) const |
| |
| void | operator() (const Handle &handle, const Args &... args) |
| |
| void | operator() (const Args &... args) const |
| |
| virtual | ~OperatorBase ()=default |
| |
| virtual std::size_t | workspace_size_in_bytes () const |
| |
| void | set_handle (const Handle &handle) |
| |
| void | set_config (const Config &config) |
| |
| void | set_stream (void *stream) |
| |
| void | set_workspace (void *workspace) |
| |
| void | set_workspace_size_in_bytes (std::size_t workspace_size_in_bytes) |
| |
| virtual | ~OperatorBase ()=default |
| |
| virtual std::size_t | workspace_size_in_bytes () const |
| |
| void | set_handle (const Handle &handle) |
| |
| void | set_config (const Config &config) |
| |
| void | set_stream (void *stream) |
| |
| void | set_workspace (void *workspace) |
| |
| void | set_workspace_size_in_bytes (std::size_t workspace_size_in_bytes) |
| |
|
| static void | clear_cache () |
| |
| static void | clear_cache () |
| |
| static std::unique_ptr< Operator > | Make (const Config &config, const Tensor tensor, Args &&... args) |
| |
| static std::unique_ptr< Operator > | Make (const Tensor tensor, Args &&... args) |
| |
| static std::unique_ptr< Operator > | Make (const Config &config, const std::vector< Tensor > tensors, Args &&... args) |
| |
| static std::unique_ptr< Operator > | Make (const std::vector< Tensor > tensors, Args &&... args) |
| |
| static std::unique_ptr< Operator > | Make (const Config &config, const Tensor tensor, Args &&... args) |
| |
| static std::unique_ptr< Operator > | Make (const Tensor tensor, Args &&... args) |
| |
| static std::unique_ptr< Operator > | Make (const Config &config, const std::vector< Tensor > tensors, Args &&... args) |
| |
| static std::unique_ptr< Operator > | Make (const std::vector< Tensor > tensors, Args &&... args) |
| |
| static void | Call (const Handle &handle, const Config &config, const Args &... args) |
| |
| static void | Call (const Tensor tensor, const Args &... args) |
| |
| static auto | Call (const TensorLike &tensor, const Args &... args) |
| |
| static void | Call (const Handle &handle, const Config &config, const Args &... args) |
| |
| static void | Call (const Tensor tensor, const Args &... args) |
| |
| static auto | Call (const TensorLike &tensor, const Args &... args) |
| |
| static std::vector< std::size_t > | active_implementation_indices (Device::Type dev_type) |
| |
| static std::vector< std::size_t > | active_implementation_indices (Device::Type dev_type) |
| |
| static constexpr Device::Type | device_type_ |
| |
| static constexpr std::size_t | implementation_index_ |
| |