vllm/csrc/attention.cpp

#include <torch/extension.h>
#include <c10/util/Optional.h>

void single_query_cached_kv_attention(
  torch::Tensor& out,
  torch::Tensor& query,
  torch::Tensor& key_cache,
  torch::Tensor& value_cache,
  float scale,
  torch::Tensor& block_tables,
  torch::Tensor& context_lens,
  int block_size,
  int max_context_len,
  const c10::optional<torch::Tensor>& alibi_slopes);

PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
  m.def(
    "single_query_cached_kv_attention",
    &single_query_cached_kv_attention,
    "Compute the attention between an input query and the cached key/value tensors");
}
Implement `single_query_cached_kv_attention` kernel (#3) 2023-03-01 15:02:19 -08:00			`#include <torch/extension.h>`
Add support for BLOOM (#331) 2023-07-03 13:12:35 -07:00			`#include <c10/util/Optional.h>`
Implement `single_query_cached_kv_attention` kernel (#3) 2023-03-01 15:02:19 -08:00
			`void single_query_cached_kv_attention(`
			`torch::Tensor& out,`
			`torch::Tensor& query,`
			`torch::Tensor& key_cache,`
			`torch::Tensor& value_cache,`
			`float scale,`
			`torch::Tensor& block_tables,`
			`torch::Tensor& context_lens,`
			`int block_size,`
Add support for BLOOM (#331) 2023-07-03 13:12:35 -07:00			`int max_context_len,`
			`const c10::optional<torch::Tensor>& alibi_slopes);`
Implement `single_query_cached_kv_attention` kernel (#3) 2023-03-01 15:02:19 -08:00
			`PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {`
			`m.def(`
			`"single_query_cached_kv_attention",`
			`&single_query_cached_kv_attention,`
			`"Compute the attention between an input query and the cached key/value tensors");`
			`}`