26 lines
929 B
C++
26 lines
929 B
C++
#include "core/registration.h"
|
|
#include "moe_ops.h"
|
|
|
|
TORCH_LIBRARY_EXPAND(TORCH_EXTENSION_NAME, m) {
|
|
// Apply topk softmax to the gating outputs.
|
|
m.def(
|
|
"topk_softmax(Tensor! topk_weights, Tensor! topk_indices, Tensor! "
|
|
"token_expert_indices, Tensor gating_output) -> ()");
|
|
m.impl("topk_softmax", torch::kCUDA, &topk_softmax);
|
|
|
|
#ifndef USE_ROCM
|
|
m.def(
|
|
"marlin_gemm_moe(Tensor! a, Tensor! b_q_weights, Tensor! sorted_ids, "
|
|
"Tensor! topk_weights, Tensor! topk_ids, Tensor! b_scales, Tensor! "
|
|
"b_zeros, Tensor! g_idx, Tensor! perm, Tensor! workspace, "
|
|
"int b_q_type, SymInt size_m, "
|
|
"SymInt size_n, SymInt size_k, bool is_k_full, int num_experts, int "
|
|
"topk, "
|
|
"int moe_block_size, bool replicate_input, bool apply_weights)"
|
|
" -> Tensor");
|
|
// conditionally compiled so impl registration is in source file
|
|
#endif
|
|
}
|
|
|
|
REGISTER_EXTENSION(TORCH_EXTENSION_NAME)
|