vllm/csrc/reduction_utils.cuh

/*
 * Adapted from https://github.com/NVIDIA/FasterTransformer/blob/release/v5.3_tag/src/fastertransformer/kernels/reduce_kernel_utils.cuh
 * Copyright (c) 2023, The vLLM team.
 * Copyright (c) 2020-2023, NVIDIA CORPORATION.  All rights reserved.
 *
 * Licensed under the Apache License, Version 2.0 (the "License");
 * you may not use this file except in compliance with the License.
 * You may obtain a copy of the License at
 *
 *     http://www.apache.org/licenses/LICENSE-2.0
 *
 * Unless required by applicable law or agreed to in writing, software
 * distributed under the License is distributed on an "AS IS" BASIS,
 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
 * See the License for the specific language governing permissions and
 * limitations under the License.
 */
#pragma once

#include "cuda_compat.h"

namespace vllm {

template<typename T>
__inline__ __device__ T warpReduceSum(T val) {
#pragma unroll
  for (int mask = WARP_SIZE/2; mask > 0; mask >>= 1)
    val += VLLM_SHFL_XOR_SYNC(val, mask);
  return val;
}

__inline__ __device__ constexpr int _calculateLaneMask(int warp_size) {
  return warp_size - 1;
}

__inline__ __device__ constexpr int _calculateWidShift(int warp_size) {
  return 5 + (warp_size >> 6);
}

/* Calculate the sum of all elements in a block */
template<typename T>
__inline__ __device__ T blockReduceSum(T val) {
  static __shared__ T shared[WARP_SIZE];
  constexpr auto LANE_MASK = _calculateLaneMask(WARP_SIZE);
  constexpr auto WID_SHIFT = _calculateWidShift(WARP_SIZE);
  int lane = threadIdx.x & LANE_MASK;
  int wid = threadIdx.x >> WID_SHIFT;

  val = warpReduceSum<T>(val);

  if (lane == 0)
    shared[wid] = val;

  __syncthreads();

  // Modify from blockDim.x << 5 to blockDim.x / 32. to prevent
  // blockDim.x is not divided by 32
  val = (threadIdx.x < (blockDim.x / (WARP_SIZE * 1.0f))) ? shared[lane] : (T)(0.0f);
  val = warpReduceSum<T>(val);
  return val;
}

} // namespace vllm
Add copyright headers to source files adapted from FT (#104) 2023-05-14 22:19:19 -07:00			`/*`
			`* Adapted from https://github.com/NVIDIA/FasterTransformer/blob/release/v5.3_tag/src/fastertransformer/kernels/reduce_kernel_utils.cuh`
Change the name to vLLM (#150) 2023-06-17 03:07:40 -07:00			`* Copyright (c) 2023, The vLLM team.`
Add copyright headers to source files adapted from FT (#104) 2023-05-14 22:19:19 -07:00			`* Copyright (c) 2020-2023, NVIDIA CORPORATION. All rights reserved.`
			`*`
			`* Licensed under the Apache License, Version 2.0 (the "License");`
			`* you may not use this file except in compliance with the License.`
			`* You may obtain a copy of the License at`
			`*`
			`* http://www.apache.org/licenses/LICENSE-2.0`
			`*`
			`* Unless required by applicable law or agreed to in writing, software`
			`* distributed under the License is distributed on an "AS IS" BASIS,`
			`* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.`
			`* See the License for the specific language governing permissions and`
			`* limitations under the License.`
			`*/`
Refactor attention kernels (#53) 2023-05-03 13:40:13 -07:00			`#pragma once`

Merge EmbeddedLLM/vllm-rocm into vLLM main (#1836) Co-authored-by: Philipp Moritz <pcmoritz@gmail.com> Co-authored-by: Amir Balwel <amoooori04@gmail.com> Co-authored-by: root <kuanfu.liu@akirakan.com> Co-authored-by: tjtanaa <tunjian.tan@embeddedllm.com> Co-authored-by: kuanfu <kuanfu.liu@embeddedllm.com> Co-authored-by: miloice <17350011+kliuae@users.noreply.github.com> 2023-12-08 15:16:52 +08:00			`#include "cuda_compat.h"`

Change the name to vLLM (#150) 2023-06-17 03:07:40 -07:00			`namespace vllm {`
Refactor attention kernels (#53) 2023-05-03 13:40:13 -07:00
			`template<typename T>`
			`__inline__ __device__ T warpReduceSum(T val) {`
			`#pragma unroll`
[ROCM] Fix blockReduceSum to use correct warp counts for ROCm and CUDA (#3262) 2024-03-10 17:27:45 -05:00			`for (int mask = WARP_SIZE/2; mask > 0; mask >>= 1)`
Merge EmbeddedLLM/vllm-rocm into vLLM main (#1836) Co-authored-by: Philipp Moritz <pcmoritz@gmail.com> Co-authored-by: Amir Balwel <amoooori04@gmail.com> Co-authored-by: root <kuanfu.liu@akirakan.com> Co-authored-by: tjtanaa <tunjian.tan@embeddedllm.com> Co-authored-by: kuanfu <kuanfu.liu@embeddedllm.com> Co-authored-by: miloice <17350011+kliuae@users.noreply.github.com> 2023-12-08 15:16:52 +08:00			`val += VLLM_SHFL_XOR_SYNC(val, mask);`
Refactor attention kernels (#53) 2023-05-03 13:40:13 -07:00			`return val;`
			`}`

[ROCm] Fix warp and lane calculation in blockReduceSum (#3321) 2024-03-12 04:14:07 +08:00			`__inline__ __device__ constexpr int _calculateLaneMask(int warp_size) {`
			`return warp_size - 1;`
			`}`

			`__inline__ __device__ constexpr int _calculateWidShift(int warp_size) {`
			`return 5 + (warp_size >> 6);`
			`}`

Refactor attention kernels (#53) 2023-05-03 13:40:13 -07:00			`/* Calculate the sum of all elements in a block */`
			`template<typename T>`
			`__inline__ __device__ T blockReduceSum(T val) {`
[ROCM] Fix blockReduceSum to use correct warp counts for ROCm and CUDA (#3262) 2024-03-10 17:27:45 -05:00			`static __shared__ T shared[WARP_SIZE];`
[ROCm] Fix warp and lane calculation in blockReduceSum (#3321) 2024-03-12 04:14:07 +08:00			`constexpr auto LANE_MASK = _calculateLaneMask(WARP_SIZE);`
			`constexpr auto WID_SHIFT = _calculateWidShift(WARP_SIZE);`
			`int lane = threadIdx.x & LANE_MASK;`
			`int wid = threadIdx.x >> WID_SHIFT;`
Refactor attention kernels (#53) 2023-05-03 13:40:13 -07:00
			`val = warpReduceSum<T>(val);`

			`if (lane == 0)`
			`shared[wid] = val;`

			`__syncthreads();`

			`// Modify from blockDim.x << 5 to blockDim.x / 32. to prevent`
			`// blockDim.x is not divided by 32`
[ROCM] Fix blockReduceSum to use correct warp counts for ROCm and CUDA (#3262) 2024-03-10 17:27:45 -05:00			`val = (threadIdx.x < (blockDim.x / (WARP_SIZE * 1.0f))) ? shared[lane] : (T)(0.0f);`
Refactor attention kernels (#53) 2023-05-03 13:40:13 -07:00			`val = warpReduceSum<T>(val);`
			`return val;`
			`}`

Change the name to vLLM (#150) 2023-06-17 03:07:40 -07:00			`} // namespace vllm`