mirror of
https://gitlab.com/libeigen/eigen.git
synced 2026-04-10 11:34:33 +08:00
Disable cuda Eigen::half vectorization on host.
All cuda `__half` functions are device-only in CUDA 9, including
conversions. Host-side conversions were added in CUDA 10.
The existing code doesn't build prior to 10.0.
All arithmetic functions are always device-only, so there's
therefore no reason to use vectorization on the host at all.
Modified the code to disable vectorization for `__half` on host,
which required also updating the `TensorReductionGpu` implementation
which previously made assumptions about available packets.
(cherry picked from commit cc3573ab44)
This commit is contained in:
committed by
Antonio Sánchez
parent
277d369060
commit
c2b6df6e60
@@ -52,7 +52,7 @@ struct PacketType : internal::packet_traits<Scalar> {
|
||||
};
|
||||
|
||||
// For CUDA packet types when using a GpuDevice
|
||||
#if defined(EIGEN_USE_GPU) && defined(EIGEN_HAS_GPU_FP16)
|
||||
#if defined(EIGEN_USE_GPU) && defined(EIGEN_HAS_GPU_FP16) && defined(EIGEN_GPU_COMPILE_PHASE)
|
||||
|
||||
typedef ulonglong2 Packet4h2;
|
||||
template<>
|
||||
|
||||
@@ -98,6 +98,7 @@ __device__ inline void atomicReduce(half2* output, half2 accum, R& reducer) {
|
||||
}
|
||||
}
|
||||
}
|
||||
#ifdef EIGEN_GPU_COMPILE_PHASE
|
||||
// reduction should be associative since reduction is not atomic in wide vector but atomic in half2 operations
|
||||
template <typename R>
|
||||
__device__ inline void atomicReduce(Packet4h2* output, Packet4h2 accum, R& reducer) {
|
||||
@@ -107,6 +108,7 @@ __device__ inline void atomicReduce(Packet4h2* output, Packet4h2 accum, R& reduc
|
||||
atomicReduce(houtput+i,*(haccum+i),reducer);
|
||||
}
|
||||
}
|
||||
#endif // EIGEN_GPU_COMPILE_PHASE
|
||||
#endif // EIGEN_HAS_GPU_FP16
|
||||
|
||||
template <>
|
||||
@@ -213,8 +215,8 @@ __global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void FullReductionKernel(Reducer reducer
|
||||
#ifdef EIGEN_HAS_GPU_FP16
|
||||
template <typename Self,
|
||||
typename Reducer, typename Index>
|
||||
__global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void ReductionInitFullReduxKernelHalfFloat(Reducer reducer, const Self input, Index num_coeffs,
|
||||
packet_traits<Eigen::half>::type* scratch) {
|
||||
__global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void ReductionInitFullReduxKernelHalfFloat(
|
||||
Reducer reducer, const Self input, Index num_coeffs, half* scratch) {
|
||||
eigen_assert(blockDim.x == 1);
|
||||
eigen_assert(gridDim.x == 1);
|
||||
typedef packet_traits<Eigen::half>::type packet_type;
|
||||
@@ -224,15 +226,16 @@ __global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void ReductionInitFullReduxKernelHalfFlo
|
||||
half2* h2scratch = reinterpret_cast<half2*>(scratch);
|
||||
for (Index i = num_coeffs - packet_remainder; i + 2 <= num_coeffs; i += 2) {
|
||||
*h2scratch =
|
||||
__halves2half2(input.m_impl.coeff(i), input.m_impl.coeff(i + 1));
|
||||
__halves2half2(input.coeff(i), input.coeff(i + 1));
|
||||
h2scratch++;
|
||||
}
|
||||
if ((num_coeffs & 1) != 0) {
|
||||
half lastCoeff = input.m_impl.coeff(num_coeffs - 1);
|
||||
half lastCoeff = input.coeff(num_coeffs - 1);
|
||||
*h2scratch = __halves2half2(lastCoeff, reducer.initialize());
|
||||
}
|
||||
} else {
|
||||
*scratch = reducer.template initializePacket<packet_type>();
|
||||
packet_type reduce = reducer.template initializePacket<packet_type>();
|
||||
internal::pstoreu(scratch, reduce);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -258,8 +261,9 @@ __global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void ReductionInitKernelHalfFloat(Reduce
|
||||
|
||||
template <int BlockSize, int NumPerThread, typename Self,
|
||||
typename Reducer, typename Index>
|
||||
__global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void FullReductionKernelHalfFloat(Reducer reducer, const Self input, Index num_coeffs,
|
||||
half* output, packet_traits<Eigen::half>::type* scratch) {
|
||||
__global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void FullReductionKernelHalfFloat(
|
||||
Reducer reducer, const Self input, Index num_coeffs,
|
||||
half* output, half* scratch) {
|
||||
typedef typename packet_traits<Eigen::half>::type PacketType;
|
||||
const int packet_width = unpacket_traits<PacketType>::size;
|
||||
eigen_assert(NumPerThread % packet_width == 0);
|
||||
@@ -273,19 +277,20 @@ __global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void FullReductionKernelHalfFloat(Reduce
|
||||
int rem = num_coeffs % packet_width;
|
||||
if (rem != 0) {
|
||||
half2* p_scratch = reinterpret_cast<half2*>(scratch);
|
||||
*scratch = reducer.template initializePacket<PacketType>();
|
||||
pstoreu(scratch, reducer.template initializePacket<PacketType>());
|
||||
for (int i = 0; i < rem / 2; i++) {
|
||||
*p_scratch = __halves2half2(
|
||||
input.m_impl.coeff(num_coeffs - packet_width + 2 * i),
|
||||
input.m_impl.coeff(num_coeffs - packet_width + 2 * i + 1));
|
||||
input.coeff(num_coeffs - packet_width + 2 * i),
|
||||
input.coeff(num_coeffs - packet_width + 2 * i + 1));
|
||||
p_scratch++;
|
||||
}
|
||||
if ((num_coeffs & 1) != 0) {
|
||||
half last = input.m_impl.coeff(num_coeffs - 1);
|
||||
half last = input.coeff(num_coeffs - 1);
|
||||
*p_scratch = __halves2half2(last, reducer.initialize());
|
||||
}
|
||||
} else {
|
||||
*scratch = reducer.template initializePacket<PacketType>();
|
||||
PacketType reduce = reducer.template initializePacket<PacketType>();
|
||||
pstoreu(scratch, reduce);
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
@@ -298,7 +303,7 @@ __global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void FullReductionKernelHalfFloat(Reduce
|
||||
for (Index i = 0; i < max_iter; i += BlockSize) {
|
||||
const Index index = first_index + packet_width * i;
|
||||
eigen_assert(index + packet_width < num_coeffs);
|
||||
PacketType val = input.m_impl.template packet<Unaligned>(index);
|
||||
PacketType val = input.template packet<Unaligned>(index);
|
||||
reducer.reducePacket(val, &accum);
|
||||
}
|
||||
|
||||
@@ -337,7 +342,7 @@ __global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void FullReductionKernelHalfFloat(Reduce
|
||||
}
|
||||
|
||||
if ((threadIdx.x & (warpSize - 1)) == 0) {
|
||||
atomicReduce(scratch, accum, reducer);
|
||||
atomicReduce(reinterpret_cast<PacketType*>(scratch), accum, reducer);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
@@ -357,17 +362,21 @@ __global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void FullReductionKernelHalfFloat(Reduce
|
||||
}
|
||||
|
||||
template <typename Op>
|
||||
__global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void ReductionCleanupKernelHalfFloat(Op reducer, half* output, packet_traits<Eigen::half>::type* scratch) {
|
||||
__global__ EIGEN_HIP_LAUNCH_BOUNDS_1024 void ReductionCleanupKernelHalfFloat(Op reducer, half* output, half* scratch) {
|
||||
eigen_assert(threadIdx.x == 1);
|
||||
half2* pscratch = reinterpret_cast<half2*>(scratch);
|
||||
half tmp = __float2half(0.f);
|
||||
typedef packet_traits<Eigen::half>::type packet_type;
|
||||
for (int i = 0; i < unpacket_traits<packet_type>::size; i += 2) {
|
||||
reducer.reduce(__low2half(*pscratch), &tmp);
|
||||
reducer.reduce(__high2half(*pscratch), &tmp);
|
||||
pscratch++;
|
||||
if (unpacket_traits<packet_type>::size == 1) {
|
||||
*output = *scratch;
|
||||
} else {
|
||||
half2* pscratch = reinterpret_cast<half2*>(scratch);
|
||||
half tmp = __float2half(0.f);
|
||||
for (int i = 0; i < unpacket_traits<packet_type>::size; i += 2) {
|
||||
reducer.reduce(__low2half(*pscratch), &tmp);
|
||||
reducer.reduce(__high2half(*pscratch), &tmp);
|
||||
pscratch++;
|
||||
}
|
||||
*output = tmp;
|
||||
}
|
||||
*output = tmp;
|
||||
}
|
||||
|
||||
#endif // EIGEN_HAS_GPU_FP16
|
||||
@@ -416,13 +425,11 @@ template <typename Self, typename Op>
|
||||
struct FullReductionLauncher<Self, Op, Eigen::half, true> {
|
||||
static void run(const Self& self, Op& reducer, const GpuDevice& device, half* output, typename Self::Index num_coeffs) {
|
||||
typedef typename Self::Index Index;
|
||||
typedef typename packet_traits<Eigen::half>::type PacketType;
|
||||
|
||||
const int block_size = 256;
|
||||
const int num_per_thread = 128;
|
||||
const int num_blocks = divup<int>(num_coeffs, block_size * num_per_thread);
|
||||
PacketType* scratch = static_cast<PacketType*>(device.scratchpad());
|
||||
// half2* scratch = static_cast<half2*>(device.scratchpad());
|
||||
half* scratch = static_cast<half*>(device.scratchpad());
|
||||
|
||||
if (num_blocks > 1) {
|
||||
// We initialize the output and the scrathpad outside the reduction kernel when we can't be sure that there
|
||||
|
||||
Reference in New Issue
Block a user