mirror of
https://gitlab.com/libeigen/eigen.git
synced 2026-04-10 11:34:33 +08:00
Add workaround for using std::fma for scalar multiply-add.
This commit is contained in:
committed by
Rasmus Munk Larsen
parent
5996176b88
commit
ef3c5c1d1d
@@ -1004,8 +1004,7 @@ struct madd_impl {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
// Use FMA if there is a single CPU instruction.
|
#if EIGEN_SCALAR_MADD_USE_FMA
|
||||||
#ifdef EIGEN_VECTORIZE_FMA
|
|
||||||
template <typename Scalar>
|
template <typename Scalar>
|
||||||
struct madd_impl<Scalar, std::enable_if_t<has_fma<Scalar>::value>> {
|
struct madd_impl<Scalar, std::enable_if_t<has_fma<Scalar>::value>> {
|
||||||
static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar run(const Scalar& x, const Scalar& y, const Scalar& z) {
|
static EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar run(const Scalar& x, const Scalar& y, const Scalar& z) {
|
||||||
@@ -1927,7 +1926,6 @@ EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar arithmetic_shift_right(const Scalar
|
|||||||
return bit_cast<Scalar, SignedScalar>(bit_cast<SignedScalar, Scalar>(a) >> n);
|
return bit_cast<Scalar, SignedScalar>(bit_cast<SignedScalar, Scalar>(a) >> n);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Otherwise, rely on template implementation.
|
|
||||||
template <typename Scalar>
|
template <typename Scalar>
|
||||||
EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar fma(const Scalar& x, const Scalar& y, const Scalar& z) {
|
EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE Scalar fma(const Scalar& x, const Scalar& y, const Scalar& z) {
|
||||||
return internal::fma_impl<Scalar>::run(x, y, z);
|
return internal::fma_impl<Scalar>::run(x, y, z);
|
||||||
|
|||||||
@@ -52,6 +52,26 @@
|
|||||||
#define EIGEN_STACK_ALLOCATION_LIMIT 131072
|
#define EIGEN_STACK_ALLOCATION_LIMIT 131072
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
/* Specify whether to use std::fma for scalar multiply-add instructions.
|
||||||
|
*
|
||||||
|
* On machines that have FMA as a single instruction, this will generally
|
||||||
|
* improve precision without significant performance implications.
|
||||||
|
*
|
||||||
|
* Without a single instruction, performance has been found to be reduced 2-3x
|
||||||
|
* on Intel CPUs, and up to 30x for WASM.
|
||||||
|
*
|
||||||
|
* If unspecified, defaults to using FMA if hardware support is available.
|
||||||
|
* The default should be used in most cases to ensure consistency between
|
||||||
|
* vectorized and non-vectorized paths.
|
||||||
|
*/
|
||||||
|
#ifndef EIGEN_SCALAR_MADD_USE_FMA
|
||||||
|
#ifdef EIGEN_VECTORIZE_FMA
|
||||||
|
#define EIGEN_SCALAR_MADD_USE_FMA 1
|
||||||
|
#else
|
||||||
|
#define EIGEN_SCALAR_MADD_USE_FMA 0
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
|
||||||
//------------------------------------------------------------------------------------------
|
//------------------------------------------------------------------------------------------
|
||||||
// Compiler identification, EIGEN_COMP_*
|
// Compiler identification, EIGEN_COMP_*
|
||||||
//------------------------------------------------------------------------------------------
|
//------------------------------------------------------------------------------------------
|
||||||
|
|||||||
@@ -18,9 +18,6 @@ one option, and other parts (or libraries that you use) are compiled with anothe
|
|||||||
fail to link or exhibit subtle bugs. Nevertheless, these options can be useful for people who know what they
|
fail to link or exhibit subtle bugs. Nevertheless, these options can be useful for people who know what they
|
||||||
are doing.
|
are doing.
|
||||||
|
|
||||||
- \b EIGEN2_SUPPORT and \b EIGEN2_SUPPORT_STAGEnn_xxx are disabled starting from the 3.3 release.
|
|
||||||
Defining one of these will raise a compile-error. If you need to compile Eigen2 code,
|
|
||||||
<a href="http://eigen.tuxfamily.org/index.php?title=Eigen2">check this site</a>.
|
|
||||||
- \b EIGEN_DEFAULT_DENSE_INDEX_TYPE - the type for column and row indices in matrices, vectors and array
|
- \b EIGEN_DEFAULT_DENSE_INDEX_TYPE - the type for column and row indices in matrices, vectors and array
|
||||||
(DenseBase::Index). Set to \c std::ptrdiff_t by default.
|
(DenseBase::Index). Set to \c std::ptrdiff_t by default.
|
||||||
- \b EIGEN_DEFAULT_IO_FORMAT - the IOFormat to use when printing a matrix if no %IOFormat is specified.
|
- \b EIGEN_DEFAULT_IO_FORMAT - the IOFormat to use when printing a matrix if no %IOFormat is specified.
|
||||||
|
|||||||
Reference in New Issue
Block a user