mirror of
https://gitlab.com/libeigen/eigen.git
synced 2026-04-10 11:34:33 +08:00
Merging from eigen/eigen.
This commit is contained in:
@@ -12,17 +12,6 @@ include_directories(../../test ../../unsupported ../../Eigen
|
||||
|
||||
find_package (Threads)
|
||||
|
||||
find_package(Xsmm)
|
||||
if(XSMM_FOUND)
|
||||
add_definitions("-DEIGEN_USE_LIBXSMM")
|
||||
include_directories(${XSMM_INCLUDES})
|
||||
link_directories(${XSMM_LIBRARIES})
|
||||
set(EXTERNAL_LIBS ${EXTERNAL_LIBS} xsmm)
|
||||
ei_add_property(EIGEN_TESTED_BACKENDS "Xsmm, ")
|
||||
else(XSMM_FOUND)
|
||||
ei_add_property(EIGEN_MISSING_BACKENDS "Xsmm, ")
|
||||
endif(XSMM_FOUND)
|
||||
|
||||
find_package(GoogleHash)
|
||||
if(GOOGLEHASH_FOUND)
|
||||
add_definitions("-DEIGEN_GOOGLEHASH_SUPPORT")
|
||||
|
||||
@@ -511,8 +511,6 @@ static void test_const_inputs()
|
||||
VERIFY_IS_APPROX(mat3(1,1), mat1(1,0)*mat2(0,1) + mat1(1,1)*mat2(1,1) + mat1(1,2)*mat2(2,1));
|
||||
}
|
||||
|
||||
#if !defined(EIGEN_USE_LIBXSMM)
|
||||
|
||||
// Apply Sqrt to all output elements.
|
||||
struct SqrtOutputKernel {
|
||||
template <typename Index, typename Scalar>
|
||||
@@ -562,9 +560,6 @@ static void test_large_contraction_with_output_kernel() {
|
||||
}
|
||||
}
|
||||
|
||||
#endif // !defined(EIGEN_USE_LIBXSMM)
|
||||
|
||||
|
||||
EIGEN_DECLARE_TEST(cxx11_tensor_contraction)
|
||||
{
|
||||
CALL_SUBTEST(test_evals<ColMajor>());
|
||||
@@ -597,8 +592,6 @@ EIGEN_DECLARE_TEST(cxx11_tensor_contraction)
|
||||
CALL_SUBTEST(test_tensor_product<RowMajor>());
|
||||
CALL_SUBTEST(test_const_inputs<ColMajor>());
|
||||
CALL_SUBTEST(test_const_inputs<RowMajor>());
|
||||
#if !defined(EIGEN_USE_LIBXSMM)
|
||||
CALL_SUBTEST(test_large_contraction_with_output_kernel<ColMajor>());
|
||||
CALL_SUBTEST(test_large_contraction_with_output_kernel<RowMajor>());
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -562,37 +562,112 @@ static void test_execute_reverse_rvalue(Device d)
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int NumDims, typename Device, bool Vectorizable,
|
||||
bool Tileable, int Layout>
|
||||
static void test_async_execute_unary_expr(Device d)
|
||||
{
|
||||
static constexpr int Options = 0 | Layout;
|
||||
|
||||
// Pick a large enough tensor size to bypass small tensor block evaluation
|
||||
// optimization.
|
||||
auto dims = RandomDims<NumDims>(50 / NumDims, 100 / NumDims);
|
||||
|
||||
Tensor<T, NumDims, Options, Index> src(dims);
|
||||
Tensor<T, NumDims, Options, Index> dst(dims);
|
||||
|
||||
src.setRandom();
|
||||
const auto expr = src.square();
|
||||
|
||||
using Assign = TensorAssignOp<decltype(dst), const decltype(expr)>;
|
||||
using Executor = internal::TensorAsyncExecutor<const Assign, Device,
|
||||
Vectorizable, Tileable>;
|
||||
Eigen::Barrier done(1);
|
||||
Executor::runAsync(Assign(dst, expr), d, [&done]() { done.Notify(); });
|
||||
done.Wait();
|
||||
|
||||
for (Index i = 0; i < dst.dimensions().TotalSize(); ++i) {
|
||||
T square = src.coeff(i) * src.coeff(i);
|
||||
VERIFY_IS_EQUAL(square, dst.coeff(i));
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T, int NumDims, typename Device, bool Vectorizable,
|
||||
bool Tileable, int Layout>
|
||||
static void test_async_execute_binary_expr(Device d)
|
||||
{
|
||||
static constexpr int Options = 0 | Layout;
|
||||
|
||||
// Pick a large enough tensor size to bypass small tensor block evaluation
|
||||
// optimization.
|
||||
auto dims = RandomDims<NumDims>(50 / NumDims, 100 / NumDims);
|
||||
|
||||
Tensor<T, NumDims, Options, Index> lhs(dims);
|
||||
Tensor<T, NumDims, Options, Index> rhs(dims);
|
||||
Tensor<T, NumDims, Options, Index> dst(dims);
|
||||
|
||||
lhs.setRandom();
|
||||
rhs.setRandom();
|
||||
|
||||
const auto expr = lhs + rhs;
|
||||
|
||||
using Assign = TensorAssignOp<decltype(dst), const decltype(expr)>;
|
||||
using Executor = internal::TensorAsyncExecutor<const Assign, Device,
|
||||
Vectorizable, Tileable>;
|
||||
|
||||
Eigen::Barrier done(1);
|
||||
Executor::runAsync(Assign(dst, expr), d, [&done]() { done.Notify(); });
|
||||
done.Wait();
|
||||
|
||||
for (Index i = 0; i < dst.dimensions().TotalSize(); ++i) {
|
||||
T sum = lhs.coeff(i) + rhs.coeff(i);
|
||||
VERIFY_IS_EQUAL(sum, dst.coeff(i));
|
||||
}
|
||||
}
|
||||
|
||||
#ifdef EIGEN_DONT_VECTORIZE
|
||||
#define VECTORIZABLE(VAL) !EIGEN_DONT_VECTORIZE && VAL
|
||||
#else
|
||||
#else
|
||||
#define VECTORIZABLE(VAL) VAL
|
||||
#endif
|
||||
|
||||
#define CALL_SUBTEST_PART(PART) \
|
||||
CALL_SUBTEST_##PART
|
||||
|
||||
#define CALL_SUBTEST_COMBINATIONS(PART, NAME, T, NUM_DIMS) \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, false, false, ColMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, false, true, ColMajor>(default_device))); \
|
||||
#define CALL_SUBTEST_COMBINATIONS(PART, NAME, T, NUM_DIMS) \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, false, false, ColMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, false, true, ColMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, VECTORIZABLE(true), false, ColMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, VECTORIZABLE(true), true, ColMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, false, false, RowMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, false, true, RowMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, false, false, RowMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, false, true, RowMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, VECTORIZABLE(true), false, RowMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, DefaultDevice, VECTORIZABLE(true), true, RowMajor>(default_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, false, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, true, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, false, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, true, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, VECTORIZABLE(true), false, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, VECTORIZABLE(true), true, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, false, RowMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, true, RowMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, false, RowMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, true, RowMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, VECTORIZABLE(true), false, RowMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, VECTORIZABLE(true), true, RowMajor>(tp_device)))
|
||||
|
||||
// NOTE: Currently only ThreadPoolDevice supports async expression evaluation.
|
||||
#define CALL_ASYNC_SUBTEST_COMBINATIONS(PART, NAME, T, NUM_DIMS) \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, false, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, true, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, VECTORIZABLE(true), false, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, VECTORIZABLE(true), true, ColMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, false, RowMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, false, true, RowMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, VECTORIZABLE(true), false, RowMajor>(tp_device))); \
|
||||
CALL_SUBTEST_PART(PART)((NAME<T, NUM_DIMS, ThreadPoolDevice, VECTORIZABLE(true), true, RowMajor>(tp_device)))
|
||||
|
||||
EIGEN_DECLARE_TEST(cxx11_tensor_executor) {
|
||||
Eigen::DefaultDevice default_device;
|
||||
// Default device is unused in ASYNC tests.
|
||||
EIGEN_UNUSED_VARIABLE(default_device);
|
||||
|
||||
const auto num_threads = internal::random<int>(1, 24);
|
||||
const auto num_threads = internal::random<int>(20, 24);
|
||||
Eigen::ThreadPool tp(num_threads);
|
||||
Eigen::ThreadPoolDevice tp_device(&tp, num_threads);
|
||||
|
||||
@@ -660,8 +735,16 @@ EIGEN_DECLARE_TEST(cxx11_tensor_executor) {
|
||||
CALL_SUBTEST_COMBINATIONS(14, test_execute_reverse_rvalue, float, 4);
|
||||
CALL_SUBTEST_COMBINATIONS(14, test_execute_reverse_rvalue, float, 5);
|
||||
|
||||
CALL_ASYNC_SUBTEST_COMBINATIONS(15, test_async_execute_unary_expr, float, 3);
|
||||
CALL_ASYNC_SUBTEST_COMBINATIONS(15, test_async_execute_unary_expr, float, 4);
|
||||
CALL_ASYNC_SUBTEST_COMBINATIONS(15, test_async_execute_unary_expr, float, 5);
|
||||
|
||||
CALL_ASYNC_SUBTEST_COMBINATIONS(16, test_async_execute_binary_expr, float, 3);
|
||||
CALL_ASYNC_SUBTEST_COMBINATIONS(16, test_async_execute_binary_expr, float, 4);
|
||||
CALL_ASYNC_SUBTEST_COMBINATIONS(16, test_async_execute_binary_expr, float, 5);
|
||||
|
||||
// Force CMake to split this test.
|
||||
// EIGEN_SUFFIXES;1;2;3;4;5;6;7;8;9;10;11;12;13;14
|
||||
// EIGEN_SUFFIXES;1;2;3;4;5;6;7;8;9;10;11;12;13;14;15;16
|
||||
}
|
||||
|
||||
#undef CALL_SUBTEST_COMBINATIONS
|
||||
|
||||
@@ -19,8 +19,8 @@ static void test_0d()
|
||||
Tensor<int, 0> scalar1;
|
||||
Tensor<int, 0, RowMajor> scalar2;
|
||||
|
||||
TensorMap<Tensor<const int, 0> > scalar3(scalar1.data());
|
||||
TensorMap<Tensor<const int, 0, RowMajor> > scalar4(scalar2.data());
|
||||
TensorMap<const Tensor<int, 0> > scalar3(scalar1.data());
|
||||
TensorMap<const Tensor<int, 0, RowMajor> > scalar4(scalar2.data());
|
||||
|
||||
scalar1() = 7;
|
||||
scalar2() = 13;
|
||||
@@ -37,8 +37,8 @@ static void test_1d()
|
||||
Tensor<int, 1> vec1(6);
|
||||
Tensor<int, 1, RowMajor> vec2(6);
|
||||
|
||||
TensorMap<Tensor<const int, 1> > vec3(vec1.data(), 6);
|
||||
TensorMap<Tensor<const int, 1, RowMajor> > vec4(vec2.data(), 6);
|
||||
TensorMap<const Tensor<int, 1> > vec3(vec1.data(), 6);
|
||||
TensorMap<const Tensor<int, 1, RowMajor> > vec4(vec2.data(), 6);
|
||||
|
||||
vec1(0) = 4; vec2(0) = 0;
|
||||
vec1(1) = 8; vec2(1) = 1;
|
||||
@@ -85,8 +85,8 @@ static void test_2d()
|
||||
mat2(1,1) = 4;
|
||||
mat2(1,2) = 5;
|
||||
|
||||
TensorMap<Tensor<const int, 2> > mat3(mat1.data(), 2, 3);
|
||||
TensorMap<Tensor<const int, 2, RowMajor> > mat4(mat2.data(), 2, 3);
|
||||
TensorMap<const Tensor<int, 2> > mat3(mat1.data(), 2, 3);
|
||||
TensorMap<const Tensor<int, 2, RowMajor> > mat4(mat2.data(), 2, 3);
|
||||
|
||||
VERIFY_IS_EQUAL(mat3.rank(), 2);
|
||||
VERIFY_IS_EQUAL(mat3.size(), 6);
|
||||
@@ -129,8 +129,8 @@ static void test_3d()
|
||||
}
|
||||
}
|
||||
|
||||
TensorMap<Tensor<const int, 3> > mat3(mat1.data(), 2, 3, 7);
|
||||
TensorMap<Tensor<const int, 3, RowMajor> > mat4(mat2.data(), 2, 3, 7);
|
||||
TensorMap<const Tensor<int, 3> > mat3(mat1.data(), 2, 3, 7);
|
||||
TensorMap<const Tensor<int, 3, RowMajor> > mat4(mat2.data(), 2, 3, 7);
|
||||
|
||||
VERIFY_IS_EQUAL(mat3.rank(), 3);
|
||||
VERIFY_IS_EQUAL(mat3.size(), 2*3*7);
|
||||
@@ -265,6 +265,53 @@ static void test_casting()
|
||||
VERIFY_IS_EQUAL(sum1, 861);
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static const T& add_const(T& value) {
|
||||
return value;
|
||||
}
|
||||
|
||||
static void test_0d_const_tensor()
|
||||
{
|
||||
Tensor<int, 0> scalar1;
|
||||
Tensor<int, 0, RowMajor> scalar2;
|
||||
|
||||
TensorMap<const Tensor<int, 0> > scalar3(add_const(scalar1).data());
|
||||
TensorMap<const Tensor<int, 0, RowMajor> > scalar4(add_const(scalar2).data());
|
||||
|
||||
scalar1() = 7;
|
||||
scalar2() = 13;
|
||||
|
||||
VERIFY_IS_EQUAL(scalar1.rank(), 0);
|
||||
VERIFY_IS_EQUAL(scalar1.size(), 1);
|
||||
|
||||
VERIFY_IS_EQUAL(scalar3(), 7);
|
||||
VERIFY_IS_EQUAL(scalar4(), 13);
|
||||
}
|
||||
|
||||
static void test_0d_const_tensor_map()
|
||||
{
|
||||
Tensor<int, 0> scalar1;
|
||||
Tensor<int, 0, RowMajor> scalar2;
|
||||
|
||||
const TensorMap<Tensor<int, 0> > scalar3(scalar1.data());
|
||||
const TensorMap<Tensor<int, 0, RowMajor> > scalar4(scalar2.data());
|
||||
|
||||
// Although TensorMap is constant, we still can write to the underlying
|
||||
// storage, because we map over non-constant Tensor.
|
||||
scalar3() = 7;
|
||||
scalar4() = 13;
|
||||
|
||||
VERIFY_IS_EQUAL(scalar1(), 7);
|
||||
VERIFY_IS_EQUAL(scalar2(), 13);
|
||||
|
||||
// Pointer to the underlying storage is also non-const.
|
||||
scalar3.data()[0] = 8;
|
||||
scalar4.data()[0] = 14;
|
||||
|
||||
VERIFY_IS_EQUAL(scalar1(), 8);
|
||||
VERIFY_IS_EQUAL(scalar2(), 14);
|
||||
}
|
||||
|
||||
EIGEN_DECLARE_TEST(cxx11_tensor_map)
|
||||
{
|
||||
CALL_SUBTEST(test_0d());
|
||||
@@ -274,4 +321,7 @@ EIGEN_DECLARE_TEST(cxx11_tensor_map)
|
||||
|
||||
CALL_SUBTEST(test_from_tensor());
|
||||
CALL_SUBTEST(test_casting());
|
||||
|
||||
CALL_SUBTEST(test_0d_const_tensor());
|
||||
CALL_SUBTEST(test_0d_const_tensor_map());
|
||||
}
|
||||
|
||||
@@ -38,9 +38,9 @@ class TestAllocator : public Allocator {
|
||||
|
||||
void test_multithread_elementwise()
|
||||
{
|
||||
Tensor<float, 3> in1(2,3,7);
|
||||
Tensor<float, 3> in2(2,3,7);
|
||||
Tensor<float, 3> out(2,3,7);
|
||||
Tensor<float, 3> in1(200, 30, 70);
|
||||
Tensor<float, 3> in2(200, 30, 70);
|
||||
Tensor<float, 3> out(200, 30, 70);
|
||||
|
||||
in1.setRandom();
|
||||
in2.setRandom();
|
||||
@@ -49,15 +49,39 @@ void test_multithread_elementwise()
|
||||
Eigen::ThreadPoolDevice thread_pool_device(&tp, internal::random<int>(3, 11));
|
||||
out.device(thread_pool_device) = in1 + in2 * 3.14f;
|
||||
|
||||
for (int i = 0; i < 2; ++i) {
|
||||
for (int j = 0; j < 3; ++j) {
|
||||
for (int k = 0; k < 7; ++k) {
|
||||
VERIFY_IS_APPROX(out(i,j,k), in1(i,j,k) + in2(i,j,k) * 3.14f);
|
||||
for (int i = 0; i < 200; ++i) {
|
||||
for (int j = 0; j < 30; ++j) {
|
||||
for (int k = 0; k < 70; ++k) {
|
||||
VERIFY_IS_APPROX(out(i, j, k), in1(i, j, k) + in2(i, j, k) * 3.14f);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void test_async_multithread_elementwise()
|
||||
{
|
||||
Tensor<float, 3> in1(200, 30, 70);
|
||||
Tensor<float, 3> in2(200, 30, 70);
|
||||
Tensor<float, 3> out(200, 30, 70);
|
||||
|
||||
in1.setRandom();
|
||||
in2.setRandom();
|
||||
|
||||
Eigen::ThreadPool tp(internal::random<int>(3, 11));
|
||||
Eigen::ThreadPoolDevice thread_pool_device(&tp, internal::random<int>(3, 11));
|
||||
|
||||
Eigen::Barrier b(1);
|
||||
out.device(thread_pool_device, [&b]() { b.Notify(); }) = in1 + in2 * 3.14f;
|
||||
b.Wait();
|
||||
|
||||
for (int i = 0; i < 200; ++i) {
|
||||
for (int j = 0; j < 30; ++j) {
|
||||
for (int k = 0; k < 70; ++k) {
|
||||
VERIFY_IS_APPROX(out(i, j, k), in1(i, j, k) + in2(i, j, k) * 3.14f);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void test_multithread_compound_assignment()
|
||||
{
|
||||
@@ -306,6 +330,52 @@ static void test_multithread_contraction_with_output_kernel() {
|
||||
}
|
||||
}
|
||||
|
||||
template<int DataLayout>
|
||||
void test_async_multithread_contraction_agrees_with_singlethread()
|
||||
{
|
||||
int contract_size = internal::random<int>(100, 500);
|
||||
|
||||
Tensor<float, 3, DataLayout> left(internal::random<int>(10, 40),
|
||||
contract_size,
|
||||
internal::random<int>(10, 40));
|
||||
|
||||
Tensor<float, 4, DataLayout> right(
|
||||
internal::random<int>(1, 20), internal::random<int>(1, 20), contract_size,
|
||||
internal::random<int>(1, 20));
|
||||
|
||||
left.setRandom();
|
||||
right.setRandom();
|
||||
|
||||
// add constants to shift values away from 0 for more precision
|
||||
left += left.constant(1.5f);
|
||||
right += right.constant(1.5f);
|
||||
|
||||
typedef Tensor<float, 1>::DimensionPair DimPair;
|
||||
Eigen::array<DimPair, 1> dims({{DimPair(1, 2)}});
|
||||
|
||||
Eigen::ThreadPool tp(internal::random<int>(2, 11));
|
||||
Eigen::ThreadPoolDevice thread_pool_device(&tp, internal::random<int>(8, 32));
|
||||
|
||||
Tensor<float, 5, DataLayout> st_result;
|
||||
st_result = left.contract(right, dims);
|
||||
|
||||
Tensor<float, 5, DataLayout> tp_result(st_result.dimensions());
|
||||
|
||||
Eigen::Barrier barrier(1);
|
||||
tp_result.device(thread_pool_device, [&barrier]() { barrier.Notify(); }) =
|
||||
left.contract(right, dims);
|
||||
barrier.Wait();
|
||||
|
||||
VERIFY(dimensions_match(st_result.dimensions(), tp_result.dimensions()));
|
||||
for (ptrdiff_t i = 0; i < st_result.size(); i++) {
|
||||
// if both of the values are very small, then do nothing (because the test
|
||||
// will fail due to numerical precision issues when values are small)
|
||||
if (numext::abs(st_result.data()[i] - tp_result.data()[i]) >= 1e-4f) {
|
||||
VERIFY_IS_APPROX(st_result.data()[i], tp_result.data()[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// We are triggering 'evalShardedByInnerDim' optimization.
|
||||
template <int DataLayout>
|
||||
static void test_sharded_by_inner_dim_contraction()
|
||||
@@ -386,6 +456,93 @@ static void test_sharded_by_inner_dim_contraction_with_output_kernel()
|
||||
}
|
||||
}
|
||||
|
||||
// We are triggering 'evalShardedByInnerDim' optimization.
|
||||
template <int DataLayout>
|
||||
static void test_async_sharded_by_inner_dim_contraction()
|
||||
{
|
||||
typedef Tensor<float, 1>::DimensionPair DimPair;
|
||||
|
||||
const int num_threads = internal::random<int>(4, 16);
|
||||
ThreadPool threads(num_threads);
|
||||
Eigen::ThreadPoolDevice device(&threads, num_threads);
|
||||
|
||||
Tensor<float, 2, DataLayout> t_left(2, 10000);
|
||||
Tensor<float, 2, DataLayout> t_right(10000, 10);
|
||||
Tensor<float, 2, DataLayout> t_result(2, 10);
|
||||
|
||||
t_left.setRandom();
|
||||
t_right.setRandom();
|
||||
// Put trash in t_result to verify contraction clears output memory.
|
||||
t_result.setRandom();
|
||||
|
||||
// Add a little offset so that the results won't be close to zero.
|
||||
t_left += t_left.constant(1.0f);
|
||||
t_right += t_right.constant(1.0f);
|
||||
|
||||
typedef Map<Eigen::Matrix<float, Dynamic, Dynamic, DataLayout>> MapXf;
|
||||
MapXf m_left(t_left.data(), 2, 10000);
|
||||
MapXf m_right(t_right.data(), 10000, 10);
|
||||
Eigen::Matrix<float, Dynamic, Dynamic, DataLayout> m_result(2, 10);
|
||||
|
||||
// this contraction should be equivalent to a single matrix multiplication
|
||||
Eigen::array<DimPair, 1> dims({{DimPair(1, 0)}});
|
||||
|
||||
// compute results by separate methods
|
||||
Eigen::Barrier barrier(1);
|
||||
t_result.device(device, [&barrier]() { barrier.Notify(); }) =
|
||||
t_left.contract(t_right, dims);
|
||||
barrier.Wait();
|
||||
|
||||
m_result = m_left * m_right;
|
||||
|
||||
for (Index i = 0; i < t_result.dimensions().TotalSize(); i++) {
|
||||
VERIFY_IS_APPROX(t_result.data()[i], m_result.data()[i]);
|
||||
}
|
||||
}
|
||||
|
||||
// We are triggering 'evalShardedByInnerDim' optimization with output kernel.
|
||||
template <int DataLayout>
|
||||
static void test_async_sharded_by_inner_dim_contraction_with_output_kernel()
|
||||
{
|
||||
typedef Tensor<float, 1>::DimensionPair DimPair;
|
||||
|
||||
const int num_threads = internal::random<int>(4, 16);
|
||||
ThreadPool threads(num_threads);
|
||||
Eigen::ThreadPoolDevice device(&threads, num_threads);
|
||||
|
||||
Tensor<float, 2, DataLayout> t_left(2, 10000);
|
||||
Tensor<float, 2, DataLayout> t_right(10000, 10);
|
||||
Tensor<float, 2, DataLayout> t_result(2, 10);
|
||||
|
||||
t_left.setRandom();
|
||||
t_right.setRandom();
|
||||
// Put trash in t_result to verify contraction clears output memory.
|
||||
t_result.setRandom();
|
||||
|
||||
// Add a little offset so that the results won't be close to zero.
|
||||
t_left += t_left.constant(1.0f);
|
||||
t_right += t_right.constant(1.0f);
|
||||
|
||||
typedef Map<Eigen::Matrix<float, Dynamic, Dynamic, DataLayout>> MapXf;
|
||||
MapXf m_left(t_left.data(), 2, 10000);
|
||||
MapXf m_right(t_right.data(), 10000, 10);
|
||||
Eigen::Matrix<float, Dynamic, Dynamic, DataLayout> m_result(2, 10);
|
||||
|
||||
// this contraction should be equivalent to a single matrix multiplication
|
||||
Eigen::array<DimPair, 1> dims({{DimPair(1, 0)}});
|
||||
|
||||
// compute results by separate methods
|
||||
Eigen::Barrier barrier(1);
|
||||
t_result.device(device, [&barrier]() { barrier.Notify(); }) =
|
||||
t_left.contract(t_right, dims, SqrtOutputKernel());
|
||||
barrier.Wait();
|
||||
m_result = m_left * m_right;
|
||||
|
||||
for (Index i = 0; i < t_result.dimensions().TotalSize(); i++) {
|
||||
VERIFY_IS_APPROX(t_result.data()[i], std::sqrt(m_result.data()[i]));
|
||||
}
|
||||
}
|
||||
|
||||
template<int DataLayout>
|
||||
void test_full_contraction() {
|
||||
int contract_size1 = internal::random<int>(1, 500);
|
||||
@@ -516,6 +673,7 @@ void test_threadpool_allocate(TestAllocator* allocator)
|
||||
EIGEN_DECLARE_TEST(cxx11_tensor_thread_pool)
|
||||
{
|
||||
CALL_SUBTEST_1(test_multithread_elementwise());
|
||||
CALL_SUBTEST_1(test_async_multithread_elementwise());
|
||||
CALL_SUBTEST_1(test_multithread_compound_assignment());
|
||||
|
||||
CALL_SUBTEST_2(test_multithread_contraction<ColMajor>());
|
||||
@@ -525,11 +683,18 @@ EIGEN_DECLARE_TEST(cxx11_tensor_thread_pool)
|
||||
CALL_SUBTEST_3(test_multithread_contraction_agrees_with_singlethread<RowMajor>());
|
||||
CALL_SUBTEST_3(test_multithread_contraction_with_output_kernel<ColMajor>());
|
||||
CALL_SUBTEST_3(test_multithread_contraction_with_output_kernel<RowMajor>());
|
||||
CALL_SUBTEST_3(test_async_multithread_contraction_agrees_with_singlethread<ColMajor>());
|
||||
CALL_SUBTEST_3(test_async_multithread_contraction_agrees_with_singlethread<RowMajor>());
|
||||
|
||||
// Test EvalShardedByInnerDimContext parallelization strategy.
|
||||
CALL_SUBTEST_4(test_sharded_by_inner_dim_contraction<ColMajor>());
|
||||
CALL_SUBTEST_4(test_sharded_by_inner_dim_contraction<RowMajor>());
|
||||
CALL_SUBTEST_4(test_sharded_by_inner_dim_contraction_with_output_kernel<ColMajor>());
|
||||
CALL_SUBTEST_4(test_sharded_by_inner_dim_contraction_with_output_kernel<RowMajor>());
|
||||
CALL_SUBTEST_4(test_async_sharded_by_inner_dim_contraction<ColMajor>());
|
||||
CALL_SUBTEST_4(test_async_sharded_by_inner_dim_contraction<RowMajor>());
|
||||
CALL_SUBTEST_4(test_async_sharded_by_inner_dim_contraction_with_output_kernel<ColMajor>());
|
||||
CALL_SUBTEST_4(test_async_sharded_by_inner_dim_contraction_with_output_kernel<RowMajor>());
|
||||
|
||||
// Exercise various cases that have been problematic in the past.
|
||||
CALL_SUBTEST_5(test_contraction_corner_cases<ColMajor>());
|
||||
|
||||
Reference in New Issue
Block a user