mirror of
https://gitlab.com/libeigen/eigen.git
synced 2026-04-10 11:34:33 +08:00
GPU: Add dense cuSOLVER solvers (QR, SVD, EigenSolver)
Add QR (geqrf + ormqr + trsm), SVD (gesvd), and self-adjoint eigenvalue decomposition (syevd) via cuSOLVER. All support host and DeviceMatrix input. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -528,7 +528,7 @@ if(CUDA_FOUND AND EIGEN_TEST_CUDA)
|
||||
# compiler and linked against CUDA runtime + cuSOLVER. This avoids NVCC
|
||||
# instantiating Eigen's CPU packet operations for CUDA vector types.
|
||||
unset(EIGEN_ADD_TEST_FILENAME_EXTENSION)
|
||||
foreach(_cusolver_test IN ITEMS gpu_cusolver_llt gpu_cusolver_lu)
|
||||
foreach(_cusolver_test IN ITEMS gpu_cusolver_llt gpu_cusolver_lu gpu_cusolver_qr gpu_cusolver_svd gpu_cusolver_eigen)
|
||||
add_executable(${_cusolver_test} ${_cusolver_test}.cpp)
|
||||
target_include_directories(${_cusolver_test} PRIVATE
|
||||
"${CUDA_TOOLKIT_ROOT_DIR}/include"
|
||||
|
||||
@@ -16,6 +16,32 @@
|
||||
|
||||
using namespace Eigen;
|
||||
|
||||
// Unit roundoff for GPU GEMM compute precision.
|
||||
// TF32 (opt-in via EIGEN_CUDA_TF32) has eps ~ 2^{-10}.
|
||||
template <typename Scalar>
|
||||
typename NumTraits<Scalar>::Real gpu_unit_roundoff() {
|
||||
#if defined(EIGEN_CUDA_TF32) && !defined(EIGEN_NO_CUDA_TENSOR_OPS)
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
if (std::is_same<RealScalar, float>::value) return RealScalar(9.8e-4);
|
||||
#endif
|
||||
return NumTraits<Scalar>::epsilon();
|
||||
}
|
||||
|
||||
// Higham-Mary probabilistic error bound for GEMM:
|
||||
// ||C - fl(C)||_F <= lambda * sqrt(k) * u * ||A||_F * ||B||_F
|
||||
// where k is the inner dimension, u is the unit roundoff, and
|
||||
// lambda = sqrt(2 * ln(2/delta)) with delta = failure probability.
|
||||
// lambda = 5 corresponds to delta ~ 10^{-6}.
|
||||
// Reference: Higham & Mary, "Probabilistic Error Analysis for Inner Products",
|
||||
// SIAM J. Matrix Anal. Appl., 2019.
|
||||
template <typename Scalar>
|
||||
typename NumTraits<Scalar>::Real gemm_error_bound(Index k, typename NumTraits<Scalar>::Real normA,
|
||||
typename NumTraits<Scalar>::Real normB) {
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
constexpr RealScalar lambda = 5;
|
||||
return lambda * std::sqrt(static_cast<RealScalar>(k)) * gpu_unit_roundoff<Scalar>() * normA * normB;
|
||||
}
|
||||
|
||||
// ---- Basic GEMM: C = A * B -------------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
@@ -36,7 +62,7 @@ void test_gemm_basic(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = A * B;
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -59,7 +85,7 @@ void test_gemm_adjoint_lhs(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = A.adjoint() * B;
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -82,7 +108,7 @@ void test_gemm_transpose_rhs(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = A * B.transpose();
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -106,7 +132,7 @@ void test_gemm_scaled(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = alpha * A * B;
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -130,7 +156,7 @@ void test_gemm_accumulate(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = C_init + A * B;
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -153,7 +179,7 @@ void test_gemm_accumulate_empty(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = A * B;
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -178,7 +204,7 @@ void test_gemm_subtract(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = C_init - A * B;
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -202,7 +228,7 @@ void test_gemm_subtract_empty(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = -(A * B);
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -226,7 +252,7 @@ void test_gemm_scaled_rhs(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = A * (alpha * B);
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -266,7 +292,7 @@ void test_gemm_explicit_context(Index m, Index n, Index k) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = A * B;
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -296,7 +322,7 @@ void test_gemm_cross_context_reuse(Index n) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = A * B + D * E;
|
||||
|
||||
RealScalar tol = RealScalar(2) * RealScalar(n) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(n, A.norm(), B.norm()) + gemm_error_bound<Scalar>(n, D.norm(), E.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -326,7 +352,7 @@ void test_gemm_cross_context_resize() {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = D * E;
|
||||
|
||||
RealScalar tol = RealScalar(16) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(16, D.norm(), E.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -353,7 +379,9 @@ void test_gemm_chain(Index n) {
|
||||
Mat D = d_D.toHost();
|
||||
Mat D_ref = (A * B) * E;
|
||||
|
||||
RealScalar tol = RealScalar(2) * RealScalar(n) * NumTraits<Scalar>::epsilon() * D_ref.norm();
|
||||
Mat C_ref = A * B;
|
||||
RealScalar tol =
|
||||
gemm_error_bound<Scalar>(n, A.norm(), B.norm()) * E.norm() + gemm_error_bound<Scalar>(n, C_ref.norm(), E.norm());
|
||||
VERIFY((D - D_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -401,7 +429,7 @@ void test_llt_solve_expr(Index n, Index nrhs) {
|
||||
|
||||
Mat X = d_X.toHost();
|
||||
RealScalar residual = (A * X - B).norm() / B.norm();
|
||||
VERIFY(residual < RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
VERIFY(residual < RealScalar(n) * gpu_unit_roundoff<Scalar>());
|
||||
}
|
||||
|
||||
// ---- LLT solve with explicit context ----------------------------------------
|
||||
@@ -423,7 +451,7 @@ void test_llt_solve_expr_context(Index n, Index nrhs) {
|
||||
|
||||
Mat X = d_X.toHost();
|
||||
RealScalar residual = (A * X - B).norm() / B.norm();
|
||||
VERIFY(residual < RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
VERIFY(residual < RealScalar(n) * gpu_unit_roundoff<Scalar>());
|
||||
}
|
||||
|
||||
// ---- LU solve expression: d_X = d_A.lu().solve(d_B) ------------------------
|
||||
@@ -444,7 +472,7 @@ void test_lu_solve_expr(Index n, Index nrhs) {
|
||||
|
||||
Mat X = d_X.toHost();
|
||||
RealScalar residual = (A * X - B).norm() / (A.norm() * X.norm());
|
||||
VERIFY(residual < RealScalar(10) * RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
VERIFY(residual < RealScalar(10) * RealScalar(n) * gpu_unit_roundoff<Scalar>());
|
||||
}
|
||||
|
||||
// ---- GEMM + solver chain: C = A * B, X = C.llt().solve(D) ------------------
|
||||
@@ -474,7 +502,7 @@ void test_gemm_then_solve(Index n) {
|
||||
|
||||
Mat X = d_X.toHost();
|
||||
RealScalar residual = (C * X - D).norm() / D.norm();
|
||||
VERIFY(residual < RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
VERIFY(residual < RealScalar(n) * gpu_unit_roundoff<Scalar>());
|
||||
}
|
||||
|
||||
// ---- LLT solve with Upper triangle -----------------------------------------
|
||||
@@ -495,7 +523,7 @@ void test_llt_solve_upper(Index n, Index nrhs) {
|
||||
|
||||
Mat X = d_X.toHost();
|
||||
RealScalar residual = (A * X - B).norm() / B.norm();
|
||||
VERIFY(residual < RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
VERIFY(residual < RealScalar(n) * gpu_unit_roundoff<Scalar>());
|
||||
}
|
||||
|
||||
// ---- LU solve with explicit context -----------------------------------------
|
||||
@@ -517,7 +545,7 @@ void test_lu_solve_expr_context(Index n, Index nrhs) {
|
||||
|
||||
Mat X = d_X.toHost();
|
||||
RealScalar residual = (A * X - B).norm() / (A.norm() * X.norm());
|
||||
VERIFY(residual < RealScalar(10) * RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
VERIFY(residual < RealScalar(10) * RealScalar(n) * gpu_unit_roundoff<Scalar>());
|
||||
}
|
||||
|
||||
// ---- Zero-nrhs solver expressions ------------------------------------------
|
||||
@@ -581,7 +609,7 @@ void test_trsm(Index n, Index nrhs) {
|
||||
|
||||
Mat X = d_X.toHost();
|
||||
RealScalar residual = (A * X - B).norm() / B.norm();
|
||||
VERIFY(residual < RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
VERIFY(residual < RealScalar(n) * gpu_unit_roundoff<Scalar>());
|
||||
}
|
||||
|
||||
// ---- SYMM/HEMM: selfadjointView<UpLo>() * B --------------------------------
|
||||
@@ -603,7 +631,7 @@ void test_symm(Index n, Index nrhs) {
|
||||
Mat C = d_C.toHost();
|
||||
Mat C_ref = A * B; // A is symmetric, so full multiply == symm
|
||||
|
||||
RealScalar tol = RealScalar(n) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(n, A.norm(), B.norm());
|
||||
VERIFY((C - C_ref).norm() < tol);
|
||||
}
|
||||
|
||||
@@ -629,7 +657,7 @@ void test_syrk(Index n, Index k) {
|
||||
Mat C_lower = C.template triangularView<Lower>();
|
||||
Mat C_ref_lower = C_ref.template triangularView<Lower>();
|
||||
|
||||
RealScalar tol = RealScalar(k) * NumTraits<Scalar>::epsilon() * C_ref.norm();
|
||||
RealScalar tol = gemm_error_bound<Scalar>(k, A.norm(), A.norm());
|
||||
VERIFY((C_lower - C_ref_lower).norm() < tol);
|
||||
}
|
||||
|
||||
|
||||
180
test/gpu_cusolver_eigen.cpp
Normal file
180
test/gpu_cusolver_eigen.cpp
Normal file
@@ -0,0 +1,180 @@
|
||||
// This file is part of Eigen, a lightweight C++ template library
|
||||
// for linear algebra.
|
||||
//
|
||||
// Copyright (C) 2026 Rasmus Munk Larsen <rmlarsen@gmail.com>
|
||||
//
|
||||
// This Source Code Form is subject to the terms of the Mozilla
|
||||
// Public License v. 2.0. If a copy of the MPL was not distributed
|
||||
// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
// Tests for GpuSelfAdjointEigenSolver: GPU symmetric/Hermitian eigenvalue
|
||||
// decomposition via cuSOLVER.
|
||||
|
||||
#define EIGEN_USE_GPU
|
||||
#include "main.h"
|
||||
#include <Eigen/Eigenvalues>
|
||||
#include <Eigen/GPU>
|
||||
|
||||
using namespace Eigen;
|
||||
|
||||
// ---- Reconstruction: V * diag(W) * V^H ≈ A ---------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_eigen_reconstruction(Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
// Build a symmetric/Hermitian matrix.
|
||||
Mat R = Mat::Random(n, n);
|
||||
Mat A = R + R.adjoint();
|
||||
|
||||
GpuSelfAdjointEigenSolver<Scalar> es(A);
|
||||
VERIFY_IS_EQUAL(es.info(), Success);
|
||||
|
||||
auto W = es.eigenvalues();
|
||||
Mat V = es.eigenvectors();
|
||||
|
||||
VERIFY_IS_EQUAL(W.size(), n);
|
||||
VERIFY_IS_EQUAL(V.rows(), n);
|
||||
VERIFY_IS_EQUAL(V.cols(), n);
|
||||
|
||||
// Reconstruct: A_hat = V * diag(W) * V^H.
|
||||
Mat A_hat = V * W.asDiagonal() * V.adjoint();
|
||||
RealScalar tol = RealScalar(5) * std::sqrt(static_cast<RealScalar>(n)) * NumTraits<Scalar>::epsilon() * A.norm();
|
||||
VERIFY((A_hat - A).norm() < tol);
|
||||
|
||||
// Orthogonality: V^H * V ≈ I.
|
||||
Mat VhV = V.adjoint() * V;
|
||||
Mat eye = Mat::Identity(n, n);
|
||||
VERIFY((VhV - eye).norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Eigenvalues match CPU SelfAdjointEigenSolver ---------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_eigen_values(Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat R = Mat::Random(n, n);
|
||||
Mat A = R + R.adjoint();
|
||||
|
||||
GpuSelfAdjointEigenSolver<Scalar> gpu_es(A);
|
||||
VERIFY_IS_EQUAL(gpu_es.info(), Success);
|
||||
auto W_gpu = gpu_es.eigenvalues();
|
||||
|
||||
SelfAdjointEigenSolver<Mat> cpu_es(A);
|
||||
auto W_cpu = cpu_es.eigenvalues();
|
||||
|
||||
RealScalar tol = RealScalar(5) * std::sqrt(static_cast<RealScalar>(n)) * NumTraits<Scalar>::epsilon() *
|
||||
W_cpu.cwiseAbs().maxCoeff();
|
||||
VERIFY((W_gpu - W_cpu).norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Eigenvalues-only mode --------------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_eigen_values_only(Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat R = Mat::Random(n, n);
|
||||
Mat A = R + R.adjoint();
|
||||
|
||||
GpuSelfAdjointEigenSolver<Scalar> gpu_es(A, GpuSelfAdjointEigenSolver<Scalar>::EigenvaluesOnly);
|
||||
VERIFY_IS_EQUAL(gpu_es.info(), Success);
|
||||
auto W_gpu = gpu_es.eigenvalues();
|
||||
|
||||
SelfAdjointEigenSolver<Mat> cpu_es(A, EigenvaluesOnly);
|
||||
auto W_cpu = cpu_es.eigenvalues();
|
||||
|
||||
RealScalar tol = RealScalar(5) * std::sqrt(static_cast<RealScalar>(n)) * NumTraits<Scalar>::epsilon() *
|
||||
W_cpu.cwiseAbs().maxCoeff();
|
||||
VERIFY((W_gpu - W_cpu).norm() < tol);
|
||||
}
|
||||
|
||||
// ---- DeviceMatrix input path ------------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_eigen_device_matrix(Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat R = Mat::Random(n, n);
|
||||
Mat A = R + R.adjoint();
|
||||
|
||||
auto d_A = DeviceMatrix<Scalar>::fromHost(A);
|
||||
GpuSelfAdjointEigenSolver<Scalar> es;
|
||||
es.compute(d_A);
|
||||
VERIFY_IS_EQUAL(es.info(), Success);
|
||||
|
||||
auto W_gpu = es.eigenvalues();
|
||||
Mat V = es.eigenvectors();
|
||||
|
||||
// Verify reconstruction.
|
||||
Mat A_hat = V * W_gpu.asDiagonal() * V.adjoint();
|
||||
RealScalar tol = RealScalar(5) * std::sqrt(static_cast<RealScalar>(n)) * NumTraits<Scalar>::epsilon() * A.norm();
|
||||
VERIFY((A_hat - A).norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Recompute (reuse solver object) ----------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_eigen_recompute(Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
GpuSelfAdjointEigenSolver<Scalar> es;
|
||||
|
||||
for (int trial = 0; trial < 3; ++trial) {
|
||||
Mat R = Mat::Random(n, n);
|
||||
Mat A = R + R.adjoint();
|
||||
es.compute(A);
|
||||
VERIFY_IS_EQUAL(es.info(), Success);
|
||||
|
||||
auto W = es.eigenvalues();
|
||||
Mat V = es.eigenvectors();
|
||||
Mat A_hat = V * W.asDiagonal() * V.adjoint();
|
||||
RealScalar tol = RealScalar(5) * std::sqrt(static_cast<RealScalar>(n)) * NumTraits<Scalar>::epsilon() * A.norm();
|
||||
VERIFY((A_hat - A).norm() < tol);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- Empty matrix -----------------------------------------------------------
|
||||
|
||||
void test_eigen_empty() {
|
||||
GpuSelfAdjointEigenSolver<double> es(MatrixXd(0, 0));
|
||||
VERIFY_IS_EQUAL(es.info(), Success);
|
||||
VERIFY_IS_EQUAL(es.rows(), 0);
|
||||
VERIFY_IS_EQUAL(es.cols(), 0);
|
||||
}
|
||||
|
||||
// ---- Per-scalar driver ------------------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_scalar() {
|
||||
// Reconstruction + orthogonality.
|
||||
CALL_SUBTEST(test_eigen_reconstruction<Scalar>(64));
|
||||
CALL_SUBTEST(test_eigen_reconstruction<Scalar>(128));
|
||||
|
||||
// Eigenvalues match CPU.
|
||||
CALL_SUBTEST(test_eigen_values<Scalar>(64));
|
||||
CALL_SUBTEST(test_eigen_values<Scalar>(128));
|
||||
|
||||
// Values-only mode.
|
||||
CALL_SUBTEST(test_eigen_values_only<Scalar>(64));
|
||||
|
||||
// DeviceMatrix input.
|
||||
CALL_SUBTEST(test_eigen_device_matrix<Scalar>(64));
|
||||
|
||||
// Recompute.
|
||||
CALL_SUBTEST(test_eigen_recompute<Scalar>(32));
|
||||
}
|
||||
|
||||
EIGEN_DECLARE_TEST(gpu_cusolver_eigen) {
|
||||
CALL_SUBTEST(test_scalar<float>());
|
||||
CALL_SUBTEST(test_scalar<double>());
|
||||
CALL_SUBTEST(test_scalar<std::complex<float>>());
|
||||
CALL_SUBTEST(test_scalar<std::complex<double>>());
|
||||
CALL_SUBTEST(test_eigen_empty());
|
||||
}
|
||||
185
test/gpu_cusolver_qr.cpp
Normal file
185
test/gpu_cusolver_qr.cpp
Normal file
@@ -0,0 +1,185 @@
|
||||
// This file is part of Eigen, a lightweight C++ template library
|
||||
// for linear algebra.
|
||||
//
|
||||
// Copyright (C) 2026 Rasmus Munk Larsen <rmlarsen@gmail.com>
|
||||
//
|
||||
// This Source Code Form is subject to the terms of the Mozilla
|
||||
// Public License v. 2.0. If a copy of the MPL was not distributed
|
||||
// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
// Tests for GpuQR: GPU QR decomposition via cuSOLVER.
|
||||
|
||||
#define EIGEN_USE_GPU
|
||||
#include "main.h"
|
||||
#include <Eigen/QR>
|
||||
#include <Eigen/GPU>
|
||||
|
||||
using namespace Eigen;
|
||||
|
||||
// ---- Solve square system: A * X = B -----------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_qr_solve_square(Index n, Index nrhs) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(n, n);
|
||||
Mat B = Mat::Random(n, nrhs);
|
||||
|
||||
GpuQR<Scalar> qr(A);
|
||||
VERIFY_IS_EQUAL(qr.info(), Success);
|
||||
|
||||
Mat X = qr.solve(B);
|
||||
RealScalar residual = (A * X - B).norm() / (A.norm() * X.norm());
|
||||
VERIFY(residual < RealScalar(10) * RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
}
|
||||
|
||||
// ---- Solve overdetermined system: m > n (least-squares) ---------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_qr_solve_overdetermined(Index m, Index n, Index nrhs) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
eigen_assert(m >= n);
|
||||
Mat A = Mat::Random(m, n);
|
||||
Mat B = Mat::Random(m, nrhs);
|
||||
|
||||
GpuQR<Scalar> qr(A);
|
||||
VERIFY_IS_EQUAL(qr.info(), Success);
|
||||
|
||||
Mat X = qr.solve(B);
|
||||
VERIFY_IS_EQUAL(X.rows(), n);
|
||||
VERIFY_IS_EQUAL(X.cols(), nrhs);
|
||||
|
||||
// Compare with CPU QR.
|
||||
Mat X_cpu = HouseholderQR<Mat>(A).solve(B);
|
||||
RealScalar tol = RealScalar(100) * RealScalar(m) * NumTraits<Scalar>::epsilon();
|
||||
VERIFY((X - X_cpu).norm() / X_cpu.norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Solve with DeviceMatrix input ------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_qr_solve_device(Index n, Index nrhs) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(n, n);
|
||||
Mat B = Mat::Random(n, nrhs);
|
||||
|
||||
auto d_A = DeviceMatrix<Scalar>::fromHost(A);
|
||||
auto d_B = DeviceMatrix<Scalar>::fromHost(B);
|
||||
|
||||
GpuQR<Scalar> qr;
|
||||
qr.compute(d_A);
|
||||
VERIFY_IS_EQUAL(qr.info(), Success);
|
||||
|
||||
DeviceMatrix<Scalar> d_X = qr.solve(d_B);
|
||||
Mat X = d_X.toHost();
|
||||
|
||||
RealScalar residual = (A * X - B).norm() / (A.norm() * X.norm());
|
||||
VERIFY(residual < RealScalar(10) * RealScalar(n) * NumTraits<Scalar>::epsilon());
|
||||
}
|
||||
|
||||
// ---- Solve overdetermined via device path -----------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_qr_solve_overdetermined_device(Index m, Index n, Index nrhs) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
eigen_assert(m >= n);
|
||||
Mat A = Mat::Random(m, n);
|
||||
Mat B = Mat::Random(m, nrhs);
|
||||
|
||||
auto d_A = DeviceMatrix<Scalar>::fromHost(A);
|
||||
auto d_B = DeviceMatrix<Scalar>::fromHost(B);
|
||||
|
||||
GpuQR<Scalar> qr;
|
||||
qr.compute(d_A);
|
||||
VERIFY_IS_EQUAL(qr.info(), Success);
|
||||
|
||||
DeviceMatrix<Scalar> d_X = qr.solve(d_B);
|
||||
VERIFY_IS_EQUAL(d_X.rows(), n);
|
||||
VERIFY_IS_EQUAL(d_X.cols(), nrhs);
|
||||
|
||||
Mat X = d_X.toHost();
|
||||
Mat X_cpu = HouseholderQR<Mat>(A).solve(B);
|
||||
RealScalar tol = RealScalar(100) * RealScalar(m) * NumTraits<Scalar>::epsilon();
|
||||
VERIFY((X - X_cpu).norm() / X_cpu.norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Multiple solves reuse the factorization --------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_qr_multiple_solves(Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(n, n);
|
||||
GpuQR<Scalar> qr(A);
|
||||
VERIFY_IS_EQUAL(qr.info(), Success);
|
||||
|
||||
RealScalar tol = RealScalar(10) * RealScalar(n) * NumTraits<Scalar>::epsilon();
|
||||
for (int k = 0; k < 5; ++k) {
|
||||
Mat B = Mat::Random(n, 3);
|
||||
Mat X = qr.solve(B);
|
||||
RealScalar residual = (A * X - B).norm() / (A.norm() * X.norm());
|
||||
VERIFY(residual < tol);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- Agreement with CPU HouseholderQR ---------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_qr_vs_cpu(Index n, Index nrhs) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(n, n);
|
||||
Mat B = Mat::Random(n, nrhs);
|
||||
|
||||
GpuQR<Scalar> gpu_qr(A);
|
||||
VERIFY_IS_EQUAL(gpu_qr.info(), Success);
|
||||
|
||||
Mat X_gpu = gpu_qr.solve(B);
|
||||
Mat X_cpu = HouseholderQR<Mat>(A).solve(B);
|
||||
|
||||
RealScalar tol = RealScalar(100) * RealScalar(n) * NumTraits<Scalar>::epsilon();
|
||||
VERIFY((X_gpu - X_cpu).norm() / X_cpu.norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Per-scalar driver ------------------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_scalar() {
|
||||
CALL_SUBTEST(test_qr_solve_square<Scalar>(1, 1));
|
||||
CALL_SUBTEST(test_qr_solve_square<Scalar>(64, 1));
|
||||
CALL_SUBTEST(test_qr_solve_square<Scalar>(64, 4));
|
||||
CALL_SUBTEST(test_qr_solve_square<Scalar>(256, 8));
|
||||
|
||||
CALL_SUBTEST(test_qr_solve_overdetermined<Scalar>(128, 64, 4));
|
||||
CALL_SUBTEST(test_qr_solve_overdetermined<Scalar>(256, 128, 1));
|
||||
|
||||
CALL_SUBTEST(test_qr_solve_device<Scalar>(64, 4));
|
||||
CALL_SUBTEST(test_qr_solve_overdetermined_device<Scalar>(128, 64, 4));
|
||||
CALL_SUBTEST(test_qr_multiple_solves<Scalar>(64));
|
||||
CALL_SUBTEST(test_qr_vs_cpu<Scalar>(64, 4));
|
||||
CALL_SUBTEST(test_qr_vs_cpu<Scalar>(256, 8));
|
||||
}
|
||||
|
||||
void test_qr_empty() {
|
||||
GpuQR<double> qr(MatrixXd(0, 0));
|
||||
VERIFY_IS_EQUAL(qr.info(), Success);
|
||||
VERIFY_IS_EQUAL(qr.rows(), 0);
|
||||
VERIFY_IS_EQUAL(qr.cols(), 0);
|
||||
}
|
||||
|
||||
EIGEN_DECLARE_TEST(gpu_cusolver_qr) {
|
||||
CALL_SUBTEST(test_scalar<float>());
|
||||
CALL_SUBTEST(test_scalar<double>());
|
||||
CALL_SUBTEST(test_scalar<std::complex<float>>());
|
||||
CALL_SUBTEST(test_scalar<std::complex<double>>());
|
||||
CALL_SUBTEST(test_qr_empty());
|
||||
}
|
||||
194
test/gpu_cusolver_svd.cpp
Normal file
194
test/gpu_cusolver_svd.cpp
Normal file
@@ -0,0 +1,194 @@
|
||||
// This file is part of Eigen, a lightweight C++ template library
|
||||
// for linear algebra.
|
||||
//
|
||||
// Copyright (C) 2026 Rasmus Munk Larsen <rmlarsen@gmail.com>
|
||||
//
|
||||
// This Source Code Form is subject to the terms of the Mozilla
|
||||
// Public License v. 2.0. If a copy of the MPL was not distributed
|
||||
// with this file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
// Tests for GpuSVD: GPU SVD via cuSOLVER.
|
||||
|
||||
#define EIGEN_USE_GPU
|
||||
#include "main.h"
|
||||
#include <Eigen/SVD>
|
||||
#include <Eigen/GPU>
|
||||
|
||||
using namespace Eigen;
|
||||
|
||||
// ---- SVD reconstruction: U * diag(S) * VT ≈ A ------------------------------
|
||||
|
||||
template <typename Scalar, unsigned int Options>
|
||||
void test_svd_reconstruction(Index m, Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(m, n);
|
||||
GpuSVD<Scalar> svd(A, Options);
|
||||
VERIFY_IS_EQUAL(svd.info(), Success);
|
||||
|
||||
auto S = svd.singularValues();
|
||||
Mat U = svd.matrixU();
|
||||
Mat VT = svd.matrixVT();
|
||||
|
||||
const Index k = (std::min)(m, n);
|
||||
|
||||
// Reconstruct: A_hat = U[:,:k] * diag(S) * VT[:k,:].
|
||||
Mat A_hat = U.leftCols(k) * S.asDiagonal() * VT.topRows(k);
|
||||
RealScalar tol = RealScalar(5) * std::sqrt(static_cast<RealScalar>(k)) * NumTraits<Scalar>::epsilon() * A.norm();
|
||||
VERIFY((A_hat - A).norm() < tol);
|
||||
|
||||
// Orthogonality: U^H * U ≈ I.
|
||||
Mat UtU = U.adjoint() * U;
|
||||
Mat I_u = Mat::Identity(U.cols(), U.cols());
|
||||
VERIFY((UtU - I_u).norm() < tol);
|
||||
|
||||
// Orthogonality: VT * VT^H ≈ I.
|
||||
Mat VtVh = VT * VT.adjoint();
|
||||
Mat I_v = Mat::Identity(VT.rows(), VT.rows());
|
||||
VERIFY((VtVh - I_v).norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Singular values match CPU BDCSVD ---------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_svd_singular_values(Index m, Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(m, n);
|
||||
GpuSVD<Scalar> svd(A, 0); // values only
|
||||
VERIFY_IS_EQUAL(svd.info(), Success);
|
||||
|
||||
auto S_gpu = svd.singularValues();
|
||||
auto S_cpu = BDCSVD<Mat>(A, 0).singularValues();
|
||||
|
||||
RealScalar tol =
|
||||
RealScalar(5) * std::sqrt(static_cast<RealScalar>((std::min)(m, n))) * NumTraits<Scalar>::epsilon() * S_cpu(0);
|
||||
VERIFY((S_gpu - S_cpu).norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Solve: pseudoinverse ---------------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_svd_solve(Index m, Index n, Index nrhs) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(m, n);
|
||||
Mat B = Mat::Random(m, nrhs);
|
||||
|
||||
GpuSVD<Scalar> svd(A, ComputeThinU | ComputeThinV);
|
||||
VERIFY_IS_EQUAL(svd.info(), Success);
|
||||
|
||||
Mat X = svd.solve(B);
|
||||
VERIFY_IS_EQUAL(X.rows(), n);
|
||||
VERIFY_IS_EQUAL(X.cols(), nrhs);
|
||||
|
||||
// Compare with CPU BDCSVD solve.
|
||||
Mat X_cpu = BDCSVD<Mat>(A, ComputeThinU | ComputeThinV).solve(B);
|
||||
RealScalar tol = RealScalar(100) * RealScalar((std::max)(m, n)) * NumTraits<Scalar>::epsilon();
|
||||
VERIFY((X - X_cpu).norm() / X_cpu.norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Solve: truncated -------------------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_svd_solve_truncated(Index m, Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(m, n);
|
||||
Mat B = Mat::Random(m, 1);
|
||||
const Index k = (std::min)(m, n);
|
||||
const Index trunc = k / 2;
|
||||
eigen_assert(trunc > 0);
|
||||
|
||||
GpuSVD<Scalar> svd(A, ComputeThinU | ComputeThinV);
|
||||
Mat X_trunc = svd.solve(B, trunc);
|
||||
|
||||
// Build CPU reference: truncated pseudoinverse.
|
||||
auto cpu_svd = BDCSVD<Mat>(A, ComputeThinU | ComputeThinV);
|
||||
auto S = cpu_svd.singularValues();
|
||||
Mat U = cpu_svd.matrixU();
|
||||
Mat V = cpu_svd.matrixV();
|
||||
|
||||
// D_ii = 1/S_i for i < trunc, 0 otherwise.
|
||||
Matrix<RealScalar, Dynamic, 1> D = Matrix<RealScalar, Dynamic, 1>::Zero(k);
|
||||
for (Index i = 0; i < trunc; ++i) D(i) = RealScalar(1) / S(i);
|
||||
Mat X_ref = V * D.asDiagonal() * U.adjoint() * B;
|
||||
|
||||
RealScalar tol = RealScalar(100) * RealScalar(k) * NumTraits<Scalar>::epsilon();
|
||||
VERIFY((X_trunc - X_ref).norm() / X_ref.norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Solve: Tikhonov regularized --------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_svd_solve_regularized(Index m, Index n) {
|
||||
using Mat = Matrix<Scalar, Dynamic, Dynamic>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
Mat A = Mat::Random(m, n);
|
||||
Mat B = Mat::Random(m, 1);
|
||||
RealScalar lambda = RealScalar(0.1);
|
||||
const Index k = (std::min)(m, n);
|
||||
|
||||
GpuSVD<Scalar> svd(A, ComputeThinU | ComputeThinV);
|
||||
Mat X_reg = svd.solve(B, lambda);
|
||||
|
||||
// CPU reference: D_ii = S_i / (S_i^2 + lambda^2).
|
||||
auto cpu_svd = BDCSVD<Mat>(A, ComputeThinU | ComputeThinV);
|
||||
auto S = cpu_svd.singularValues();
|
||||
Mat U = cpu_svd.matrixU();
|
||||
Mat V = cpu_svd.matrixV();
|
||||
|
||||
Matrix<RealScalar, Dynamic, 1> D(k);
|
||||
for (Index i = 0; i < k; ++i) D(i) = S(i) / (S(i) * S(i) + lambda * lambda);
|
||||
Mat X_ref = V * D.asDiagonal() * U.adjoint() * B;
|
||||
|
||||
RealScalar tol = RealScalar(100) * RealScalar(k) * NumTraits<Scalar>::epsilon();
|
||||
VERIFY((X_reg - X_ref).norm() / X_ref.norm() < tol);
|
||||
}
|
||||
|
||||
// ---- Empty matrix -----------------------------------------------------------
|
||||
|
||||
void test_svd_empty() {
|
||||
GpuSVD<double> svd(MatrixXd(0, 0), 0);
|
||||
VERIFY_IS_EQUAL(svd.info(), Success);
|
||||
VERIFY_IS_EQUAL(svd.rows(), 0);
|
||||
VERIFY_IS_EQUAL(svd.cols(), 0);
|
||||
}
|
||||
|
||||
// ---- Per-scalar driver ------------------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
void test_scalar() {
|
||||
// Reconstruction + orthogonality (thin and full, identical test logic).
|
||||
CALL_SUBTEST((test_svd_reconstruction<Scalar, ComputeThinU | ComputeThinV>(64, 64)));
|
||||
CALL_SUBTEST((test_svd_reconstruction<Scalar, ComputeThinU | ComputeThinV>(128, 64)));
|
||||
CALL_SUBTEST((test_svd_reconstruction<Scalar, ComputeThinU | ComputeThinV>(64, 128))); // wide (m < n)
|
||||
CALL_SUBTEST((test_svd_reconstruction<Scalar, ComputeFullU | ComputeFullV>(64, 64)));
|
||||
CALL_SUBTEST((test_svd_reconstruction<Scalar, ComputeFullU | ComputeFullV>(128, 64)));
|
||||
|
||||
// Singular values.
|
||||
CALL_SUBTEST(test_svd_singular_values<Scalar>(64, 64));
|
||||
CALL_SUBTEST(test_svd_singular_values<Scalar>(128, 64));
|
||||
|
||||
// Solve.
|
||||
CALL_SUBTEST(test_svd_solve<Scalar>(64, 64, 4));
|
||||
CALL_SUBTEST(test_svd_solve<Scalar>(128, 64, 4));
|
||||
CALL_SUBTEST(test_svd_solve<Scalar>(64, 128, 4)); // wide (m < n)
|
||||
|
||||
// Truncated and regularized solve.
|
||||
CALL_SUBTEST(test_svd_solve_truncated<Scalar>(64, 64));
|
||||
CALL_SUBTEST(test_svd_solve_regularized<Scalar>(64, 64));
|
||||
}
|
||||
|
||||
EIGEN_DECLARE_TEST(gpu_cusolver_svd) {
|
||||
CALL_SUBTEST(test_scalar<float>());
|
||||
CALL_SUBTEST(test_scalar<double>());
|
||||
CALL_SUBTEST(test_scalar<std::complex<float>>());
|
||||
CALL_SUBTEST(test_scalar<std::complex<double>>());
|
||||
CALL_SUBTEST(test_svd_empty());
|
||||
}
|
||||
@@ -35,7 +35,6 @@ void test_allocate(Index rows, Index cols) {
|
||||
VERIFY(!dm.empty());
|
||||
VERIFY_IS_EQUAL(dm.rows(), rows);
|
||||
VERIFY_IS_EQUAL(dm.cols(), cols);
|
||||
VERIFY_IS_EQUAL(dm.outerStride(), rows);
|
||||
VERIFY(dm.data() != nullptr);
|
||||
VERIFY_IS_EQUAL(dm.sizeInBytes(), size_t(rows) * size_t(cols) * sizeof(Scalar));
|
||||
}
|
||||
@@ -69,7 +68,7 @@ void test_roundtrip_async(Index rows, Index cols) {
|
||||
EIGEN_CUDA_RUNTIME_CHECK(cudaStreamCreate(&stream));
|
||||
|
||||
// Async upload from raw pointer.
|
||||
auto dm = DeviceMatrix<Scalar>::fromHostAsync(host.data(), rows, cols, rows, stream);
|
||||
auto dm = DeviceMatrix<Scalar>::fromHostAsync(host.data(), rows, cols, stream);
|
||||
VERIFY_IS_EQUAL(dm.rows(), rows);
|
||||
VERIFY_IS_EQUAL(dm.cols(), cols);
|
||||
|
||||
@@ -185,7 +184,6 @@ void test_resize() {
|
||||
dm.resize(50, 30);
|
||||
VERIFY_IS_EQUAL(dm.rows(), 50);
|
||||
VERIFY_IS_EQUAL(dm.cols(), 30);
|
||||
VERIFY_IS_EQUAL(dm.outerStride(), 50);
|
||||
VERIFY(dm.data() != nullptr);
|
||||
|
||||
// Resize to same dimensions is a no-op.
|
||||
|
||||
Reference in New Issue
Block a user