mirror of
https://gitlab.com/libeigen/eigen.git
synced 2026-04-10 11:34:33 +08:00
GPU: Add BLAS-1 ops, DeviceScalar, device-resident SpMV, and CG interop (5/5)
Add the operator interface needed for GPU iterative solvers: - BLAS Level-1 on DeviceMatrix: dot(), norm(), squaredNorm(), setZero(), noalias(), operator+=/-=/\*= dispatching to cuBLAS axpy/scal/dot/nrm2. - DeviceScalar<Scalar>: device-resident scalar returned by reductions. Defers host sync until value is read (implicit conversion). Device-side division via NPP for real types. - GpuContext: stream-borrowing constructor, setThreadLocal(), cublasLtHandle(), cusparseHandle(). - GEMM upgraded from cublasGemmEx to cublasLtMatmul with heuristic algorithm selection and plan caching. - GpuSparseContext: GpuContext& constructor for same-stream execution, deviceView() returning DeviceSparseView with operator* for device-resident SpMV (d_y = d_A * d_x). - geam expressions: d_C = d_A + alpha * d_B via cublasXgeam. - GpuSVD::matrixV() convenience wrapper. These additions make DeviceMatrix usable as a VectorType in Eigen algorithm templates. Conjugate gradient is the motivating example and is tested against CPU ConjugateGradient for correctness. Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -11,7 +11,7 @@
|
||||
# ncu --set full -o profile ./build-bench-gpu/bench_gpu_solvers --benchmark_filter=BM_GpuLLT_Compute/4096
|
||||
|
||||
cmake_minimum_required(VERSION 3.18)
|
||||
project(EigenGpuBenchmarks CXX)
|
||||
project(EigenGpuBenchmarks CXX CUDA)
|
||||
|
||||
find_package(benchmark REQUIRED)
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
@@ -55,3 +55,37 @@ eigen_add_gpu_benchmark(bench_gpu_batching_float bench_gpu_batching.cpp DEFINITI
|
||||
# FFT benchmarks: 1D/2D C2C, R2C, C2R throughput and plan reuse.
|
||||
eigen_add_gpu_benchmark(bench_gpu_fft bench_gpu_fft.cpp LIBRARIES CUDA::cufft)
|
||||
eigen_add_gpu_benchmark(bench_gpu_fft_double bench_gpu_fft.cpp LIBRARIES CUDA::cufft DEFINITIONS SCALAR=double)
|
||||
|
||||
# CG sync overhead benchmark: host vs device pointer mode for reductions.
|
||||
# Uses CUDA kernels for device scalar arithmetic.
|
||||
add_executable(bench_gpu_cg_sync bench_gpu_cg_sync.cu)
|
||||
target_include_directories(bench_gpu_cg_sync PRIVATE
|
||||
${EIGEN_SOURCE_DIR}
|
||||
${CUDAToolkit_INCLUDE_DIRS})
|
||||
target_link_libraries(bench_gpu_cg_sync PRIVATE
|
||||
benchmark::benchmark benchmark::benchmark_main
|
||||
CUDA::cudart CUDA::cusolver CUDA::cublas CUDA::cusparse CUDA::npps CUDA::nppc)
|
||||
target_compile_options(bench_gpu_cg_sync PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:-O3 --expt-relaxed-constexpr>)
|
||||
target_compile_definitions(bench_gpu_cg_sync PRIVATE EIGEN_USE_GPU)
|
||||
|
||||
# GPU CG vs CPU CG comparison benchmark.
|
||||
add_executable(bench_gpu_cg_vs_cpu bench_gpu_cg_vs_cpu.cu)
|
||||
target_include_directories(bench_gpu_cg_vs_cpu PRIVATE
|
||||
${EIGEN_SOURCE_DIR}
|
||||
${CUDAToolkit_INCLUDE_DIRS})
|
||||
target_link_libraries(bench_gpu_cg_vs_cpu PRIVATE
|
||||
benchmark::benchmark benchmark::benchmark_main
|
||||
CUDA::cudart CUDA::cusolver CUDA::cublas CUDA::cusparse CUDA::npps CUDA::nppc)
|
||||
target_compile_options(bench_gpu_cg_vs_cpu PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:-O3 --expt-relaxed-constexpr>)
|
||||
target_compile_definitions(bench_gpu_cg_vs_cpu PRIVATE EIGEN_USE_GPU)
|
||||
|
||||
# Bundle Adjustment benchmark: GPU CG vs CPU CG on real BAL datasets.
|
||||
add_executable(bench_gpu_ba bench_gpu_ba.cu)
|
||||
target_include_directories(bench_gpu_ba PRIVATE
|
||||
${EIGEN_SOURCE_DIR}
|
||||
${CUDAToolkit_INCLUDE_DIRS})
|
||||
target_link_libraries(bench_gpu_ba PRIVATE
|
||||
benchmark::benchmark
|
||||
CUDA::cudart CUDA::cusolver CUDA::cublas CUDA::cusparse CUDA::npps CUDA::nppc)
|
||||
target_compile_options(bench_gpu_ba PRIVATE $<$<COMPILE_LANGUAGE:CUDA>:-O3 --expt-relaxed-constexpr>)
|
||||
target_compile_definitions(bench_gpu_ba PRIVATE EIGEN_USE_GPU)
|
||||
|
||||
149
benchmarks/GPU/ba_results.md
Normal file
149
benchmarks/GPU/ba_results.md
Normal file
@@ -0,0 +1,149 @@
|
||||
# Bundle Adjustment: GPU CG vs CPU CG Results
|
||||
|
||||
Benchmark of Eigen's GPU CG pipeline on normal equations arising from bundle
|
||||
adjustment (BAL datasets). Compares CPU `ConjugateGradient` (Jacobi preconditioner)
|
||||
against GPU CG using `DeviceMatrix` + `GpuSparseContext` + `DeviceScalar`.
|
||||
|
||||
## Hardware
|
||||
|
||||
- **CPU**: Intel Core i7-13700HX (Raptor Lake, 12 cores / 24 threads, single thread for Eigen CG)
|
||||
- **GPU**: NVIDIA GeForce RTX 4070 Laptop GPU (Ada Lovelace, 4608 CUDA cores, 8 GB GDDR6)
|
||||
- **CUDA**: 13.2 / Driver 595.79
|
||||
- **OS**: Ubuntu 24.04 (WSL2, kernel 6.6.87)
|
||||
|
||||
## Software
|
||||
|
||||
- Eigen: `eigen-gpu-cg` branch
|
||||
- Google Benchmark 1.9.1
|
||||
- Compiler: nvcc 13.2 + g++ 13.3
|
||||
- Normal equations: H = J^T*J + I (Levenberg-Marquardt damping lambda=1.0)
|
||||
- CG tolerance: 1e-8, max iterations: 10000
|
||||
|
||||
## Method
|
||||
|
||||
For each BAL problem file:
|
||||
1. Parse the BAL file (cameras, 3D points, 2D observations)
|
||||
2. Compute the full Jacobian J using the BAL camera model (Rodrigues rotation +
|
||||
perspective projection + radial distortion) with central finite differences
|
||||
3. Form the normal equations H = J^T*J + lambda*I (sparse, symmetric positive definite)
|
||||
4. Solve H*dx = -J^T*r using CG with Jacobi preconditioner on CPU and GPU
|
||||
5. Report wall-clock time (mean of 3 repetitions)
|
||||
|
||||
GPU CG uses: `GpuSparseContext` for SpMV, `DeviceMatrix` for vectors,
|
||||
`DeviceScalar` with `CUBLAS_POINTER_MODE_DEVICE` for dot/norm reductions,
|
||||
in-place `cwiseProduct` via NPP for Jacobi preconditioner application,
|
||||
device-pointer-mode `scal` to avoid host sync on the beta update.
|
||||
|
||||
## Results
|
||||
|
||||
### Summary table
|
||||
|
||||
| Dataset | Cameras | Points | Obs | H size | H nnz | CG iters | CPU CG (ms) | GPU CG (ms) | Speedup |
|
||||
|---------|---------|--------|-----|--------|-------|----------|-------------|-------------|---------|
|
||||
| Ladybug-49 | 49 | 7,776 | 31,843 | 23,769 | 1.8M | 4,421 | 4,006 | 1,152 | **3.5x** |
|
||||
| Ladybug-138 | 138 | 19,878 | 85,217 | 60,876 | 4.8M | 7,008 | 21,498 | 3,553 | **6.1x** |
|
||||
| Ladybug-646 | 646 | 73,584 | 327,297 | 226,566 | 18.4M | 10,000* | 123,727 | 14,268 | **8.7x** |
|
||||
| Dubrovnik-356 | 356 | 226,730 | 1,255,268 | 683,394 | 69.8M | 4,308 | 216,149 | 24,493 | **8.8x** |
|
||||
|
||||
\* Hit 10,000 iteration cap (poorly conditioned problem). Both CPU and GPU
|
||||
hit the same cap, so timing comparison remains valid.
|
||||
|
||||
### Profile breakdown (Ladybug-138, nsys)
|
||||
|
||||
GPU kernel time is dominated by SpMV (91%). The remaining 9% is BLAS-1
|
||||
operations (dot, axpy, scal) and NPP element-wise ops (cwiseProduct).
|
||||
|
||||
| Kernel | Time (ms) | % | Calls |
|
||||
|--------|-----------|---|-------|
|
||||
| cuSPARSE csrmv (SpMV) | 2507 | 91.3% | 7,006 |
|
||||
| cuBLAS dot | 92 | 3.4% | 21,020 |
|
||||
| cuBLAS axpy (device ptr) | 27 | 1.0% | 14,012 |
|
||||
| cuSPARSE partition | 19 | 0.7% | 7,006 |
|
||||
| NPP cwiseProduct | 16 + 13 | 1.1% | 14,011 + 7,006 |
|
||||
| cuBLAS axpy (host ptr) | 12 | 0.5% | 7,005 |
|
||||
| cuBLAS scal (device ptr) | 11 | 0.4% | 7,005 |
|
||||
| NPP scalar ops | 7 | 0.2% | 7,006 |
|
||||
|
||||
### Optimizations applied
|
||||
|
||||
Three profiling-driven optimizations reduced GPU CG time by **1.8x**
|
||||
(6.5s → 3.6s on Ladybug-138):
|
||||
|
||||
1. **In-place `cwiseProduct`**: The Jacobi preconditioner apply
|
||||
(`z = invdiag .* residual`) was allocating a new DeviceMatrix every
|
||||
iteration. Added `z.cwiseProduct(ctx, a, b)` that reuses `z`'s buffer.
|
||||
Reduced `cudaMalloc` calls from 7,053 to 23 (saving 2.3s).
|
||||
|
||||
2. **`squaredNorm` via `dot(x,x)`**: cuBLAS `nrm2` uses a numerically
|
||||
careful scaled-sum-of-squares algorithm (29µs/call). Replaced with
|
||||
`dot(x,x)` (6.4µs/call) — 4.5x faster per call, saving ~320ms.
|
||||
|
||||
3. **Device-pointer `scal`**: `p *= beta` was converting `DeviceScalar`
|
||||
beta to host (triggering a stream sync), then calling host-pointer-mode
|
||||
scal. Added `operator*=(DeviceScalar)` that uses device-pointer-mode
|
||||
scal, eliminating one sync per iteration. Halved `cudaStreamSynchronize`
|
||||
calls from 14K to 7K.
|
||||
|
||||
### Observations
|
||||
|
||||
1. **GPU speedup scales with problem size**: from 3.5x on small problems
|
||||
(24K variables) to 8.8x on large problems (683K variables). This is
|
||||
expected — larger problems have more parallelism for the GPU to exploit.
|
||||
|
||||
2. **Iteration counts match**: CPU and GPU CG converge in the same number
|
||||
of iterations (within 1%), confirming numerical equivalence.
|
||||
|
||||
3. **Bottleneck is SpMV**: CG iteration time is dominated (91%) by the
|
||||
sparse matrix-vector product on H. Further speedup requires either
|
||||
faster SpMV (e.g., block-sparse formats) or algorithmic improvements
|
||||
(Schur complement, better preconditioners).
|
||||
|
||||
4. **Remaining overhead**: CUDA API calls (cudaMemcpyAsync for 8-byte
|
||||
DeviceScalar transfers) account for ~50% of non-kernel time. Batching
|
||||
multiple scalar reductions into a single transfer would help.
|
||||
|
||||
5. **Jacobi preconditioner is weak for BA**: The Ladybug-646 problem does
|
||||
not converge in 10K iterations. Ceres uses block Jacobi or Schur
|
||||
complement preconditioners that would also benefit from GPU acceleration.
|
||||
|
||||
### Scaling plot data
|
||||
|
||||
```
|
||||
# n nnz_H cpu_ms gpu_ms speedup
|
||||
23769 1793475 4006 1152 3.48
|
||||
60876 4791762 21498 3553 6.05
|
||||
226566 18387948 123727 14268 8.67
|
||||
683394 69827066 216149 24493 8.82
|
||||
```
|
||||
|
||||
## BAL datasets
|
||||
|
||||
Downloaded from http://grail.cs.washington.edu/projects/bal/
|
||||
|
||||
| File | Source |
|
||||
|------|--------|
|
||||
| problem-49-7776-pre.txt | Ladybug sequence |
|
||||
| problem-138-19878-pre.txt | Ladybug sequence |
|
||||
| problem-646-73584-pre.txt | Ladybug sequence |
|
||||
| problem-356-226730-pre.txt | Dubrovnik reconstruction |
|
||||
|
||||
## Reproducing
|
||||
|
||||
```bash
|
||||
# Build
|
||||
cmake -G Ninja -B build-bench-gpu -S benchmarks/GPU -DCMAKE_CUDA_ARCHITECTURES=89
|
||||
cmake --build build-bench-gpu --target bench_gpu_ba
|
||||
|
||||
# Download BAL datasets
|
||||
wget http://grail.cs.washington.edu/projects/bal/data/ladybug/problem-49-7776-pre.txt.bz2
|
||||
wget http://grail.cs.washington.edu/projects/bal/data/ladybug/problem-138-19878-pre.txt.bz2
|
||||
wget http://grail.cs.washington.edu/projects/bal/data/ladybug/problem-646-73584-pre.txt.bz2
|
||||
wget http://grail.cs.washington.edu/projects/bal/data/dubrovnik/problem-356-226730-pre.txt.bz2
|
||||
bunzip2 *.bz2
|
||||
|
||||
# Run (one at a time)
|
||||
BAL_FILE=problem-49-7776-pre.txt ./build-bench-gpu/bench_gpu_ba --benchmark_repetitions=3
|
||||
BAL_FILE=problem-138-19878-pre.txt ./build-bench-gpu/bench_gpu_ba --benchmark_repetitions=3
|
||||
BAL_FILE=problem-646-73584-pre.txt ./build-bench-gpu/bench_gpu_ba --benchmark_repetitions=3
|
||||
BAL_FILE=problem-356-226730-pre.txt ./build-bench-gpu/bench_gpu_ba --benchmark_repetitions=3
|
||||
```
|
||||
533
benchmarks/GPU/bench_gpu_ba.cu
Normal file
533
benchmarks/GPU/bench_gpu_ba.cu
Normal file
@@ -0,0 +1,533 @@
|
||||
// Bundle Adjustment benchmark: GPU CG vs CPU CG on real BAL datasets.
|
||||
//
|
||||
// Tests Eigen's GPU CG pipeline (DeviceMatrix + GpuSparseContext + DeviceScalar)
|
||||
// on the normal equations (J^T*J) arising from bundle adjustment problems.
|
||||
//
|
||||
// Reads a BAL (Bundle Adjustment in the Large) format file, computes the
|
||||
// Jacobian and residual, forms the normal equations H = J^T*J + lambda*I,
|
||||
// then solves H*dx = -J^T*r with both CPU and GPU conjugate gradients.
|
||||
//
|
||||
// BAL format: http://grail.cs.washington.edu/projects/bal/
|
||||
//
|
||||
// Usage:
|
||||
// cmake --build build-bench-gpu --target bench_gpu_ba
|
||||
//
|
||||
// # Download a BAL dataset (bz2-compressed):
|
||||
// wget http://grail.cs.washington.edu/projects/bal/data/ladybug/problem-49-7776-pre.txt.bz2
|
||||
// bunzip2 problem-49-7776-pre.txt.bz2
|
||||
//
|
||||
// # Run on a specific problem:
|
||||
// BAL_FILE=problem-49-7776-pre.txt ./build-bench-gpu/bench_gpu_ba
|
||||
//
|
||||
// # Append results to the log:
|
||||
// BAL_FILE=problem-49-7776-pre.txt ./build-bench-gpu/bench_gpu_ba \
|
||||
// --benchmark_format=console 2>&1 | tee -a benchmarks/GPU/ba_results.log
|
||||
|
||||
#include <benchmark/benchmark.h>
|
||||
|
||||
#include <Eigen/Sparse>
|
||||
#include <Eigen/IterativeLinearSolvers>
|
||||
#include <Eigen/GPU>
|
||||
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <fstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
using namespace Eigen;
|
||||
|
||||
// ============================================================================
|
||||
// BAL problem data
|
||||
// ============================================================================
|
||||
|
||||
struct BALProblem {
|
||||
int num_cameras = 0;
|
||||
int num_points = 0;
|
||||
int num_observations = 0;
|
||||
|
||||
// Observations: (camera_idx, point_idx, observed_x, observed_y).
|
||||
std::vector<int> camera_index;
|
||||
std::vector<int> point_index;
|
||||
std::vector<double> observations_x;
|
||||
std::vector<double> observations_y;
|
||||
|
||||
// Camera parameters: 9 per camera (Rodrigues r[3], translation t[3], f, k1, k2).
|
||||
std::vector<double> cameras; // [num_cameras * 9]
|
||||
|
||||
// 3D points: 3 per point.
|
||||
std::vector<double> points; // [num_points * 3]
|
||||
|
||||
const double* camera(int i) const { return &cameras[i * 9]; }
|
||||
const double* point(int i) const { return &points[i * 3]; }
|
||||
|
||||
bool load(const std::string& filename) {
|
||||
std::ifstream in(filename);
|
||||
if (!in) {
|
||||
fprintf(stderr, "ERROR: Cannot open BAL file: %s\n", filename.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
in >> num_cameras >> num_points >> num_observations;
|
||||
if (!in || num_cameras <= 0 || num_points <= 0 || num_observations <= 0) {
|
||||
fprintf(stderr, "ERROR: Invalid BAL header in %s\n", filename.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
camera_index.resize(num_observations);
|
||||
point_index.resize(num_observations);
|
||||
observations_x.resize(num_observations);
|
||||
observations_y.resize(num_observations);
|
||||
for (int i = 0; i < num_observations; ++i) {
|
||||
in >> camera_index[i] >> point_index[i] >> observations_x[i] >> observations_y[i];
|
||||
}
|
||||
|
||||
cameras.resize(num_cameras * 9);
|
||||
for (int i = 0; i < num_cameras * 9; ++i) {
|
||||
in >> cameras[i];
|
||||
}
|
||||
|
||||
points.resize(num_points * 3);
|
||||
for (int i = 0; i < num_points * 3; ++i) {
|
||||
in >> points[i];
|
||||
}
|
||||
|
||||
if (!in) {
|
||||
fprintf(stderr, "ERROR: Truncated BAL file: %s\n", filename.c_str());
|
||||
return false;
|
||||
}
|
||||
|
||||
fprintf(stderr, "Loaded BAL: %d cameras, %d points, %d observations\n", num_cameras, num_points, num_observations);
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
// ============================================================================
|
||||
// Camera projection model (BAL convention)
|
||||
// ============================================================================
|
||||
|
||||
// Rodrigues rotation: rotate point X by axis-angle vector omega.
|
||||
static void rodrigues_rotate(const double* omega, const double* X, double* result) {
|
||||
double theta2 = omega[0] * omega[0] + omega[1] * omega[1] + omega[2] * omega[2];
|
||||
if (theta2 > 1e-30) {
|
||||
double theta = std::sqrt(theta2);
|
||||
double costh = std::cos(theta);
|
||||
double sinth = std::sin(theta);
|
||||
double k = (1.0 - costh) / theta2;
|
||||
|
||||
// Cross product omega x X.
|
||||
double wx = omega[1] * X[2] - omega[2] * X[1];
|
||||
double wy = omega[2] * X[0] - omega[0] * X[2];
|
||||
double wz = omega[0] * X[1] - omega[1] * X[0];
|
||||
|
||||
// Dot product omega . X.
|
||||
double dot = omega[0] * X[0] + omega[1] * X[1] + omega[2] * X[2];
|
||||
|
||||
result[0] = X[0] * costh + wx * (sinth / theta) + omega[0] * dot * k;
|
||||
result[1] = X[1] * costh + wy * (sinth / theta) + omega[1] * dot * k;
|
||||
result[2] = X[2] * costh + wz * (sinth / theta) + omega[2] * dot * k;
|
||||
} else {
|
||||
// Small angle: R ≈ I + [omega]×.
|
||||
result[0] = X[0] + omega[1] * X[2] - omega[2] * X[1];
|
||||
result[1] = X[1] + omega[2] * X[0] - omega[0] * X[2];
|
||||
result[2] = X[2] + omega[0] * X[1] - omega[1] * X[0];
|
||||
}
|
||||
}
|
||||
|
||||
// Project a 3D point through a camera, returning the 2D residual.
|
||||
// camera: [r0,r1,r2, t0,t1,t2, f, k1, k2]
|
||||
// point: [X, Y, Z]
|
||||
// observed: [ox, oy]
|
||||
// residual: [rx, ry] = projected - observed
|
||||
static void project(const double* camera, const double* point, const double* observed, double* residual) {
|
||||
// Rotate.
|
||||
double P[3];
|
||||
rodrigues_rotate(camera, point, P);
|
||||
|
||||
// Translate.
|
||||
P[0] += camera[3];
|
||||
P[1] += camera[4];
|
||||
P[2] += camera[5];
|
||||
|
||||
// Normalize (BAL convention: negative z).
|
||||
double xp = -P[0] / P[2];
|
||||
double yp = -P[1] / P[2];
|
||||
|
||||
// Radial distortion.
|
||||
double r2 = xp * xp + yp * yp;
|
||||
double distortion = 1.0 + camera[7] * r2 + camera[8] * r2 * r2;
|
||||
|
||||
// Apply focal length.
|
||||
double predicted_x = camera[6] * distortion * xp;
|
||||
double predicted_y = camera[6] * distortion * yp;
|
||||
|
||||
residual[0] = predicted_x - observed[0];
|
||||
residual[1] = predicted_y - observed[1];
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Jacobian computation (numerical differentiation)
|
||||
// ============================================================================
|
||||
|
||||
// Compute the 2x9 Jacobian block w.r.t. camera params and 2x3 block w.r.t.
|
||||
// point coords for a single observation, using central finite differences.
|
||||
static void compute_jacobian_block(const double* camera, const double* point, const double* observed,
|
||||
double* J_cam, // 2x9, row-major
|
||||
double* J_point) // 2x3, row-major
|
||||
{
|
||||
constexpr double eps = 1e-8;
|
||||
|
||||
// Camera parameters (9).
|
||||
double cam_pert[9];
|
||||
std::copy(camera, camera + 9, cam_pert);
|
||||
for (int j = 0; j < 9; ++j) {
|
||||
double orig = cam_pert[j];
|
||||
double rp[2], rm[2];
|
||||
|
||||
cam_pert[j] = orig + eps;
|
||||
project(cam_pert, point, observed, rp);
|
||||
cam_pert[j] = orig - eps;
|
||||
project(cam_pert, point, observed, rm);
|
||||
cam_pert[j] = orig;
|
||||
|
||||
J_cam[0 * 9 + j] = (rp[0] - rm[0]) / (2.0 * eps);
|
||||
J_cam[1 * 9 + j] = (rp[1] - rm[1]) / (2.0 * eps);
|
||||
}
|
||||
|
||||
// Point coordinates (3).
|
||||
double pt_pert[3];
|
||||
std::copy(point, point + 3, pt_pert);
|
||||
for (int j = 0; j < 3; ++j) {
|
||||
double orig = pt_pert[j];
|
||||
double rp[2], rm[2];
|
||||
|
||||
pt_pert[j] = orig + eps;
|
||||
project(camera, pt_pert, observed, rp);
|
||||
pt_pert[j] = orig - eps;
|
||||
project(camera, pt_pert, observed, rm);
|
||||
pt_pert[j] = orig;
|
||||
|
||||
J_point[0 * 3 + j] = (rp[0] - rm[0]) / (2.0 * eps);
|
||||
J_point[1 * 3 + j] = (rp[1] - rm[1]) / (2.0 * eps);
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Build normal equations: H = J^T*J + lambda*I, g = -J^T*r
|
||||
// ============================================================================
|
||||
|
||||
struct NormalEquations {
|
||||
SparseMatrix<double, ColMajor, int> H;
|
||||
VectorXd g;
|
||||
VectorXd residual;
|
||||
double residual_norm;
|
||||
int jacobian_rows;
|
||||
int jacobian_cols;
|
||||
long jacobian_nnz;
|
||||
};
|
||||
|
||||
static NormalEquations build_normal_equations(const BALProblem& problem, double lambda = 1.0) {
|
||||
const int num_cam_params = problem.num_cameras * 9;
|
||||
const int num_pt_params = problem.num_points * 3;
|
||||
const int num_params = num_cam_params + num_pt_params;
|
||||
const int num_residuals = problem.num_observations * 2;
|
||||
|
||||
fprintf(stderr, "Building Jacobian: %d x %d, %ld nonzeros\n", num_residuals, num_params,
|
||||
(long)problem.num_observations * 24);
|
||||
|
||||
// Build J as a triplet list.
|
||||
using Triplet = Eigen::Triplet<double>;
|
||||
std::vector<Triplet> triplets;
|
||||
triplets.reserve(problem.num_observations * 24); // 2 rows × 12 nonzeros = 24 entries per obs
|
||||
|
||||
VectorXd residual(num_residuals);
|
||||
|
||||
for (int obs = 0; obs < problem.num_observations; ++obs) {
|
||||
int ci = problem.camera_index[obs];
|
||||
int pi = problem.point_index[obs];
|
||||
double observed[2] = {problem.observations_x[obs], problem.observations_y[obs]};
|
||||
|
||||
// Compute residual.
|
||||
double r[2];
|
||||
project(problem.camera(ci), problem.point(pi), observed, r);
|
||||
residual[obs * 2 + 0] = r[0];
|
||||
residual[obs * 2 + 1] = r[1];
|
||||
|
||||
// Compute Jacobian blocks.
|
||||
double J_cam[18], J_pt[6]; // 2x9 and 2x3
|
||||
compute_jacobian_block(problem.camera(ci), problem.point(pi), observed, J_cam, J_pt);
|
||||
|
||||
// Insert camera block: rows [2*obs, 2*obs+1], cols [9*ci, 9*ci+8].
|
||||
for (int row = 0; row < 2; ++row) {
|
||||
for (int col = 0; col < 9; ++col) {
|
||||
double val = J_cam[row * 9 + col];
|
||||
if (val != 0.0) {
|
||||
triplets.emplace_back(obs * 2 + row, ci * 9 + col, val);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Insert point block: rows [2*obs, 2*obs+1], cols [num_cam_params + 3*pi, ...].
|
||||
for (int row = 0; row < 2; ++row) {
|
||||
for (int col = 0; col < 3; ++col) {
|
||||
double val = J_pt[row * 3 + col];
|
||||
if (val != 0.0) {
|
||||
triplets.emplace_back(obs * 2 + row, num_cam_params + pi * 3 + col, val);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Build sparse Jacobian.
|
||||
SparseMatrix<double, ColMajor, int> J(num_residuals, num_params);
|
||||
J.setFromTriplets(triplets.begin(), triplets.end());
|
||||
|
||||
fprintf(stderr, "Jacobian: %dx%d, nnz=%ld\n", (int)J.rows(), (int)J.cols(), (long)J.nonZeros());
|
||||
|
||||
// Form normal equations: H = J^T*J + lambda*I.
|
||||
SparseMatrix<double, ColMajor, int> H = (J.transpose() * J).pruned();
|
||||
|
||||
// Add Levenberg-Marquardt damping.
|
||||
for (int i = 0; i < num_params; ++i) {
|
||||
H.coeffRef(i, i) += lambda;
|
||||
}
|
||||
H.makeCompressed();
|
||||
|
||||
// Gradient: g = -J^T * r.
|
||||
VectorXd g = -(J.transpose() * residual);
|
||||
|
||||
double rnorm = residual.norm();
|
||||
fprintf(stderr, "Normal equations: H is %dx%d, nnz=%ld, |r|=%.6e\n", (int)H.rows(), (int)H.cols(), (long)H.nonZeros(),
|
||||
rnorm);
|
||||
|
||||
return {std::move(H), std::move(g), std::move(residual), rnorm, num_residuals, num_params, (long)J.nonZeros()};
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Global problem state (loaded once before benchmarks run)
|
||||
// ============================================================================
|
||||
|
||||
static BALProblem g_problem;
|
||||
static NormalEquations g_neq;
|
||||
static bool g_loaded = false;
|
||||
|
||||
static void ensure_loaded() {
|
||||
if (g_loaded) return;
|
||||
|
||||
const char* bal_file = std::getenv("BAL_FILE");
|
||||
if (!bal_file) {
|
||||
fprintf(stderr,
|
||||
"ERROR: Set BAL_FILE environment variable to a BAL problem file.\n"
|
||||
" Download from: http://grail.cs.washington.edu/projects/bal/\n"
|
||||
" Example:\n"
|
||||
" wget http://grail.cs.washington.edu/projects/bal/data/ladybug/"
|
||||
"problem-49-7776-pre.txt.bz2\n"
|
||||
" bunzip2 problem-49-7776-pre.txt.bz2\n"
|
||||
" BAL_FILE=problem-49-7776-pre.txt ./build-bench-gpu/bench_gpu_ba\n");
|
||||
std::exit(1);
|
||||
}
|
||||
|
||||
if (!g_problem.load(bal_file)) {
|
||||
std::exit(1);
|
||||
}
|
||||
|
||||
g_neq = build_normal_equations(g_problem);
|
||||
g_loaded = true;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// CPU CG benchmark
|
||||
// ============================================================================
|
||||
|
||||
static void BM_BA_CPU_CG(benchmark::State& state) {
|
||||
ensure_loaded();
|
||||
const auto& H = g_neq.H;
|
||||
const auto& g = g_neq.g;
|
||||
|
||||
ConjugateGradient<SparseMatrix<double, ColMajor, int>, Lower | Upper> cg;
|
||||
cg.setMaxIterations(10000);
|
||||
cg.setTolerance(1e-8);
|
||||
cg.compute(H);
|
||||
|
||||
int last_iters = 0;
|
||||
double last_error = 0;
|
||||
for (auto _ : state) {
|
||||
VectorXd dx = cg.solve(g);
|
||||
benchmark::DoNotOptimize(dx.data());
|
||||
last_iters = cg.iterations();
|
||||
last_error = cg.error();
|
||||
}
|
||||
|
||||
state.counters["n"] = H.rows();
|
||||
state.counters["nnz"] = H.nonZeros();
|
||||
state.counters["iters"] = last_iters;
|
||||
state.counters["error"] = last_error;
|
||||
state.counters["cameras"] = g_problem.num_cameras;
|
||||
state.counters["points"] = g_problem.num_points;
|
||||
state.counters["observations"] = g_problem.num_observations;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// GPU CG benchmark (with Jacobi preconditioner)
|
||||
// ============================================================================
|
||||
|
||||
static void cuda_warmup() {
|
||||
static bool done = false;
|
||||
if (!done) {
|
||||
void* p;
|
||||
cudaMalloc(&p, 1);
|
||||
cudaFree(p);
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
static void BM_BA_GPU_CG(benchmark::State& state) {
|
||||
ensure_loaded();
|
||||
cuda_warmup();
|
||||
|
||||
const auto& H = g_neq.H;
|
||||
const auto& g = g_neq.g;
|
||||
const Index n = H.rows();
|
||||
|
||||
// Extract inverse diagonal (Jacobi preconditioner).
|
||||
using SpMat = SparseMatrix<double, ColMajor, int>;
|
||||
VectorXd invdiag(n);
|
||||
for (Index j = 0; j < H.outerSize(); ++j) {
|
||||
SpMat::InnerIterator it(H, j);
|
||||
while (it && it.index() != j) ++it;
|
||||
if (it && it.index() == j && it.value() != 0.0)
|
||||
invdiag(j) = 1.0 / it.value();
|
||||
else
|
||||
invdiag(j) = 1.0;
|
||||
}
|
||||
|
||||
// Set up GPU context and upload data.
|
||||
GpuContext ctx;
|
||||
GpuContext::setThreadLocal(&ctx);
|
||||
GpuSparseContext<double> spmv_ctx(ctx);
|
||||
auto mat = spmv_ctx.deviceView(H);
|
||||
auto d_invdiag = DeviceMatrix<double>::fromHost(invdiag, ctx.stream());
|
||||
auto d_g = DeviceMatrix<double>::fromHost(g, ctx.stream());
|
||||
|
||||
int last_iters = 0;
|
||||
double last_error = 0;
|
||||
|
||||
for (auto _ : state) {
|
||||
DeviceMatrix<double> d_x(n, 1);
|
||||
d_x.setZero(ctx);
|
||||
DeviceMatrix<double> residual(n, 1);
|
||||
residual.copyFrom(ctx, d_g);
|
||||
|
||||
double rhsNorm2 = d_g.squaredNorm(ctx);
|
||||
double threshold = 1e-8 * 1e-8 * rhsNorm2;
|
||||
double residualNorm2 = residual.squaredNorm(ctx);
|
||||
|
||||
DeviceMatrix<double> p = d_invdiag.cwiseProduct(ctx, residual);
|
||||
DeviceMatrix<double> z(n, 1), tmp(n, 1);
|
||||
|
||||
auto absNew = residual.dot(ctx, p);
|
||||
Index i = 0;
|
||||
Index maxIters = 10000;
|
||||
while (i < maxIters) {
|
||||
tmp.noalias() = mat * p;
|
||||
auto alpha = absNew / p.dot(ctx, tmp);
|
||||
d_x += alpha * p;
|
||||
residual -= alpha * tmp;
|
||||
|
||||
residualNorm2 = residual.squaredNorm(ctx);
|
||||
if (residualNorm2 < threshold) break;
|
||||
|
||||
z.cwiseProduct(ctx, d_invdiag, residual); // in-place, no allocation
|
||||
auto absOld = std::move(absNew);
|
||||
absNew = residual.dot(ctx, z);
|
||||
auto beta = absNew / absOld;
|
||||
|
||||
p *= beta; // device-pointer scal, no host sync
|
||||
p += z;
|
||||
i++;
|
||||
}
|
||||
benchmark::DoNotOptimize(d_x.data());
|
||||
last_iters = i;
|
||||
last_error = std::sqrt(residualNorm2 / rhsNorm2);
|
||||
}
|
||||
|
||||
GpuContext::setThreadLocal(nullptr);
|
||||
|
||||
state.counters["n"] = n;
|
||||
state.counters["nnz"] = H.nonZeros();
|
||||
state.counters["iters"] = last_iters;
|
||||
state.counters["error"] = last_error;
|
||||
state.counters["cameras"] = g_problem.num_cameras;
|
||||
state.counters["points"] = g_problem.num_points;
|
||||
state.counters["observations"] = g_problem.num_observations;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// CPU CG with Jacobi preconditioner (apples-to-apples comparison)
|
||||
// ============================================================================
|
||||
|
||||
static void BM_BA_CPU_CG_Jacobi(benchmark::State& state) {
|
||||
ensure_loaded();
|
||||
const auto& H = g_neq.H;
|
||||
const auto& g = g_neq.g;
|
||||
|
||||
// Eigen's DiagonalPreconditioner is effectively Jacobi.
|
||||
ConjugateGradient<SparseMatrix<double, ColMajor, int>, Lower | Upper> cg;
|
||||
cg.setMaxIterations(10000);
|
||||
cg.setTolerance(1e-8);
|
||||
cg.compute(H);
|
||||
|
||||
int last_iters = 0;
|
||||
double last_error = 0;
|
||||
for (auto _ : state) {
|
||||
VectorXd dx = cg.solve(g);
|
||||
benchmark::DoNotOptimize(dx.data());
|
||||
last_iters = cg.iterations();
|
||||
last_error = cg.error();
|
||||
}
|
||||
|
||||
state.counters["n"] = H.rows();
|
||||
state.counters["nnz"] = H.nonZeros();
|
||||
state.counters["iters"] = last_iters;
|
||||
state.counters["error"] = last_error;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Register benchmarks
|
||||
// ============================================================================
|
||||
|
||||
BENCHMARK(BM_BA_CPU_CG)->Unit(benchmark::kMillisecond);
|
||||
BENCHMARK(BM_BA_CPU_CG_Jacobi)->Unit(benchmark::kMillisecond);
|
||||
BENCHMARK(BM_BA_GPU_CG)->Unit(benchmark::kMillisecond);
|
||||
|
||||
// ============================================================================
|
||||
// Custom main: print summary after benchmarks
|
||||
// ============================================================================
|
||||
|
||||
int main(int argc, char** argv) {
|
||||
benchmark::Initialize(&argc, argv);
|
||||
|
||||
// Print problem info before benchmarks.
|
||||
const char* bal_file = std::getenv("BAL_FILE");
|
||||
if (bal_file) {
|
||||
ensure_loaded();
|
||||
fprintf(stderr,
|
||||
"\n"
|
||||
"=== Bundle Adjustment GPU CG Benchmark ===\n"
|
||||
"BAL file: %s\n"
|
||||
"Cameras: %d\n"
|
||||
"Points: %d\n"
|
||||
"Observations: %d\n"
|
||||
"J size: %d x %d, nnz=%ld\n"
|
||||
"H size: %d x %d, nnz=%ld\n"
|
||||
"|residual|: %.6e\n"
|
||||
"==========================================\n\n",
|
||||
bal_file, g_problem.num_cameras, g_problem.num_points, g_problem.num_observations, g_neq.jacobian_rows,
|
||||
g_neq.jacobian_cols, g_neq.jacobian_nnz, (int)g_neq.H.rows(), (int)g_neq.H.cols(), (long)g_neq.H.nonZeros(),
|
||||
g_neq.residual_norm);
|
||||
}
|
||||
|
||||
benchmark::RunSpecifiedBenchmarks();
|
||||
benchmark::Shutdown();
|
||||
return 0;
|
||||
}
|
||||
291
benchmarks/GPU/bench_gpu_cg_sync.cu
Normal file
291
benchmarks/GPU/bench_gpu_cg_sync.cu
Normal file
@@ -0,0 +1,291 @@
|
||||
// Benchmark: GPU Conjugate Gradient via DeviceMatrix operators.
|
||||
//
|
||||
// Shows the path to running Eigen's CG on GPU with minimal code changes.
|
||||
// The DeviceMatrix benchmark mirrors Eigen's conjugate_gradient() line-by-line.
|
||||
// A raw cuBLAS device-pointer-mode implementation is included as a lower bound.
|
||||
//
|
||||
// The only change needed in Eigen's CG template to support DeviceMatrix:
|
||||
// Line 34: typedef Dest VectorType; (instead of Matrix<Scalar, Dynamic, 1>)
|
||||
//
|
||||
// Usage:
|
||||
// cmake --build build-bench-gpu --target bench_gpu_cg_sync
|
||||
// ./build-bench-gpu/bench_gpu_cg_sync
|
||||
|
||||
#include <benchmark/benchmark.h>
|
||||
|
||||
#include <Eigen/Sparse>
|
||||
#include <Eigen/GPU>
|
||||
#include <cusparse.h>
|
||||
|
||||
using namespace Eigen;
|
||||
|
||||
using Scalar = double;
|
||||
using RealScalar = double;
|
||||
using Vec = Matrix<Scalar, Dynamic, 1>;
|
||||
using SpMat = SparseMatrix<Scalar, ColMajor, int>;
|
||||
|
||||
static SpMat make_spd(Index n) {
|
||||
SpMat A(n, n);
|
||||
A.reserve(VectorXi::Constant(n, 3));
|
||||
for (Index i = 0; i < n; ++i) {
|
||||
A.insert(i, i) = 4.0;
|
||||
if (i > 0) A.insert(i, i - 1) = -1.0;
|
||||
if (i < n - 1) A.insert(i, i + 1) = -1.0;
|
||||
}
|
||||
A.makeCompressed();
|
||||
return A;
|
||||
}
|
||||
|
||||
static void cuda_warmup() {
|
||||
static bool done = false;
|
||||
if (!done) {
|
||||
void* p;
|
||||
cudaMalloc(&p, 1);
|
||||
cudaFree(p);
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
// ==========================================================================
|
||||
// GPU CG using DeviceMatrix operators — mirrors Eigen's conjugate_gradient()
|
||||
// ==========================================================================
|
||||
//
|
||||
// Compare with Eigen/src/IterativeLinearSolvers/ConjugateGradient.h lines 29-84.
|
||||
// Left column: Eigen CG code. Right column: this benchmark.
|
||||
//
|
||||
// Eigen CG GPU CG (this benchmark)
|
||||
// -------- -----------------------
|
||||
// VectorType residual = rhs - mat * x; residual.copyFrom(ctx, rhs); [x=0 so r=b]
|
||||
// RealScalar rhsNorm2 = rhs.sqNorm(); RealScalar rhsNorm2 = rhs.squaredNorm();
|
||||
// ...
|
||||
// tmp.noalias() = mat * p; tmp.noalias() = mat * p; [identical]
|
||||
// Scalar alpha = absNew / p.dot(tmp); Scalar alpha = absNew / p.dot(tmp); [identical]
|
||||
// x += alpha * p; x += alpha * p; [identical]
|
||||
// residual -= alpha * tmp; residual -= alpha * tmp; [identical]
|
||||
// residualNorm2 = residual.sqNorm(); residualNorm2 = residual.squaredNorm(); [identical]
|
||||
// ...
|
||||
// p = z + beta * p; p *= beta; p += z; [equivalent, no alloc]
|
||||
|
||||
static void BM_CG_DeviceMatrixOps(benchmark::State& state) {
|
||||
cuda_warmup();
|
||||
const Index n = state.range(0);
|
||||
|
||||
SpMat A = make_spd(n);
|
||||
Vec b = Vec::Random(n);
|
||||
|
||||
// One shared context: SpMV + BLAS-1 on same stream, zero event overhead.
|
||||
GpuContext ctx;
|
||||
GpuContext::setThreadLocal(&ctx);
|
||||
GpuSparseContext<Scalar> spmv(ctx);
|
||||
auto mat = spmv.deviceView(A);
|
||||
|
||||
// Upload RHS once.
|
||||
auto rhs = DeviceMatrix<Scalar>::fromHost(b, ctx.stream());
|
||||
|
||||
for (auto _ : state) {
|
||||
// --- Eigen CG lines 34-63: initialization ---
|
||||
// typedef Dest VectorType; // GPU CHANGE: was Matrix<Scalar,Dynamic,1>
|
||||
// VectorType residual = rhs - mat * x; // x=0, so residual = rhs
|
||||
DeviceMatrix<Scalar> x(n, 1);
|
||||
x.setZero();
|
||||
DeviceMatrix<Scalar> residual(n, 1);
|
||||
residual.copyFrom(ctx, rhs);
|
||||
|
||||
// RealScalar rhsNorm2 = rhs.squaredNorm();
|
||||
RealScalar rhsNorm2 = rhs.squaredNorm();
|
||||
if (rhsNorm2 == 0) continue;
|
||||
|
||||
RealScalar tol = 1e-10;
|
||||
const RealScalar considerAsZero = (std::numeric_limits<RealScalar>::min)();
|
||||
RealScalar threshold = numext::maxi(RealScalar(tol * tol * rhsNorm2), considerAsZero);
|
||||
|
||||
// RealScalar residualNorm2 = residual.squaredNorm();
|
||||
RealScalar residualNorm2 = residual.squaredNorm();
|
||||
if (residualNorm2 < threshold) continue;
|
||||
|
||||
// VectorType p(n);
|
||||
// p = precond.solve(residual); // no preconditioner: p = residual
|
||||
DeviceMatrix<Scalar> p(n, 1);
|
||||
p.copyFrom(ctx, residual);
|
||||
|
||||
// VectorType z(n), tmp(n);
|
||||
DeviceMatrix<Scalar> z(n, 1), tmp(n, 1);
|
||||
|
||||
// auto absNew = numext::real(residual.dot(p));
|
||||
// DeviceScalar — stays on device, no sync.
|
||||
auto absNew = residual.dot(p); // DeviceScalar, no sync
|
||||
|
||||
// while (i < maxIters) {
|
||||
Index maxIters = 200;
|
||||
Index i = 0;
|
||||
while (i < maxIters) {
|
||||
// tmp.noalias() = mat * p;
|
||||
tmp.noalias() = mat * p; // SpMV, device-resident
|
||||
|
||||
// auto alpha = absNew / p.dot(tmp);
|
||||
// DeviceScalar / DeviceScalar → device kernel, no sync!
|
||||
auto alpha = absNew / p.dot(tmp); // DeviceScalar, no sync
|
||||
|
||||
// x += alpha * p;
|
||||
// DeviceScalar * DeviceMatrix → device-pointer axpy, no sync!
|
||||
x += alpha * p;
|
||||
|
||||
// residual -= alpha * tmp;
|
||||
residual -= alpha * tmp; // device-pointer axpy, no sync
|
||||
|
||||
// residualNorm2 = residual.squaredNorm();
|
||||
residualNorm2 = residual.squaredNorm(); // THE one sync per iteration
|
||||
|
||||
// if (residualNorm2 < threshold) break;
|
||||
if (residualNorm2 < threshold) break;
|
||||
|
||||
// z = precond.solve(residual);
|
||||
z.copyFrom(ctx, residual); // no preconditioner
|
||||
|
||||
// auto absOld = std::move(absNew);
|
||||
auto absOld = std::move(absNew); // no sync, no alloc
|
||||
|
||||
// absNew = numext::real(residual.dot(z));
|
||||
absNew = residual.dot(z); // DeviceScalar, no sync
|
||||
|
||||
// auto beta = absNew / absOld;
|
||||
// DeviceScalar / DeviceScalar → device kernel, no sync!
|
||||
auto beta = absNew / absOld; // DeviceScalar, no sync
|
||||
|
||||
// p = z + beta * p;
|
||||
p *= beta; // device-pointer scal, no host sync
|
||||
p += z;
|
||||
|
||||
i++;
|
||||
}
|
||||
}
|
||||
|
||||
GpuContext::setThreadLocal(nullptr);
|
||||
state.SetItemsProcessed(state.iterations() * 200);
|
||||
}
|
||||
|
||||
BENCHMARK(BM_CG_DeviceMatrixOps)->RangeMultiplier(4)->Range(1 << 10, 1 << 20);
|
||||
|
||||
// ==========================================================================
|
||||
// Raw cuBLAS device-pointer-mode CG (1 sync/iter) — performance lower bound
|
||||
// ==========================================================================
|
||||
|
||||
__global__ void scalar_div_kernel(const Scalar* a, const Scalar* b, Scalar* out) { *out = *a / *b; }
|
||||
__global__ void scalar_neg_kernel(const Scalar* in, Scalar* out) { *out = -(*in); }
|
||||
|
||||
static void BM_CG_DevicePointerMode(benchmark::State& state) {
|
||||
cuda_warmup();
|
||||
const Index n = state.range(0);
|
||||
const int maxIters = 200;
|
||||
|
||||
SpMat A = make_spd(n);
|
||||
Vec b = Vec::Random(n);
|
||||
|
||||
cudaStream_t stream;
|
||||
cudaStreamCreate(&stream);
|
||||
cublasHandle_t cublas;
|
||||
cublasCreate(&cublas);
|
||||
cublasSetStream(cublas, stream);
|
||||
|
||||
cusparseHandle_t cusparse;
|
||||
cusparseCreate(&cusparse);
|
||||
cusparseSetStream(cusparse, stream);
|
||||
|
||||
internal::DeviceBuffer d_outer((n + 1) * sizeof(int));
|
||||
internal::DeviceBuffer d_inner(A.nonZeros() * sizeof(int));
|
||||
internal::DeviceBuffer d_vals(A.nonZeros() * sizeof(Scalar));
|
||||
cudaMemcpy(d_outer.ptr, A.outerIndexPtr(), (n + 1) * sizeof(int), cudaMemcpyHostToDevice);
|
||||
cudaMemcpy(d_inner.ptr, A.innerIndexPtr(), A.nonZeros() * sizeof(int), cudaMemcpyHostToDevice);
|
||||
cudaMemcpy(d_vals.ptr, A.valuePtr(), A.nonZeros() * sizeof(Scalar), cudaMemcpyHostToDevice);
|
||||
|
||||
cusparseSpMatDescr_t matA;
|
||||
cusparseCreateCsc(&matA, n, n, A.nonZeros(), d_outer.ptr, d_inner.ptr, d_vals.ptr, CUSPARSE_INDEX_32I,
|
||||
CUSPARSE_INDEX_32I, CUSPARSE_INDEX_BASE_ZERO, CUDA_R_64F);
|
||||
|
||||
internal::DeviceBuffer d_tmp_buf(n * sizeof(Scalar));
|
||||
cusparseDnVecDescr_t tmp_x, tmp_y;
|
||||
cusparseCreateDnVec(&tmp_x, n, d_tmp_buf.ptr, CUDA_R_64F);
|
||||
cusparseCreateDnVec(&tmp_y, n, d_tmp_buf.ptr, CUDA_R_64F);
|
||||
Scalar spmv_alpha = 1.0, spmv_beta = 0.0;
|
||||
size_t ws_size = 0;
|
||||
cusparseSpMV_bufferSize(cusparse, CUSPARSE_OPERATION_NON_TRANSPOSE, &spmv_alpha, matA, tmp_x, &spmv_beta, tmp_y,
|
||||
CUDA_R_64F, CUSPARSE_SPMV_ALG_DEFAULT, &ws_size);
|
||||
internal::DeviceBuffer d_workspace(ws_size);
|
||||
cusparseDestroyDnVec(tmp_x);
|
||||
cusparseDestroyDnVec(tmp_y);
|
||||
|
||||
internal::DeviceBuffer d_x(n * sizeof(Scalar)), d_r(n * sizeof(Scalar));
|
||||
internal::DeviceBuffer d_p(n * sizeof(Scalar)), d_tmp(n * sizeof(Scalar));
|
||||
internal::DeviceBuffer d_b(n * sizeof(Scalar));
|
||||
internal::DeviceBuffer d_absNew(sizeof(Scalar)), d_absOld(sizeof(Scalar));
|
||||
internal::DeviceBuffer d_pdot(sizeof(Scalar)), d_alpha(sizeof(Scalar));
|
||||
internal::DeviceBuffer d_neg_alpha(sizeof(Scalar)), d_beta(sizeof(Scalar));
|
||||
internal::DeviceBuffer d_rnorm(sizeof(RealScalar));
|
||||
|
||||
cudaMemcpy(d_b.ptr, b.data(), n * sizeof(Scalar), cudaMemcpyHostToDevice);
|
||||
|
||||
auto spmv = [&](Scalar* x_ptr, Scalar* y_ptr) {
|
||||
cusparseDnVecDescr_t vx, vy;
|
||||
cusparseCreateDnVec(&vx, n, x_ptr, CUDA_R_64F);
|
||||
cusparseCreateDnVec(&vy, n, y_ptr, CUDA_R_64F);
|
||||
cusparseSpMV(cusparse, CUSPARSE_OPERATION_NON_TRANSPOSE, &spmv_alpha, matA, vx, &spmv_beta, vy, CUDA_R_64F,
|
||||
CUSPARSE_SPMV_ALG_DEFAULT, d_workspace.ptr);
|
||||
cusparseDestroyDnVec(vx);
|
||||
cusparseDestroyDnVec(vy);
|
||||
};
|
||||
|
||||
for (auto _ : state) {
|
||||
cudaMemsetAsync(static_cast<Scalar*>(d_x.ptr), 0, n * sizeof(Scalar), stream);
|
||||
cudaMemcpyAsync(d_r.ptr, d_b.ptr, n * sizeof(Scalar), cudaMemcpyDeviceToDevice, stream);
|
||||
cudaMemcpyAsync(d_p.ptr, d_b.ptr, n * sizeof(Scalar), cudaMemcpyDeviceToDevice, stream);
|
||||
|
||||
cublasSetPointerMode(cublas, CUBLAS_POINTER_MODE_DEVICE);
|
||||
cublasDdot(cublas, n, static_cast<Scalar*>(d_r.ptr), 1, static_cast<Scalar*>(d_p.ptr), 1,
|
||||
static_cast<Scalar*>(d_absNew.ptr));
|
||||
|
||||
for (int i = 0; i < maxIters; ++i) {
|
||||
spmv(static_cast<Scalar*>(d_p.ptr), static_cast<Scalar*>(d_tmp.ptr));
|
||||
|
||||
cublasDdot(cublas, n, static_cast<Scalar*>(d_p.ptr), 1, static_cast<Scalar*>(d_tmp.ptr), 1,
|
||||
static_cast<Scalar*>(d_pdot.ptr));
|
||||
|
||||
scalar_div_kernel<<<1, 1, 0, stream>>>(static_cast<Scalar*>(d_absNew.ptr), static_cast<Scalar*>(d_pdot.ptr),
|
||||
static_cast<Scalar*>(d_alpha.ptr));
|
||||
scalar_neg_kernel<<<1, 1, 0, stream>>>(static_cast<Scalar*>(d_alpha.ptr), static_cast<Scalar*>(d_neg_alpha.ptr));
|
||||
|
||||
cublasDaxpy(cublas, n, static_cast<Scalar*>(d_alpha.ptr), static_cast<Scalar*>(d_p.ptr), 1,
|
||||
static_cast<Scalar*>(d_x.ptr), 1);
|
||||
cublasDaxpy(cublas, n, static_cast<Scalar*>(d_neg_alpha.ptr), static_cast<Scalar*>(d_tmp.ptr), 1,
|
||||
static_cast<Scalar*>(d_r.ptr), 1);
|
||||
|
||||
cublasDnrm2(cublas, n, static_cast<Scalar*>(d_r.ptr), 1, static_cast<RealScalar*>(d_rnorm.ptr));
|
||||
|
||||
RealScalar rnorm;
|
||||
cudaMemcpyAsync(&rnorm, d_rnorm.ptr, sizeof(RealScalar), cudaMemcpyDeviceToHost, stream);
|
||||
cudaStreamSynchronize(stream);
|
||||
if (rnorm * rnorm < 1e-20) break;
|
||||
|
||||
cudaMemcpyAsync(d_absOld.ptr, d_absNew.ptr, sizeof(Scalar), cudaMemcpyDeviceToDevice, stream);
|
||||
cublasDdot(cublas, n, static_cast<Scalar*>(d_r.ptr), 1, static_cast<Scalar*>(d_r.ptr), 1,
|
||||
static_cast<Scalar*>(d_absNew.ptr));
|
||||
|
||||
scalar_div_kernel<<<1, 1, 0, stream>>>(static_cast<Scalar*>(d_absNew.ptr), static_cast<Scalar*>(d_absOld.ptr),
|
||||
static_cast<Scalar*>(d_beta.ptr));
|
||||
|
||||
cublasDscal(cublas, n, static_cast<Scalar*>(d_beta.ptr), static_cast<Scalar*>(d_p.ptr), 1);
|
||||
cublasSetPointerMode(cublas, CUBLAS_POINTER_MODE_HOST);
|
||||
Scalar one = 1.0;
|
||||
cublasDaxpy(cublas, n, &one, static_cast<Scalar*>(d_r.ptr), 1, static_cast<Scalar*>(d_p.ptr), 1);
|
||||
cublasSetPointerMode(cublas, CUBLAS_POINTER_MODE_DEVICE);
|
||||
}
|
||||
cudaStreamSynchronize(stream);
|
||||
}
|
||||
|
||||
state.SetItemsProcessed(state.iterations() * maxIters);
|
||||
cusparseDestroySpMat(matA);
|
||||
cusparseDestroy(cusparse);
|
||||
cublasDestroy(cublas);
|
||||
cudaStreamDestroy(stream);
|
||||
}
|
||||
|
||||
BENCHMARK(BM_CG_DevicePointerMode)->RangeMultiplier(4)->Range(1 << 10, 1 << 20);
|
||||
216
benchmarks/GPU/bench_gpu_cg_vs_cpu.cu
Normal file
216
benchmarks/GPU/bench_gpu_cg_vs_cpu.cu
Normal file
@@ -0,0 +1,216 @@
|
||||
// Benchmark: GPU CG vs CPU CG on realistic sparse systems.
|
||||
//
|
||||
// Tests 2D Laplacian (5-point stencil) and 3D Laplacian (7-point stencil)
|
||||
// in both float and double precision.
|
||||
//
|
||||
// Usage:
|
||||
// cmake --build build-bench-gpu --target bench_gpu_cg_vs_cpu
|
||||
// ./build-bench-gpu/bench_gpu_cg_vs_cpu
|
||||
|
||||
#include <benchmark/benchmark.h>
|
||||
|
||||
#include <Eigen/Sparse>
|
||||
#include <Eigen/IterativeLinearSolvers>
|
||||
#include <Eigen/GPU>
|
||||
|
||||
using namespace Eigen;
|
||||
|
||||
// ---- Sparse matrix generators -----------------------------------------------
|
||||
|
||||
template <typename Scalar>
|
||||
SparseMatrix<Scalar, ColMajor, int> make_laplacian_2d(int grid_n) {
|
||||
using SpMat = SparseMatrix<Scalar, ColMajor, int>;
|
||||
const int n = grid_n * grid_n;
|
||||
SpMat A(n, n);
|
||||
A.reserve(VectorXi::Constant(n, 5));
|
||||
for (int i = 0; i < grid_n; ++i) {
|
||||
for (int j = 0; j < grid_n; ++j) {
|
||||
int idx = i * grid_n + j;
|
||||
A.insert(idx, idx) = Scalar(4);
|
||||
if (i > 0) A.insert(idx, idx - grid_n) = Scalar(-1);
|
||||
if (i < grid_n - 1) A.insert(idx, idx + grid_n) = Scalar(-1);
|
||||
if (j > 0) A.insert(idx, idx - 1) = Scalar(-1);
|
||||
if (j < grid_n - 1) A.insert(idx, idx + 1) = Scalar(-1);
|
||||
}
|
||||
}
|
||||
A.makeCompressed();
|
||||
return A;
|
||||
}
|
||||
|
||||
template <typename Scalar>
|
||||
SparseMatrix<Scalar, ColMajor, int> make_laplacian_3d(int grid_n) {
|
||||
using SpMat = SparseMatrix<Scalar, ColMajor, int>;
|
||||
const int n = grid_n * grid_n * grid_n;
|
||||
const int n2 = grid_n * grid_n;
|
||||
SpMat A(n, n);
|
||||
A.reserve(VectorXi::Constant(n, 7));
|
||||
for (int i = 0; i < grid_n; ++i) {
|
||||
for (int j = 0; j < grid_n; ++j) {
|
||||
for (int k = 0; k < grid_n; ++k) {
|
||||
int idx = i * n2 + j * grid_n + k;
|
||||
A.insert(idx, idx) = Scalar(6);
|
||||
if (i > 0) A.insert(idx, idx - n2) = Scalar(-1);
|
||||
if (i < grid_n - 1) A.insert(idx, idx + n2) = Scalar(-1);
|
||||
if (j > 0) A.insert(idx, idx - grid_n) = Scalar(-1);
|
||||
if (j < grid_n - 1) A.insert(idx, idx + grid_n) = Scalar(-1);
|
||||
if (k > 0) A.insert(idx, idx - 1) = Scalar(-1);
|
||||
if (k < grid_n - 1) A.insert(idx, idx + 1) = Scalar(-1);
|
||||
}
|
||||
}
|
||||
}
|
||||
A.makeCompressed();
|
||||
return A;
|
||||
}
|
||||
|
||||
static void cuda_warmup() {
|
||||
static bool done = false;
|
||||
if (!done) {
|
||||
void* p;
|
||||
cudaMalloc(&p, 1);
|
||||
cudaFree(p);
|
||||
done = true;
|
||||
}
|
||||
}
|
||||
|
||||
// ---- CPU CG -----------------------------------------------------------------
|
||||
|
||||
template <typename Scalar, typename MatGen>
|
||||
void run_cpu_cg(benchmark::State& state, MatGen make_matrix) {
|
||||
using SpMat = SparseMatrix<Scalar, ColMajor, int>;
|
||||
using Vec = Matrix<Scalar, Dynamic, 1>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
const int grid_n = state.range(0);
|
||||
SpMat A = make_matrix(grid_n);
|
||||
Vec b = Vec::Random(A.rows());
|
||||
|
||||
ConjugateGradient<SpMat, Lower | Upper> cg;
|
||||
cg.setMaxIterations(10000);
|
||||
cg.setTolerance(RealScalar(1e-8));
|
||||
cg.compute(A);
|
||||
|
||||
int last_iters = 0;
|
||||
for (auto _ : state) {
|
||||
Vec x = cg.solve(b);
|
||||
benchmark::DoNotOptimize(x.data());
|
||||
last_iters = cg.iterations();
|
||||
}
|
||||
state.counters["n"] = A.rows();
|
||||
state.counters["nnz"] = A.nonZeros();
|
||||
state.counters["iters"] = last_iters;
|
||||
state.counters["error"] = cg.error();
|
||||
}
|
||||
|
||||
// ---- GPU CG -----------------------------------------------------------------
|
||||
|
||||
template <typename Scalar, typename MatGen>
|
||||
void run_gpu_cg(benchmark::State& state, MatGen make_matrix) {
|
||||
using SpMat = SparseMatrix<Scalar, ColMajor, int>;
|
||||
using Vec = Matrix<Scalar, Dynamic, 1>;
|
||||
using RealScalar = typename NumTraits<Scalar>::Real;
|
||||
|
||||
cuda_warmup();
|
||||
const int grid_n = state.range(0);
|
||||
SpMat A = make_matrix(grid_n);
|
||||
const Index n = A.rows();
|
||||
Vec b = Vec::Random(n);
|
||||
|
||||
// Extract inverse diagonal.
|
||||
Vec invdiag(n);
|
||||
for (Index j = 0; j < A.outerSize(); ++j) {
|
||||
typename SpMat::InnerIterator it(A, j);
|
||||
while (it && it.index() != j) ++it;
|
||||
if (it && it.index() == j && it.value() != Scalar(0))
|
||||
invdiag(j) = Scalar(1) / it.value();
|
||||
else
|
||||
invdiag(j) = Scalar(1);
|
||||
}
|
||||
|
||||
GpuContext ctx;
|
||||
GpuContext::setThreadLocal(&ctx);
|
||||
GpuSparseContext<Scalar> spmv_ctx(ctx);
|
||||
auto mat = spmv_ctx.deviceView(A);
|
||||
auto d_invdiag = DeviceMatrix<Scalar>::fromHost(invdiag, ctx.stream());
|
||||
auto d_b = DeviceMatrix<Scalar>::fromHost(b, ctx.stream());
|
||||
|
||||
int last_iters = 0;
|
||||
RealScalar last_error = 0;
|
||||
|
||||
for (auto _ : state) {
|
||||
DeviceMatrix<Scalar> d_x(n, 1);
|
||||
d_x.setZero(ctx);
|
||||
DeviceMatrix<Scalar> residual(n, 1);
|
||||
residual.copyFrom(ctx, d_b);
|
||||
|
||||
RealScalar rhsNorm2 = d_b.squaredNorm(ctx);
|
||||
RealScalar tol = RealScalar(1e-8);
|
||||
RealScalar threshold = tol * tol * rhsNorm2;
|
||||
RealScalar residualNorm2 = residual.squaredNorm(ctx);
|
||||
|
||||
DeviceMatrix<Scalar> p = d_invdiag.cwiseProduct(ctx, residual);
|
||||
DeviceMatrix<Scalar> z(n, 1), tmp(n, 1);
|
||||
|
||||
auto absNew = residual.dot(ctx, p);
|
||||
Index i = 0;
|
||||
Index maxIters = 10000;
|
||||
while (i < maxIters) {
|
||||
tmp.noalias() = mat * p;
|
||||
auto alpha = absNew / p.dot(ctx, tmp);
|
||||
d_x += alpha * p;
|
||||
residual -= alpha * tmp;
|
||||
|
||||
residualNorm2 = residual.squaredNorm(ctx);
|
||||
if (residualNorm2 < threshold) break;
|
||||
|
||||
z.cwiseProduct(ctx, d_invdiag, residual);
|
||||
auto absOld = std::move(absNew);
|
||||
absNew = residual.dot(ctx, z);
|
||||
auto beta = absNew / absOld;
|
||||
|
||||
p *= beta;
|
||||
p += z;
|
||||
i++;
|
||||
}
|
||||
benchmark::DoNotOptimize(d_x.data());
|
||||
last_iters = i;
|
||||
last_error = numext::sqrt(residualNorm2 / rhsNorm2);
|
||||
}
|
||||
|
||||
GpuContext::setThreadLocal(nullptr);
|
||||
state.counters["n"] = n;
|
||||
state.counters["nnz"] = A.nonZeros();
|
||||
state.counters["iters"] = last_iters;
|
||||
state.counters["error"] = last_error;
|
||||
}
|
||||
|
||||
// ---- 2D Laplacian, double ---------------------------------------------------
|
||||
|
||||
static void BM_CG_CPU_2D_double(benchmark::State& state) { run_cpu_cg<double>(state, make_laplacian_2d<double>); }
|
||||
static void BM_CG_GPU_2D_double(benchmark::State& state) { run_gpu_cg<double>(state, make_laplacian_2d<double>); }
|
||||
|
||||
BENCHMARK(BM_CG_CPU_2D_double)->ArgsProduct({{32, 64, 128, 256, 512}});
|
||||
BENCHMARK(BM_CG_GPU_2D_double)->ArgsProduct({{32, 64, 128, 256, 512}});
|
||||
|
||||
// ---- 2D Laplacian, float ----------------------------------------------------
|
||||
|
||||
static void BM_CG_CPU_2D_float(benchmark::State& state) { run_cpu_cg<float>(state, make_laplacian_2d<float>); }
|
||||
static void BM_CG_GPU_2D_float(benchmark::State& state) { run_gpu_cg<float>(state, make_laplacian_2d<float>); }
|
||||
|
||||
BENCHMARK(BM_CG_CPU_2D_float)->ArgsProduct({{32, 64, 128, 256, 512}});
|
||||
BENCHMARK(BM_CG_GPU_2D_float)->ArgsProduct({{32, 64, 128, 256, 512}});
|
||||
|
||||
// ---- 3D Laplacian, double ---------------------------------------------------
|
||||
|
||||
static void BM_CG_CPU_3D_double(benchmark::State& state) { run_cpu_cg<double>(state, make_laplacian_3d<double>); }
|
||||
static void BM_CG_GPU_3D_double(benchmark::State& state) { run_gpu_cg<double>(state, make_laplacian_3d<double>); }
|
||||
|
||||
BENCHMARK(BM_CG_CPU_3D_double)->ArgsProduct({{16, 32, 48, 64}});
|
||||
BENCHMARK(BM_CG_GPU_3D_double)->ArgsProduct({{16, 32, 48, 64}});
|
||||
|
||||
// ---- 3D Laplacian, float ----------------------------------------------------
|
||||
|
||||
static void BM_CG_CPU_3D_float(benchmark::State& state) { run_cpu_cg<float>(state, make_laplacian_3d<float>); }
|
||||
static void BM_CG_GPU_3D_float(benchmark::State& state) { run_gpu_cg<float>(state, make_laplacian_3d<float>); }
|
||||
|
||||
BENCHMARK(BM_CG_CPU_3D_float)->ArgsProduct({{16, 32, 48, 64}});
|
||||
BENCHMARK(BM_CG_GPU_3D_float)->ArgsProduct({{16, 32, 48, 64}});
|
||||
Reference in New Issue
Block a user