CUDA CGNR, Part 4: CudaCgnrSolver

* Added CudaCgnrSolver, a new CUDA-accelerated CGNR.
* To use CudaCgnrSolver, the user must select CGNR as the linear_solver
  and CUDA_SPARSE as the sparse_linear_algebra_library.
* Updated ConjugateGradientSolver to work with an array of pointers to
  scratch to support CudaVectors as scratch.
* Moved CUDA initialization to run in Solver::Solve as needed.

Some performance comparisons on an Ubuntu 20.04 desktop with an
Intel i9-9940X CPU @ 3.30GHz, and an nVidia Quadro RTX 6000,
all configurations run with 24 threads, and 10 iterations.

=================================================
CGNR + CUDA_SPARSE + IDENTITY Preconditioner
problem-1778-993923-pre.txt
=================================================
Cost:
Initial                          2.563973e+08
Final                            1.724755e+06
Change                           2.546725e+08

Minimizer iterations                       11
Successful steps                            7
Unsuccessful steps                          4

Time (in seconds):
Preprocessor                         4.020158

  Residual only evaluation           1.567092 (10)
  Jacobian & residual evaluation     7.847130 (7)
  Linear solver                     31.688898 (10)
Minimizer                           46.834987

Postprocessor                        0.353974
Total                               51.209120

=================================================
SPARSE_SCHUR (CPU) + SUITE_SPARSE + AMD
problem-1778-993923-pre.txt
=================================================
Cost:
Initial                          2.563973e+08
Final                            1.651617e+06
Change                           2.547457e+08

Minimizer iterations                       11
Successful steps                           11
Unsuccessful steps                          0

Time (in seconds):
Preprocessor                        35.812003

  Residual only evaluation           1.658980 (10)
  Jacobian & residual evaluation    12.218799 (11)
  Linear solver                     76.409992 (10)
Minimizer                           98.809773

Postprocessor                        0.372712
Total                              134.994489

=================================================
ITERATIVE_SCHUR (CPU) + JACOBI Preconditioner
problem-1778-993923-pre.txt
=================================================
Cost:
Initial                          2.563973e+08
Final                            1.684447e+06
Change                           2.547128e+08

Minimizer iterations                       11
Successful steps                            8
Unsuccessful steps                          3

Time (in seconds):
Preprocessor                        15.331614

  Residual only evaluation           1.606114 (10)
  Jacobian & residual evaluation     8.502166 (8)
  Linear solver                    351.910080 (10)
Minimizer                          368.797327

Postprocessor                        0.363536
Total                              384.492478

=================================================
CGNR + CUDA_SPARSE + IDENTITY Preconditioner
problem-13682-4456117-pre.txt
=================================================
Cost:
Initial                          1.126372e+09
Final                            2.269329e+07
Change                           1.103678e+09

Minimizer iterations                       11
Successful steps                            7
Unsuccessful steps                          4

Time (in seconds):
Preprocessor                        19.140087

  Residual only evaluation           8.721920 (10)
  Jacobian & residual evaluation    41.955923 (7)
  Linear solver                    214.121861 (10)
Minimizer                          296.636890

Postprocessor                        1.971827
Total                              317.748804

Change-Id: I3a09f31aa6903f661e91f595afd39d427583e856
This commit is contained in:
Joydeep Biswas
2022-08-14 19:14:04 -05:00
parent 6ab435d774
commit 829089053e
21 changed files with 413 additions and 59 deletions
+65 -13
View File
@@ -327,7 +327,8 @@ bool OptionsAreValidForCgnr(const Solver::Options& options, string* error) {
return false;
}
if (options.preconditioner_type == SUBSET) {
if (options.sparse_linear_algebra_library_type != CUDA_SPARSE &&
options.preconditioner_type == SUBSET) {
if (options.residual_blocks_for_subset_preconditioner.empty()) {
*error =
"When using SUBSET preconditioner, "
@@ -341,6 +342,20 @@ bool OptionsAreValidForCgnr(const Solver::Options& options, string* error) {
}
}
// Check options for CGNR with CUDA_SPARSE.
if (options.sparse_linear_algebra_library_type == CUDA_SPARSE) {
if (!IsSparseLinearAlgebraLibraryTypeAvailable(CUDA_SPARSE)) {
*error = "Can't use CGNR with sparse_linear_algebra_library_type = "
"CUDA_SPARSE because support was not enabled when Ceres was built.";
return false;
}
if (options.preconditioner_type != IDENTITY) {
StringPrintf("Can't use CGNR with preconditioner_type = %s when "
"sparse_linear_algebra_library_type = CUDA_SPARSE.",
PreconditionerTypeToString(options.preconditioner_type));
return false;
}
}
return true;
}
@@ -652,6 +667,18 @@ std::string SchurStructureToString(const int row_block_size,
return internal::StringPrintf("%s,%s,%s", row.c_str(), e.c_str(), f.c_str());
}
bool IsCudaRequired(const Solver::Options& options) {
if (options.linear_solver_type == DENSE_NORMAL_CHOLESKY ||
options.linear_solver_type == DENSE_SCHUR ||
options.linear_solver_type == DENSE_QR) {
return (options.dense_linear_algebra_library_type == CUDA);
}
if (options.linear_solver_type == CGNR) {
return (options.sparse_linear_algebra_library_type == CUDA_SPARSE);
}
return false;
}
} // namespace
bool Solver::Options::IsValid(string* error) const {
@@ -697,6 +724,13 @@ void Solver::Solve(const Solver::Options& options,
Program* program = problem_impl->mutable_program();
PreSolveSummarize(options, problem_impl, summary);
if (IsCudaRequired(options)) {
if (!problem_impl->context()->InitCUDA(&summary->message)) {
LOG(ERROR) << "Terminating: " << summary->message;
return;
}
}
// If gradient_checking is enabled, wrap all cost functions in a
// gradient checker and install a callback that terminates if any gradient
// error is detected.
@@ -866,22 +900,40 @@ string Solver::Summary::FullReport() const {
}
}
if (linear_solver_type_used == SPARSE_NORMAL_CHOLESKY ||
const bool used_sparse_linear_algebra_library =
linear_solver_type_used == SPARSE_NORMAL_CHOLESKY ||
linear_solver_type_used == SPARSE_SCHUR ||
(linear_solver_type_used == CGNR &&
preconditioner_type_used == SUBSET) ||
linear_solver_type_used == CGNR ||
(linear_solver_type_used == ITERATIVE_SCHUR &&
(preconditioner_type_used == CLUSTER_JACOBI ||
preconditioner_type_used == CLUSTER_TRIDIAGONAL))) {
(preconditioner_type_used == CLUSTER_JACOBI ||
preconditioner_type_used == CLUSTER_TRIDIAGONAL));
const bool linear_solver_ordering_required =
linear_solver_type_used == SPARSE_SCHUR ||
(linear_solver_type_used == ITERATIVE_SCHUR &&
(preconditioner_type_used == CLUSTER_JACOBI ||
preconditioner_type_used == CLUSTER_TRIDIAGONAL)) ||
(linear_solver_type_used == CGNR && preconditioner_type_used == SUBSET);
if (used_sparse_linear_algebra_library) {
const char* mixed_precision_suffix =
(mixed_precision_solves_used ? "(Mixed Precision)" : "");
StringAppendF(
&report,
"\nSparse linear algebra library %15s + %s %s\n",
SparseLinearAlgebraLibraryTypeToString(
sparse_linear_algebra_library_type),
LinearSolverOrderingTypeToString(linear_solver_ordering_type),
mixed_precision_suffix);
if (linear_solver_ordering_required) {
StringAppendF(
&report,
"\nSparse linear algebra library %15s + %s %s\n",
SparseLinearAlgebraLibraryTypeToString(
sparse_linear_algebra_library_type),
LinearSolverOrderingTypeToString(linear_solver_ordering_type),
mixed_precision_suffix);
} else {
StringAppendF(
&report,
"\nSparse linear algebra library %15s %s\n",
SparseLinearAlgebraLibraryTypeToString(
sparse_linear_algebra_library_type),
mixed_precision_suffix);
}
}
StringAppendF(&report, "\n");