mirror of
https://github.com/ceres-solver/ceres-solver.git
synced 2026-08-29 08:34:37 +08:00
f0c3b23684
Previously, the thread ID was acquired and released on every iteration of the for loop. The C++11 concurrent queue implementation is much slower than TBB's version and consequently this was a huge bottleneck. This introduces another ParallelFor API which takes the thread ID as a parameter in the evaluation function. This allows us to acquire and release the thread ID for each block of work which drastically improves the performance. This change brings us on par with OpenMP and TBB. See below for a timing comparison. Note: in this example this CLs C++11 version is faster to compute the residuals because TBB still must acquire the thread ID on every iteration, which has some overhead. Tested by building and running tests for no threading, OpenMP, TBB, and C++11 threads. Also ran bazel tests. ./bin/bundle_adjuster --input=problem-744-543562-pre.txt --num_threads=8 C++11 @Head Time (in seconds): Residual only evaluation 7.819692 (5) Jacobian & residual evaluation 11.606063 (6) Linear solver 47.860195 (5) Minimizer 70.877072 Total 90.806338 --------------------------------------------------- C++11 (This CL) Time (in seconds): Residual only evaluation 1.217500 (5) Jacobian & residual evaluation 5.796112 (6) Linear solver 44.080873 (5) Minimizer 54.635524 Total 77.640072 --------------------------------------------------- OpenMP Time (in seconds): Residual only evaluation 0.797023 (5) Jacobian & residual evaluation 5.633916 (6) Linear solver 43.280020 (5) Minimizer 53.199058 Total 76.250861 --------------------------------------------------- TBB Time (in seconds): Residual only evaluation 1.911095 (5) Jacobian & residual evaluation 5.557807 (6) Linear solver 44.074680 (5) Minimizer 55.002688 Total 78.052687 --------------------------------------------------- No Threads Time (in seconds): Residual only evaluation 2.939212 (5) Jacobian & residual evaluation 18.519874 (6) Linear solver 74.017837 (5) Minimizer 98.980080 Total 122.216391 Change-Id: I3af959b0771bbdfe8cad8c13896191d6ac903181
93 lines
3.1 KiB
C++
93 lines
3.1 KiB
C++
// Ceres Solver - A fast non-linear least squares minimizer
|
|
// Copyright 2018 Google Inc. All rights reserved.
|
|
// http://ceres-solver.org/
|
|
//
|
|
// Redistribution and use in source and binary forms, with or without
|
|
// modification, are permitted provided that the following conditions are met:
|
|
//
|
|
// * Redistributions of source code must retain the above copyright notice,
|
|
// this list of conditions and the following disclaimer.
|
|
// * Redistributions in binary form must reproduce the above copyright notice,
|
|
// this list of conditions and the following disclaimer in the documentation
|
|
// and/or other materials provided with the distribution.
|
|
// * Neither the name of Google Inc. nor the names of its contributors may be
|
|
// used to endorse or promote products derived from this software without
|
|
// specific prior written permission.
|
|
//
|
|
// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
// AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
// IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
|
// ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE
|
|
// LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
|
// CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
|
// SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
|
// INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
|
// CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
|
// ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
|
// POSSIBILITY OF SUCH DAMAGE.
|
|
//
|
|
// Author: vitus@google.com (Michael Vitus)
|
|
|
|
// This include must come before any #ifndef check on Ceres compile options.
|
|
#include "ceres/internal/port.h"
|
|
|
|
#ifdef CERES_USE_TBB
|
|
|
|
#include "ceres/parallel_for.h"
|
|
|
|
#include <tbb/parallel_for.h>
|
|
#include <tbb/task_arena.h>
|
|
|
|
#include "ceres/scoped_thread_token.h"
|
|
#include "ceres/thread_token_provider.h"
|
|
#include "glog/logging.h"
|
|
|
|
namespace ceres {
|
|
namespace internal {
|
|
|
|
void ParallelFor(ContextImpl* context,
|
|
int start,
|
|
int end,
|
|
int num_threads,
|
|
const std::function<void(int)>& function) {
|
|
CHECK_GT(num_threads, 0);
|
|
CHECK(context != NULL);
|
|
if (end <= start) {
|
|
return;
|
|
}
|
|
|
|
// Fast path for when it is single threaded.
|
|
if (num_threads == 1) {
|
|
for (int i = start; i < end; ++i) {
|
|
function(i);
|
|
}
|
|
return;
|
|
}
|
|
|
|
tbb::task_arena task_arena(num_threads);
|
|
task_arena.execute([&]{
|
|
tbb::parallel_for(start, end, function);
|
|
});
|
|
}
|
|
|
|
void ParallelFor(ContextImpl* context,
|
|
int start,
|
|
int end,
|
|
int num_threads,
|
|
const std::function<void(int thread_id, int i)>& function) {
|
|
CHECK(context != NULL);
|
|
|
|
ThreadTokenProvider thread_token_provider(num_threads);
|
|
ParallelFor(context, start, end, num_threads, [&](int i) {
|
|
const ScopedThreadToken scoped_thread_token(&thread_token_provider);
|
|
const int thread_id = scoped_thread_token.token();
|
|
function(thread_id, i);
|
|
});
|
|
}
|
|
|
|
|
|
} // namespace internal
|
|
} // namespace ceres
|
|
|
|
#endif // CERES_USE_TBB
|