mirror of
https://github.com/wjakob/tbb.git
synced 2026-08-30 17:10:41 +08:00
e32d75f876
The new release TBB is now under a new more
open license.
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
The list of most significant changes made over time in
Intel(R) Threading Building Blocks (Intel(R) TBB).
Intel TBB 2017
TBB_INTERFACE_VERSION == 9100
Changes (w.r.t. Intel TBB 4.4 Update 5):
- static_partitioner class is now a fully supported feature.
- async_node class is now a fully supported feature.
- Improved dynamic memory allocation replacement on Windows* OS to skip
DLLs for which replacement cannot be done, instead of aborting.
- Intel TBB no longer performs dynamic memory allocation replacement
for Microsoft* Visual Studio* 2008.
- For 64-bit platforms, quadrupled the worst-case limit on the amount
of memory the Intel TBB allocator can handle.
- Added TBB_USE_GLIBCXX_VERSION macro to specify the version of GNU
libstdc++ when it cannot be properly recognized, e.g. when used
with Clang on Linux* OS. Inspired by a contribution from David A.
- Added graph/stereo example to demostrate tbb::flow::async_msg.
- Removed a few cases of excessive user data copying in the flow graph.
- Reworked split_node to eliminate unnecessary overheads.
- Added support for C++11 move semantics to the argument of
tbb::parallel_do_feeder::add() method.
- Added C++11 move constructor and assignment operator to
tbb::combinable template class.
- Added tbb::this_task_arena::max_concurrency() function and
max_concurrency() method of class task_arena returning the maximal
number of threads that can work inside an arena.
- Deprecated tbb::task_arena::current_thread_index() static method;
use tbb::this_task_arena::current_thread_index() function instead.
- All examples for commercial version of library moved online:
https://software.intel.com/en-us/product-code-samples. Examples are
available as a standalone package or as a part of Intel(R) Parallel
Studio XE or Intel(R) System Studio Online Samples packages.
Changes affecting backward compatibility:
- Renamed following methods and types in async_node class:
Old New
async_gateway_type => gateway_type
async_gateway() => gateway()
async_try_put() => try_put()
async_reserve() => reserve_wait()
async_commit() => release_wait()
- Internal layout of some flow graph nodes has changed; recompilation
is recommended for all binaries that use the flow graph.
Preview Features:
- Added template class streaming_node to the flow graph API. It allows
a flow graph to offload computations to other devices through
streaming or offloading APIs.
- Template class opencl_node reimplemented as a specialization of
streaming_node that works with OpenCL*.
- Added tbb::this_task_arena::isolate() function to isolate execution
of a group of tasks or an algorithm from other tasks submitted
to the scheduler.
Bugs fixed:
- Added a workaround for GCC bug #62258 in std::rethrow_exception()
to prevent possible problems in case of exception propagation.
- Fixed parallel_scan to provide correct result if the initial value
of an accumulator is not the operation identity value.
- Fixed a memory corruption in the memory allocator when it meets
internal limits.
- Fixed the memory allocator on 64-bit platforms to align memory
to 16 bytes by default for all allocations bigger than 8 bytes.
- As a workaround for crashes in the Intel TBB library compiled with
GCC 6, added -flifetime-dse=1 to compilation options on Linux* OS.
- Fixed a race in the flow graph implementation.
Open-source contributions integrated:
- Enabling use of C++11 'override' keyword by Raf Schietekat.
------------------------------------------------------------------------
125 lines
4.4 KiB
C++
125 lines
4.4 KiB
C++
/*
|
|
Copyright (c) 2005-2016 Intel Corporation
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
http://www.apache.org/licenses/LICENSE-2.0
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
|
|
|
|
|
|
|
|
*/
|
|
|
|
#include "primes.h"
|
|
#include <cstdlib>
|
|
#include <cstdio>
|
|
#include <cstring>
|
|
#include <cctype>
|
|
#include <utility>
|
|
#include <iostream>
|
|
#include <sstream>
|
|
#include "tbb/tick_count.h"
|
|
|
|
#include "../../common/utility/utility.h"
|
|
|
|
struct RunOptions{
|
|
//! NumberType of threads to use.
|
|
utility::thread_number_range threads;
|
|
//whether to suppress additional output
|
|
bool silentFlag;
|
|
//
|
|
NumberType n;
|
|
//! Grain size parameter
|
|
NumberType grainSize;
|
|
// number of time to repeat calculation
|
|
NumberType repeatNumber;
|
|
|
|
RunOptions(utility::thread_number_range threads, NumberType grainSize, NumberType n, bool silentFlag, NumberType repeatNumber)
|
|
: threads(threads), grainSize(grainSize), n(n), silentFlag(silentFlag), repeatNumber(repeatNumber)
|
|
{}
|
|
};
|
|
|
|
int do_get_default_num_threads() {
|
|
int threads;
|
|
#if __TBB_MIC_OFFLOAD
|
|
#pragma offload target(mic) out(threads)
|
|
#endif // __TBB_MIC_OFFLOAD
|
|
threads = tbb::task_scheduler_init::default_num_threads();
|
|
return threads;
|
|
}
|
|
|
|
int get_default_num_threads() {
|
|
static int threads = do_get_default_num_threads();
|
|
return threads;
|
|
}
|
|
|
|
//! Parse the command line.
|
|
static RunOptions ParseCommandLine( int argc, const char* argv[] ) {
|
|
utility::thread_number_range threads( get_default_num_threads, 0, get_default_num_threads() );
|
|
NumberType grainSize = 1000;
|
|
bool silent = false;
|
|
NumberType number = 100000000;
|
|
NumberType repeatNumber = 1;
|
|
|
|
utility::parse_cli_arguments(argc,argv,
|
|
utility::cli_argument_pack()
|
|
//"-h" option for displaying help is present implicitly
|
|
.positional_arg(threads,"n-of-threads",utility::thread_number_range_desc)
|
|
.positional_arg(number,"number","upper bound of range to search primes in, must be a positive integer")
|
|
.positional_arg(grainSize,"grain-size","must be a positive integer")
|
|
.positional_arg(repeatNumber,"n-of-repeats","repeat the calculation this number of times, must be a positive integer")
|
|
.arg(silent,"silent","no output except elapsed time")
|
|
);
|
|
|
|
RunOptions options(threads,grainSize, number, silent, repeatNumber);
|
|
return options;
|
|
}
|
|
|
|
int main( int argc, const char* argv[] ) {
|
|
tbb::tick_count mainBeginMark = tbb::tick_count::now();
|
|
RunOptions options =ParseCommandLine(argc,argv);
|
|
|
|
// Try different numbers of threads
|
|
for( int p=options.threads.first; p<=options.threads.last; p=options.threads.step(p) ) {
|
|
for (NumberType i=0; i<options.repeatNumber;++i){
|
|
tbb::tick_count iterationBeginMark = tbb::tick_count::now();
|
|
NumberType count = 0;
|
|
NumberType n = options.n;
|
|
if( p==0 ) {
|
|
#if __TBB_MIC_OFFLOAD
|
|
#pragma offload target(mic) in(n) out(count)
|
|
#endif // __TBB_MIC_OFFLOAD
|
|
count = SerialCountPrimes(n);
|
|
} else {
|
|
NumberType grainSize = options.grainSize;
|
|
#if __TBB_MIC_OFFLOAD
|
|
#pragma offload target(mic) in(n, p, grainSize) out(count)
|
|
#endif // __TBB_MIC_OFFLOAD
|
|
count = ParallelCountPrimes(n, p, grainSize);
|
|
}
|
|
tbb::tick_count iterationEndMark = tbb::tick_count::now();
|
|
if (!options.silentFlag){
|
|
std::cout
|
|
<<"#primes from [2.." <<options.n<<"] = " << count
|
|
<<" ("<<(iterationEndMark-iterationBeginMark).seconds()<< " sec with "
|
|
;
|
|
if( 0 != p )
|
|
std::cout<<p<<"-way parallelism";
|
|
else
|
|
std::cout<<"serial code";
|
|
std::cout<<")\n" ;
|
|
}
|
|
}
|
|
}
|
|
utility::report_elapsed_time((tbb::tick_count::now()-mainBeginMark).seconds());
|
|
return 0;
|
|
}
|