mirror of
https://github.com/wjakob/tbb.git
synced 2026-08-31 01:20:42 +08:00
e32d75f876
The new release TBB is now under a new more
open license.
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
The list of most significant changes made over time in
Intel(R) Threading Building Blocks (Intel(R) TBB).
Intel TBB 2017
TBB_INTERFACE_VERSION == 9100
Changes (w.r.t. Intel TBB 4.4 Update 5):
- static_partitioner class is now a fully supported feature.
- async_node class is now a fully supported feature.
- Improved dynamic memory allocation replacement on Windows* OS to skip
DLLs for which replacement cannot be done, instead of aborting.
- Intel TBB no longer performs dynamic memory allocation replacement
for Microsoft* Visual Studio* 2008.
- For 64-bit platforms, quadrupled the worst-case limit on the amount
of memory the Intel TBB allocator can handle.
- Added TBB_USE_GLIBCXX_VERSION macro to specify the version of GNU
libstdc++ when it cannot be properly recognized, e.g. when used
with Clang on Linux* OS. Inspired by a contribution from David A.
- Added graph/stereo example to demostrate tbb::flow::async_msg.
- Removed a few cases of excessive user data copying in the flow graph.
- Reworked split_node to eliminate unnecessary overheads.
- Added support for C++11 move semantics to the argument of
tbb::parallel_do_feeder::add() method.
- Added C++11 move constructor and assignment operator to
tbb::combinable template class.
- Added tbb::this_task_arena::max_concurrency() function and
max_concurrency() method of class task_arena returning the maximal
number of threads that can work inside an arena.
- Deprecated tbb::task_arena::current_thread_index() static method;
use tbb::this_task_arena::current_thread_index() function instead.
- All examples for commercial version of library moved online:
https://software.intel.com/en-us/product-code-samples. Examples are
available as a standalone package or as a part of Intel(R) Parallel
Studio XE or Intel(R) System Studio Online Samples packages.
Changes affecting backward compatibility:
- Renamed following methods and types in async_node class:
Old New
async_gateway_type => gateway_type
async_gateway() => gateway()
async_try_put() => try_put()
async_reserve() => reserve_wait()
async_commit() => release_wait()
- Internal layout of some flow graph nodes has changed; recompilation
is recommended for all binaries that use the flow graph.
Preview Features:
- Added template class streaming_node to the flow graph API. It allows
a flow graph to offload computations to other devices through
streaming or offloading APIs.
- Template class opencl_node reimplemented as a specialization of
streaming_node that works with OpenCL*.
- Added tbb::this_task_arena::isolate() function to isolate execution
of a group of tasks or an algorithm from other tasks submitted
to the scheduler.
Bugs fixed:
- Added a workaround for GCC bug #62258 in std::rethrow_exception()
to prevent possible problems in case of exception propagation.
- Fixed parallel_scan to provide correct result if the initial value
of an accumulator is not the operation identity value.
- Fixed a memory corruption in the memory allocator when it meets
internal limits.
- Fixed the memory allocator on 64-bit platforms to align memory
to 16 bytes by default for all allocations bigger than 8 bytes.
- As a workaround for crashes in the Intel TBB library compiled with
GCC 6, added -flifetime-dse=1 to compilation options on Linux* OS.
- Fixed a race in the flow graph implementation.
Open-source contributions integrated:
- Enabling use of C++11 'override' keyword by Raf Schietekat.
------------------------------------------------------------------------
176 lines
5.1 KiB
C++
176 lines
5.1 KiB
C++
/*
|
|
Copyright (c) 2005-2016 Intel Corporation
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
http://www.apache.org/licenses/LICENSE-2.0
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
|
|
|
|
|
|
|
|
*/
|
|
|
|
// Support for GUI display for Polygon overlay demo
|
|
|
|
#define VIDEO_WINMAIN_ARGS
|
|
#include <iostream>
|
|
#include "polyover.h"
|
|
#include "polymain.h"
|
|
#include "pover_video.h"
|
|
#include "tbb/tick_count.h"
|
|
#include "tbb/task_scheduler_init.h"
|
|
#ifndef _WIN32
|
|
#include <sys/time.h>
|
|
#include <unistd.h>
|
|
|
|
void rt_sleep(int msec) {
|
|
usleep(msec*1000);
|
|
}
|
|
|
|
#else //_WIN32
|
|
|
|
#undef OLDUNIXTIME
|
|
#undef STDTIME
|
|
|
|
#include <windows.h>
|
|
|
|
void rt_sleep(int msec) {
|
|
Sleep(msec);
|
|
}
|
|
|
|
#endif /* _WIN32 */
|
|
|
|
using namespace std;
|
|
|
|
bool g_next_frame() {
|
|
if(++n_next_frame_calls >= frame_skips) { // the data race here is benign
|
|
n_next_frame_calls = 0;
|
|
return gVideo->next_frame();
|
|
}
|
|
return gVideo->running;
|
|
}
|
|
|
|
bool g_last_frame() {
|
|
if(n_next_frame_calls) return gVideo->next_frame();
|
|
return gVideo->running;
|
|
}
|
|
|
|
bool initializeVideo(int argc, char **argv) {
|
|
//pover_video *l_video = new pover_video();
|
|
//gVideo = l_video;
|
|
gVideo->init_console(); // don't check return code.
|
|
gVideo->title = g_windowTitle;
|
|
g_useGraphics = gVideo->init_window(g_xwinsize, g_ywinsize);
|
|
return true;
|
|
}
|
|
|
|
void pover_video::on_process() {
|
|
tbb::tick_count t0, t1;
|
|
double naiveParallelTime, domainSplitParallelTime;
|
|
// create map1 These could be done in parallel, if the pseudorandom number generator were re-seeded.
|
|
GenerateMap(&gPolymap1, gMapXSize, gMapYSize, gNPolygons, /*red*/255, /*green*/0, /*blue*/127);
|
|
// create map2
|
|
GenerateMap(&gPolymap2, gMapXSize, gMapYSize, gNPolygons, /*red*/0, /*green*/255, /*blue*/127);
|
|
//
|
|
// Draw source maps
|
|
gDrawXOffset = map1XLoc;
|
|
gDrawYOffset = map1YLoc;
|
|
for(int i=0; i < int(gPolymap1->size()); i++) {
|
|
(*gPolymap1)[i].drawPoly();
|
|
}
|
|
gDrawXOffset = map2XLoc;
|
|
gDrawYOffset = map2YLoc;
|
|
for(int i=0; i < int(gPolymap2->size()) ;i++) {
|
|
(*gPolymap2)[i].drawPoly();
|
|
}
|
|
gDoDraw = true;
|
|
|
|
// run serial map generation
|
|
gDrawXOffset = maprXLoc;
|
|
gDrawYOffset = maprYLoc;
|
|
{
|
|
RPolygon *xp = new RPolygon(0, 0, gMapXSize-1, gMapYSize-1, 0, 0, 0); // Clear the output space
|
|
delete xp;
|
|
t0 = tbb::tick_count::now();
|
|
SerialOverlayMaps(&gResultMap, gPolymap1, gPolymap2);
|
|
t1 = tbb::tick_count::now();
|
|
cout << "Serial overlay took " << (t1-t0).seconds()*1000 << " msec" << std::endl;
|
|
gSerialTime = (t1-t0).seconds()*1000;
|
|
#if _DEBUG
|
|
CheckPolygonMap(gResultMap);
|
|
// keep the map for comparison purposes.
|
|
#else
|
|
delete gResultMap;
|
|
#endif
|
|
if(gCsvFile.is_open()) {
|
|
gCsvFile << "Serial Time," << gSerialTime << std::endl;
|
|
gCsvFile << "Threads,";
|
|
if(gThreadsLow == THREADS_UNSET || gThreadsLow == tbb::task_scheduler_init::automatic) {
|
|
gCsvFile << "Threads,Automatic";
|
|
}
|
|
else {
|
|
for(int i=gThreadsLow; i <= gThreadsHigh; i++) {
|
|
gCsvFile << i;
|
|
if(i < gThreadsHigh) gCsvFile << ",";
|
|
}
|
|
}
|
|
gCsvFile << std::endl;
|
|
}
|
|
if(gIsGraphicalVersion) rt_sleep(2000);
|
|
}
|
|
// run naive parallel map generation
|
|
{
|
|
Polygon_map_t *resultMap;
|
|
if(gCsvFile.is_open()) {
|
|
gCsvFile << "Naive Time";
|
|
}
|
|
NaiveParallelOverlay(resultMap, *gPolymap1, *gPolymap2);
|
|
delete resultMap;
|
|
if(gIsGraphicalVersion) rt_sleep(2000);
|
|
}
|
|
// run split map generation
|
|
{
|
|
Polygon_map_t *resultMap;
|
|
if(gCsvFile.is_open()) {
|
|
gCsvFile << "Split Time";
|
|
}
|
|
SplitParallelOverlay(&resultMap, gPolymap1, gPolymap2);
|
|
delete resultMap;
|
|
if(gIsGraphicalVersion) rt_sleep(2000);
|
|
}
|
|
// split, accumulating into concurrent vector
|
|
{
|
|
concurrent_Polygon_map_t *cresultMap;
|
|
if(gCsvFile.is_open()) {
|
|
gCsvFile << "Split CV time";
|
|
}
|
|
SplitParallelOverlayCV(&cresultMap, gPolymap1, gPolymap2);
|
|
delete cresultMap;
|
|
if(gIsGraphicalVersion) rt_sleep(2000);
|
|
}
|
|
// split, accumulating into ETS
|
|
{
|
|
ETS_Polygon_map_t *cresultMap;
|
|
if(gCsvFile.is_open()) {
|
|
gCsvFile << "Split ETS time";
|
|
}
|
|
SplitParallelOverlayETS(&cresultMap, gPolymap1, gPolymap2);
|
|
delete cresultMap;
|
|
if(gIsGraphicalVersion) rt_sleep(2000);
|
|
}
|
|
if(gIsGraphicalVersion) rt_sleep(8000);
|
|
delete gPolymap1;
|
|
delete gPolymap2;
|
|
#if _DEBUG
|
|
delete gResultMap;
|
|
#endif
|
|
}
|