From d9f860fc563e5cf71d36301d7aba79a557d1b8fa Mon Sep 17 00:00:00 2001 From: andor <16273755+Andor233@users.noreply.github.com> Date: Thu, 30 May 2024 22:01:07 +0800 Subject: [PATCH 01/17] Fix #3783 : transform_inclusive_scan transform_exclusive_scan exclusive_scan inclusive_scan passes output values to the binary (reduce) operation with par policy --- stl/inc/execution | 149 ++++++++++++++---- .../test.cpp | 5 - .../test.cpp | 5 - .../test.cpp | 5 - .../test.cpp | 5 - 5 files changed, 114 insertions(+), 55 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index aef08d6da24..6c23a0d6c5f 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3619,8 +3619,9 @@ struct _Scan_decoupled_lookback { __std_execution_wake_by_address_all(&_State); } - template - void _Apply_exclusive_predecessor(_Ty& _Preceding, _FwdIt _First, const _FwdIt _Last, _BinOp _Reduce_op) { + template + void _Apply_exclusive_predecessor( + _Ty& _Preceding, _FwdIt _First, _FwdIt2 _First2, const _FwdIt _Last, _BinOp _Reduce_op) { // apply _Preceding to [_First, _Last) and _Sum._Ref(), using _Reduce_op _STD _Implicitly_construct_in_place_by_binary_op(_Sum._Ref(), _Reduce_op, _Preceding, _Local._Ref()); _State.store(_Local_available | _Sum_available); @@ -3628,19 +3629,26 @@ struct _Scan_decoupled_lookback { #pragma loop(ivdep) while (++_First != _Last) { - *_First = _Reduce_op(_Preceding, _STD move(*_First)); + ++_First2; + *_First = _Reduce_op(_Preceding, _STD move(*_First2)); } } - template - void _Apply_inclusive_predecessor(_Ty& _Preceding, _FwdIt _First, const _FwdIt _Last, _BinOp _Reduce_op) { + template + void _Apply_inclusive_predecessor( + _Ty& _Preceding, _FwdIt _First, _FwdIt2 _First2, const _FwdIt _Last, _BinOp _Reduce_op) { // apply _Preceding to [_First, _Last) and _Sum._Ref(), using _Reduce_op _STD _Implicitly_construct_in_place_by_binary_op(_Sum._Ref(), _Reduce_op, _Preceding, _Local._Ref()); _State.store(_Local_available | _Sum_available); #pragma loop(ivdep) for (; _First != _Last; ++_First) { - *_First = _Reduce_op(_Preceding, _STD move(*_First)); + if (_STD next(_First) == _Last) { + *_First = _Reduce_op(_Preceding, _Local._Ref()); + } else { + *_First = _Reduce_op(_Preceding, _STD move(*_First2)); + ++_First2; + } } } @@ -4342,7 +4350,8 @@ struct _No_init_tag { }; // tag to indicate that no initial value is to be used template -_FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val) { +_FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, + _STD vector<_Ty>& _Intermediate_result) { // local-sum for parallel exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and stores // successor sum in _Val // pre: _Val is *uninitialized* && _First != _Last @@ -4355,8 +4364,9 @@ _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ } _Ty _Tmp = _Reduce_op(_Val, *_First); // temp to enable _First == _Dest - *_Dest = _Val; - _Val = _STD move(_Tmp); + _Intermediate_result.emplace_back(_STD move(_Val)); + *_Dest = _Intermediate_result.back(); + _Val = _STD move(_Tmp); } } @@ -4420,17 +4430,20 @@ struct _Static_partitioned_exclusive_scan2 { return _Cancellation_status::_Running; } + _STD vector<_Ty> _Intermediate_result; + // Calculate local sum and publish to other threads - const auto _Last = - _STD _Exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref()); + const auto _Last = _STD _Exclusive_scan_per_chunk( + _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _Intermediate_result); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_exclusive_predecessor(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, begin(_Intermediate_result), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, begin(_Intermediate_result), _Last, _Reduce_op); } return _Cancellation_status::_Running; @@ -4485,7 +4498,33 @@ _FwdIt2 exclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _Last, _FwdI } template -_FwdIt2 _Inclusive_scan_per_chunk( +_FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, + _Ty_fwd&& _Predecessor, _STD vector<_Ty>& _Intermediate_result) { + // local-sum for parallel inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall sum in + // _Val. + // pre: _Val is *uninitialized* && _First != _Last + if constexpr (is_same_v<_No_init_tag, remove_const_t>>) { + _STD _Construct_in_place_by_deref(_Val, _First); + } else { + _STD _Implicitly_construct_in_place_by_binary_op_deref_rhs( + _Val, _Reduce_op, _STD forward<_Ty_fwd>(_Predecessor), _First); + } + + for (;;) { + *_Dest = _Val; + ++_Dest; + ++_First; + if (_First == _Last) { + return _Dest; + } + + _Intermediate_result.emplace_back(_STD move(_Val)); + _Val = _Reduce_op(_Intermediate_result.back(), *_First); + } +} + +template +_FwdIt2 _Inclusive_scan_per_chunk_complete( _FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, _Ty_fwd&& _Predecessor) { // local-sum for parallel inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall sum in // _Val. @@ -4536,7 +4575,7 @@ struct _Static_partitioned_inclusive_scan2 { // Run local inclusive_scan on this chunk const auto _Chunk = _Lookback.data() + static_cast(_Chunk_number); if (_Chunk_number == 0) { // chunk 0 is special as it has no predecessor; its local and total sums are the same - _STD _Inclusive_scan_per_chunk( + _STD _Inclusive_scan_per_chunk_complete( _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Sum._Ref(), _STD move(_Initial)); _Chunk->_Store_available_state(_Sum_available); return _Cancellation_status::_Running; @@ -4545,23 +4584,27 @@ struct _Static_partitioned_inclusive_scan2 { const auto _Prev_chunk = _STD _Prev_iter(_Chunk); if (_Prev_chunk->_State.load() & _Sum_available) { // if predecessor sum already complete, we can incorporate its value directly for 1 pass - _STD _Inclusive_scan_per_chunk( + _STD _Inclusive_scan_per_chunk_complete( _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Sum._Ref(), _Prev_chunk->_Sum._Ref()); _Chunk->_Store_available_state(_Sum_available); return _Cancellation_status::_Running; } + // Make a vector to avoid the type of *_Dest is different with _Ty + _STD vector<_Ty> _Intermediate_result; + // Calculate local sum and publish to other threads - const auto _Last = _STD _Inclusive_scan_per_chunk( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _No_init_tag{}); + const auto _Last = _STD _Inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, + _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_inclusive_predecessor(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Last, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); } return _Cancellation_status::_Running; @@ -4661,8 +4704,8 @@ _FwdIt2 inclusive_scan(_ExPo&&, _FwdIt1 _First, _FwdIt1 _Last, _FwdIt2 _Dest, _B } template -_FwdIt2 _Transform_exclusive_scan_per_chunk( - _FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _UnaryOp _Transform_op, _Ty& _Val) { +_FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, + _UnaryOp _Transform_op, _Ty& _Val, _STD vector<_Ty>& _Intermediate_result) { // Local-sum for parallel transform_exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and // stores successor sum in _Val. // pre: _Val is *uninitialized* && _First != _Last @@ -4675,8 +4718,9 @@ _FwdIt2 _Transform_exclusive_scan_per_chunk( } _Ty _Tmp = _Reduce_op(_Val, _Transform_op(*_First)); // temp to enable _First == _Dest - *_Dest = _Val; - _Val = _STD move(_Tmp); + _Intermediate_result.emplace_back(_STD move(_Val)); + *_Dest = _Intermediate_result.back(); + _Val = _STD move(_Tmp); } } @@ -4721,6 +4765,9 @@ struct _Static_partitioned_transform_exclusive_scan2 { return _Cancellation_status::_Canceled; } + // Make a vector to avoid the type of *_Dest is different with _Ty + _STD vector<_Ty> _Intermediate_result; + const auto _Chunk_number = _Key._Chunk_number; const auto _In_range = _Basis1._Get_chunk(_Key); const auto _Dest = _Basis2._Get_first(_Chunk_number, _Team._Get_chunk_offset(_Chunk_number)); @@ -4743,16 +4790,17 @@ struct _Static_partitioned_transform_exclusive_scan2 { } // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_exclusive_scan_per_chunk( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Local._Ref()); + const auto _Last = _STD _Transform_exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _Intermediate_result); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_exclusive_predecessor(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, begin(_Intermediate_result), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, begin(_Intermediate_result), _Last, _Reduce_op); } return _Cancellation_status::_Running; @@ -4808,7 +4856,34 @@ _FwdIt2 transform_exclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _L template _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, - _UnaryOp _Transform_op, _Ty& _Val, _Ty_fwd&& _Predecessor) { + _UnaryOp _Transform_op, _Ty& _Val, _Ty_fwd&& _Predecessor, _STD vector<_Ty>& _Intermediate_result) { + // local-sum for parallel transform_inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall + // sum in _Val + // pre: _Val is *uninitialized* && _First != _Last + if constexpr (is_same_v<_No_init_tag, remove_const_t>>) { + _STD _Construct_in_place_by_transform_deref(_Val, _Transform_op, _First); + } else { + _STD _Implicitly_construct_in_place_by_binary_op_transform_deref_rhs( + _Val, _Reduce_op, _Transform_op, _STD forward<_Ty_fwd>(_Predecessor), _First); + } + + for (;;) { + *_Dest = _Val; + ++_Dest; + ++_First; + if (_First == _Last) { + // The Last value is stored in the _Val + return _Dest; + } + + _Intermediate_result.emplace_back(_STD move(_Val)); + _Val = _Reduce_op(_Intermediate_result.back(), _Transform_op(*_First)); + } +} + +template +_FwdIt2 _Transform_inclusive_scan_per_chunk_complete(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, + _BinOp _Reduce_op, _UnaryOp _Transform_op, _Ty& _Val, _Ty_fwd&& _Predecessor) { // local-sum for parallel transform_inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall // sum in _Val // pre: _Val is *uninitialized* && _First != _Last @@ -4859,7 +4934,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { // Run local transform_inclusive_scan on this chunk const auto _Chunk = _Lookback.data() + static_cast(_Chunk_number); if (_Chunk_number == 0) { // chunk 0 is special as it has no predecessor; its local and total sums are the same - _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, + _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Sum._Ref(), _STD move(_Initial)); _Chunk->_Store_available_state(_Sum_available); return _Cancellation_status::_Running; @@ -4868,23 +4943,27 @@ struct _Static_partitioned_transform_inclusive_scan2 { const auto _Prev_chunk = _STD _Prev_iter(_Chunk); if (_Prev_chunk->_State.load() & _Sum_available) { // if predecessor sum already complete, we can incorporate its value directly for 1 pass - _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, + _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Sum._Ref(), _Prev_chunk->_Sum._Ref()); _Chunk->_Store_available_state(_Sum_available); return _Cancellation_status::_Running; } + // Make a vector to avoid the type of *_Dest is different with _Ty + _STD vector<_Ty> _Intermediate_result; + // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_inclusive_scan_per_chunk( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}); + const auto _Last = _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_inclusive_predecessor(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, begin(_Intermediate_result), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Last, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, begin(_Intermediate_result), _Last, _Reduce_op); } return _Cancellation_status::_Running; diff --git a/tests/std/tests/P0024R2_parallel_algorithms_exclusive_scan/test.cpp b/tests/std/tests/P0024R2_parallel_algorithms_exclusive_scan/test.cpp index 5a00a8d07b9..36a08e7b56b 100644 --- a/tests/std/tests/P0024R2_parallel_algorithms_exclusive_scan/test.cpp +++ b/tests/std/tests/P0024R2_parallel_algorithms_exclusive_scan/test.cpp @@ -184,11 +184,6 @@ struct typesBop { bopResult operator()(intermediateType&&, intermediateType&&) { return 0; } - - // *result = binary_op(tmp, move(*result)) - bopResult operator()(intermediateType&, outputType&&) { - return 0; - } }; void test_case_exclusive_scan_init_writes_intermediate_type() { diff --git a/tests/std/tests/P0024R2_parallel_algorithms_inclusive_scan/test.cpp b/tests/std/tests/P0024R2_parallel_algorithms_inclusive_scan/test.cpp index a3ecee5b3bd..517a003a200 100644 --- a/tests/std/tests/P0024R2_parallel_algorithms_inclusive_scan/test.cpp +++ b/tests/std/tests/P0024R2_parallel_algorithms_inclusive_scan/test.cpp @@ -193,11 +193,6 @@ struct typesBop { bopResult operator()(intermediateType&&, intermediateType&&) { return 0; } - - // *result = binary_op(tmp, move(*result)) - bopResult operator()(intermediateType&, outputType&&) { - return 0; - } }; void test_case_inclusive_scan_init_writes_intermediate_type() { diff --git a/tests/std/tests/P0024R2_parallel_algorithms_transform_exclusive_scan/test.cpp b/tests/std/tests/P0024R2_parallel_algorithms_transform_exclusive_scan/test.cpp index 7cb3d6f11d6..12b8440dd48 100644 --- a/tests/std/tests/P0024R2_parallel_algorithms_transform_exclusive_scan/test.cpp +++ b/tests/std/tests/P0024R2_parallel_algorithms_transform_exclusive_scan/test.cpp @@ -179,11 +179,6 @@ struct typesBop { bopResult operator()(intermediateType&&, intermediateType&&) { return 0; } - - // *result = binary_op(tmp, move(*result)) - bopResult operator()(intermediateType&, outputType&&) { - return 0; - } }; void test_case_transform_exclusive_scan_init_writes_intermediate_type() { diff --git a/tests/std/tests/P0024R2_parallel_algorithms_transform_inclusive_scan/test.cpp b/tests/std/tests/P0024R2_parallel_algorithms_transform_inclusive_scan/test.cpp index c54e01aa69d..b1aa6eaec09 100644 --- a/tests/std/tests/P0024R2_parallel_algorithms_transform_inclusive_scan/test.cpp +++ b/tests/std/tests/P0024R2_parallel_algorithms_transform_inclusive_scan/test.cpp @@ -203,11 +203,6 @@ struct typesBop { bopResult operator()(intermediateType&&, intermediateType&&) { return 0; } - - // *result = binary_op(tmp, move(*result)) - bopResult operator()(intermediateType&, outputType&&) { - return 0; - } }; void test_case_transform_inclusive_scan_init_writes_intermediate_type() { From 851efc1ba0a4b39fc685754aa4f4a29e7a663560 Mon Sep 17 00:00:00 2001 From: andor <16273755+Andor233@users.noreply.github.com> Date: Thu, 30 May 2024 23:13:28 +0800 Subject: [PATCH 02/17] Add _STD to defend ADL --- stl/inc/execution | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 6c23a0d6c5f..5aa3739ff57 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -4440,10 +4440,10 @@ struct _Static_partitioned_exclusive_scan2 { // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly _Chunk->_Apply_exclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, begin(_Intermediate_result), _Last, _Reduce_op); + _Prev_chunk->_Sum._Ref(), _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, begin(_Intermediate_result), _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); } return _Cancellation_status::_Running; @@ -4797,10 +4797,10 @@ struct _Static_partitioned_transform_exclusive_scan2 { // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly _Chunk->_Apply_exclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, begin(_Intermediate_result), _Last, _Reduce_op); + _Prev_chunk->_Sum._Ref(), _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, begin(_Intermediate_result), _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); } return _Cancellation_status::_Running; @@ -4960,10 +4960,10 @@ struct _Static_partitioned_transform_inclusive_scan2 { // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly _Chunk->_Apply_inclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, begin(_Intermediate_result), _Last, _Reduce_op); + _Prev_chunk->_Sum._Ref(), _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, begin(_Intermediate_result), _Last, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); } return _Cancellation_status::_Running; From a5ac91cc047773bdf84ae8924a4f5a22e5e2abc6 Mon Sep 17 00:00:00 2001 From: andor <16273755+Andor233@users.noreply.github.com> Date: Sat, 1 Jun 2024 03:26:04 +0800 Subject: [PATCH 03/17] 1. Add _STD to defend ADL, towards #140, 2. Use the _Intermediate_result Only when the intermediate type is different with output type --- stl/inc/execution | 232 +++++++++++++++++++++++++++++++++++----------- stl/inc/vector | 18 ++-- stl/inc/xmemory | 6 +- 3 files changed, 190 insertions(+), 66 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 5aa3739ff57..49bb59141ae 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3634,6 +3634,19 @@ struct _Scan_decoupled_lookback { } } + template + void _Apply_exclusive_predecessor_origin(_Ty& _Preceding, _FwdIt _First, const _FwdIt _Last, _BinOp _Reduce_op) { + // apply _Preceding to [_First, _Last) and _Sum._Ref(), using _Reduce_op + _STD _Implicitly_construct_in_place_by_binary_op(_Sum._Ref(), _Reduce_op, _Preceding, _Local._Ref()); + _State.store(_Local_available | _Sum_available); + *_First = _Preceding; + +#pragma loop(ivdep) + while (++_First != _Last) { + *_First = _Reduce_op(_Preceding, _STD move(*_First)); + } + } + template void _Apply_inclusive_predecessor( _Ty& _Preceding, _FwdIt _First, _FwdIt2 _First2, const _FwdIt _Last, _BinOp _Reduce_op) { @@ -3652,6 +3665,18 @@ struct _Scan_decoupled_lookback { } } + template + void _Apply_inclusive_predecessor_origin(_Ty& _Preceding, _FwdIt _First, const _FwdIt _Last, _BinOp _Reduce_op) { + // apply _Preceding to [_First, _Last) and _Sum._Ref(), using _Reduce_op + _STD _Implicitly_construct_in_place_by_binary_op(_Sum._Ref(), _Reduce_op, _Preceding, _Local._Ref()); + _State.store(_Local_available | _Sum_available); + +#pragma loop(ivdep) + for (; _First != _Last; ++_First) { + *_First = _Reduce_op(_Preceding, _STD move(*_First)); + } + } + ~_Scan_decoupled_lookback() { const auto _State_bits = _State.load(memory_order_relaxed); if (_State_bits & _Sum_available) { @@ -4370,6 +4395,25 @@ _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ } } +template +_FwdIt2 _Exclusive_scan_per_chunk_origin(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val) { + // local-sum for parallel exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and stores + // successor sum in _Val + // pre: _Val is *uninitialized* && _First != _Last + _STD _Construct_in_place(_Val, *_First); + for (;;) { + ++_First; + ++_Dest; + if (_First == _Last) { + return _Dest; + } + + _Ty _Tmp = _Reduce_op(_Val, *_First); // temp to enable _First == _Dest + *_Dest = _Val; + _Val = _STD move(_Tmp); + } +} + template void _Exclusive_scan_per_chunk_complete( _FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, _Ty& _Init) { @@ -4430,20 +4474,36 @@ struct _Static_partitioned_exclusive_scan2 { return _Cancellation_status::_Running; } - _STD vector<_Ty> _Intermediate_result; - - // Calculate local sum and publish to other threads - const auto _Last = _STD _Exclusive_scan_per_chunk( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _Intermediate_result); - _Chunk->_Store_available_state(_Local_available); + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { + // Calculate local sum and publish to other threads + const auto _Last = _STD _Exclusive_scan_per_chunk_origin( + _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref()); + _Chunk->_Store_available_state(_Local_available); - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_exclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_exclusive_predecessor_origin(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); + } } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); + _STD vector<_Ty> _Intermediate_result; + + // Calculate local sum and publish to other threads + const auto _Last = _STD _Exclusive_scan_per_chunk( + _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _Intermediate_result); + _Chunk->_Store_available_state(_Local_available); + + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_exclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } } return _Cancellation_status::_Running; @@ -4590,23 +4650,38 @@ struct _Static_partitioned_inclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to avoid the type of *_Dest is different with _Ty - _STD vector<_Ty> _Intermediate_result; - - // Calculate local sum and publish to other threads - const auto _Last = _STD _Inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, - _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); - _Chunk->_Store_available_state(_Local_available); + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { + // Calculate local sum and publish to other threads + const auto _Last = _STD _Inclusive_scan_per_chunk_complete( + _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _No_init_tag{}); + _Chunk->_Store_available_state(_Local_available); - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_inclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_inclusive_predecessor_origin(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); + } } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); + // Make a vector to avoid the type of *_Dest is different with _Ty + _STD vector<_Ty> _Intermediate_result; + + // Calculate local sum and publish to other threads + const auto _Last = _STD _Inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, + _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); + _Chunk->_Store_available_state(_Local_available); + + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_inclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } } - return _Cancellation_status::_Running; } @@ -4724,6 +4799,25 @@ _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, } } +template +_FwdIt2 _Transform_exclusive_scan_per_chunk_origin(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _UnaryOp _Transform_op, _Ty& _Val) { + // Local-sum for parallel transform_exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and + // stores successor sum in _Val. + // pre: _Val is *uninitialized* && _First != _Last + _STD _Construct_in_place_by_transform_deref(_Val, _Transform_op, _First); + for (;;) { + ++_First; + ++_Dest; + if (_First == _Last) { + return _Dest; + } + + _Ty _Tmp = _Reduce_op(_Val, _Transform_op(*_First)); // temp to enable _First == _Dest + *_Dest = _Val; + _Val = _STD move(_Tmp); + } +} + template void _Transform_exclusive_scan_per_chunk_complete(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _UnaryOp _Transform_op, _Ty& _Val, _Ty& _Init) { @@ -4765,9 +4859,6 @@ struct _Static_partitioned_transform_exclusive_scan2 { return _Cancellation_status::_Canceled; } - // Make a vector to avoid the type of *_Dest is different with _Ty - _STD vector<_Ty> _Intermediate_result; - const auto _Chunk_number = _Key._Chunk_number; const auto _In_range = _Basis1._Get_chunk(_Key); const auto _Dest = _Basis2._Get_first(_Chunk_number, _Team._Get_chunk_offset(_Chunk_number)); @@ -4789,20 +4880,38 @@ struct _Static_partitioned_transform_exclusive_scan2 { return _Cancellation_status::_Running; } - // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, - _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _Intermediate_result); - _Chunk->_Store_available_state(_Local_available); + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { + // Calculate local sum and publish to other threads + const auto _Last = _STD _Transform_exclusive_scan_per_chunk_origin( + _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Local._Ref()); + _Chunk->_Store_available_state(_Local_available); - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_exclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_exclusive_predecessor_origin(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); + } } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); + // Make a vector to avoid the type of *_Dest is different with _Ty + _STD vector<_Ty> _Intermediate_result; + + // Calculate local sum and publish to other threads + const auto _Last = _STD _Transform_exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _Intermediate_result); + _Chunk->_Store_available_state(_Local_available); + + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_exclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } } - return _Cancellation_status::_Running; } @@ -4949,23 +5058,38 @@ struct _Static_partitioned_transform_inclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to avoid the type of *_Dest is different with _Ty - _STD vector<_Ty> _Intermediate_result; + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { + // Calculate local sum and publish to other threads + const auto _Last = _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, _Dest, + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}); + _Chunk->_Store_available_state(_Local_available); - // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, - _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); - _Chunk->_Store_available_state(_Local_available); - - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_inclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_inclusive_predecessor_origin(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); + } } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _STD begin(_Intermediate_result), _Last, _Reduce_op); + // Make a vector to avoid the type of *_Dest is different with _Ty + _STD vector<_Ty> _Intermediate_result; + + // Calculate local sum and publish to other threads + const auto _Last = _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); + _Chunk->_Store_available_state(_Local_available); + + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_inclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } } - return _Cancellation_status::_Running; } diff --git a/stl/inc/vector b/stl/inc/vector index e2d7e0cc600..05f10ee9a74 100644 --- a/stl/inc/vector +++ b/stl/inc/vector @@ -788,7 +788,7 @@ private: if constexpr (conjunction_v, _Uses_default_construct<_Alloc, _Ty*, _Valty...>>) { _ASAN_VECTOR_MODIFY(1); - _Construct_in_place(*_Mylast, _STD forward<_Valty>(_Val)...); + _STD _Construct_in_place(*_Mylast, _STD forward<_Valty>(_Val)...); } else { _ASAN_VECTOR_EXTEND_GUARD(static_cast(_Mylast - _My_data._Myfirst) + 1); _Alty_traits::construct(_Getal(), _Unfancy(_Mylast), _STD forward<_Valty>(_Val)...); @@ -822,27 +822,27 @@ private: const size_type _Newsize = _Oldsize + 1; size_type _Newcapacity = _Calculate_growth(_Newsize); - const pointer _Newvec = _Allocate_at_least_helper(_Al, _Newcapacity); + const pointer _Newvec = _STD _Allocate_at_least_helper(_Al, _Newcapacity); const pointer _Constructed_last = _Newvec + _Whereoff + 1; pointer _Constructed_first = _Constructed_last; _TRY_BEGIN - _Alty_traits::construct(_Al, _Unfancy(_Newvec + _Whereoff), _STD forward<_Valty>(_Val)...); + _Alty_traits::construct(_Al, _STD _Unfancy(_Newvec + _Whereoff), _STD forward<_Valty>(_Val)...); _Constructed_first = _Newvec + _Whereoff; if (_Whereptr == _Mylast) { // at back, provide strong guarantee if constexpr (is_nothrow_move_constructible_v<_Ty> || !is_copy_constructible_v<_Ty>) { - _Uninitialized_move(_Myfirst, _Mylast, _Newvec, _Al); + _STD _Uninitialized_move(_Myfirst, _Mylast, _Newvec, _Al); } else { - _Uninitialized_copy(_Myfirst, _Mylast, _Newvec, _Al); + _STD _Uninitialized_copy(_Myfirst, _Mylast, _Newvec, _Al); } } else { // provide basic guarantee - _Uninitialized_move(_Myfirst, _Whereptr, _Newvec, _Al); + _STD _Uninitialized_move(_Myfirst, _Whereptr, _Newvec, _Al); _Constructed_first = _Newvec; - _Uninitialized_move(_Whereptr, _Mylast, _Newvec + _Whereoff + 1, _Al); + _STD _Uninitialized_move(_Whereptr, _Mylast, _Newvec + _Whereoff + 1, _Al); } _CATCH_ALL - _Destroy_range(_Constructed_first, _Constructed_last, _Al); + _STD _Destroy_range(_Constructed_first, _Constructed_last, _Al); _Al.deallocate(_Newvec, _Newcapacity); _RERAISE; _CATCH_END @@ -2024,7 +2024,7 @@ private: _My_data._Orphan_all(); if (_Myfirst) { // destroy and deallocate old array - _Destroy_range(_Myfirst, _Mylast, _Al); + _STD _Destroy_range(_Myfirst, _Mylast, _Al); _ASAN_VECTOR_REMOVE; _Al.deallocate(_Myfirst, static_cast(_Myend - _Myfirst)); } diff --git a/stl/inc/xmemory b/stl/inc/xmemory index 011c0722780..45913c396c3 100644 --- a/stl/inc/xmemory +++ b/stl/inc/xmemory @@ -1905,15 +1905,15 @@ _CONSTEXPR20 _Alloc_ptr_t<_Alloc> _Uninitialized_move( // move [_First, _Last) to raw _Dest, using _Al // note: only called internally from elsewhere in the STL using _Ptrval = typename _Alloc::value_type*; - auto _UFirst = _Get_unwrapped(_First); - const auto _ULast = _Get_unwrapped(_Last); + auto _UFirst = _STD _Get_unwrapped(_First); + const auto _ULast = _STD _Get_unwrapped(_Last); if constexpr (conjunction_v::_Bitcopy_constructible>, _Uses_default_construct<_Alloc, _Ptrval, decltype(_STD move(*_UFirst))>>) { #if _HAS_CXX20 if (!_STD is_constant_evaluated()) #endif // _HAS_CXX20 { - _Copy_memmove(_UFirst, _ULast, _Unfancy(_Dest)); + _STD _Copy_memmove(_UFirst, _ULast, _STD _Unfancy(_Dest)); return _Dest + (_ULast - _UFirst); } } From f253732eddec3ab53bb7d19bf5b687523042fbc6 Mon Sep 17 00:00:00 2001 From: andor <16273755+Andor233@users.noreply.github.com> Date: Sat, 1 Jun 2024 03:34:44 +0800 Subject: [PATCH 04/17] Format code --- stl/inc/execution | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 49bb59141ae..933b1d4cbfd 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -4396,7 +4396,8 @@ _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ } template -_FwdIt2 _Exclusive_scan_per_chunk_origin(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val) { +_FwdIt2 _Exclusive_scan_per_chunk_origin( + _FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val) { // local-sum for parallel exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and stores // successor sum in _Val // pre: _Val is *uninitialized* && _First != _Last @@ -4800,7 +4801,8 @@ _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, } template -_FwdIt2 _Transform_exclusive_scan_per_chunk_origin(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _UnaryOp _Transform_op, _Ty& _Val) { +_FwdIt2 _Transform_exclusive_scan_per_chunk_origin( + _FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _UnaryOp _Transform_op, _Ty& _Val) { // Local-sum for parallel transform_exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and // stores successor sum in _Val. // pre: _Val is *uninitialized* && _First != _Last @@ -5061,8 +5063,8 @@ struct _Static_partitioned_transform_inclusive_scan2 { // If the intermediate type is same with output type, then we can direct store the bop result in the output if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, _Dest, - _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}); + const auto _Last = _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, + _Dest, _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements From ad5b30329ae4a8eb3bfd594fc37542c83fd2b8f3 Mon Sep 17 00:00:00 2001 From: andor <16273755+Andor233@users.noreply.github.com> Date: Sat, 1 Jun 2024 18:41:33 +0800 Subject: [PATCH 05/17] Remove _STD before the typename --- stl/inc/execution | 24 ++++++++++++------------ 1 file changed, 12 insertions(+), 12 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 933b1d4cbfd..3faccfb687a 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -4376,7 +4376,7 @@ struct _No_init_tag { template _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, - _STD vector<_Ty>& _Intermediate_result) { + vector<_Ty>& _Intermediate_result) { // local-sum for parallel exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and stores // successor sum in _Val // pre: _Val is *uninitialized* && _First != _Last @@ -4476,7 +4476,7 @@ struct _Static_partitioned_exclusive_scan2 { } // If the intermediate type is same with output type, then we can direct store the bop result in the output - if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { + if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { // Calculate local sum and publish to other threads const auto _Last = _STD _Exclusive_scan_per_chunk_origin( _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref()); @@ -4490,7 +4490,7 @@ struct _Static_partitioned_exclusive_scan2 { _Chunk->_Apply_exclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); } } else { - _STD vector<_Ty> _Intermediate_result; + vector<_Ty> _Intermediate_result; // Calculate local sum and publish to other threads const auto _Last = _STD _Exclusive_scan_per_chunk( @@ -4560,7 +4560,7 @@ _FwdIt2 exclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _Last, _FwdI template _FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, - _Ty_fwd&& _Predecessor, _STD vector<_Ty>& _Intermediate_result) { + _Ty_fwd&& _Predecessor, vector<_Ty>& _Intermediate_result) { // local-sum for parallel inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall sum in // _Val. // pre: _Val is *uninitialized* && _First != _Last @@ -4652,7 +4652,7 @@ struct _Static_partitioned_inclusive_scan2 { } // If the intermediate type is same with output type, then we can direct store the bop result in the output - if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { + if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { // Calculate local sum and publish to other threads const auto _Last = _STD _Inclusive_scan_per_chunk_complete( _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _No_init_tag{}); @@ -4667,7 +4667,7 @@ struct _Static_partitioned_inclusive_scan2 { } } else { // Make a vector to avoid the type of *_Dest is different with _Ty - _STD vector<_Ty> _Intermediate_result; + vector<_Ty> _Intermediate_result; // Calculate local sum and publish to other threads const auto _Last = _STD _Inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, @@ -4781,7 +4781,7 @@ _FwdIt2 inclusive_scan(_ExPo&&, _FwdIt1 _First, _FwdIt1 _Last, _FwdIt2 _Dest, _B template _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, - _UnaryOp _Transform_op, _Ty& _Val, _STD vector<_Ty>& _Intermediate_result) { + _UnaryOp _Transform_op, _Ty& _Val, vector<_Ty>& _Intermediate_result) { // Local-sum for parallel transform_exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and // stores successor sum in _Val. // pre: _Val is *uninitialized* && _First != _Last @@ -4883,7 +4883,7 @@ struct _Static_partitioned_transform_exclusive_scan2 { } // If the intermediate type is same with output type, then we can direct store the bop result in the output - if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { + if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { // Calculate local sum and publish to other threads const auto _Last = _STD _Transform_exclusive_scan_per_chunk_origin( _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Local._Ref()); @@ -4898,7 +4898,7 @@ struct _Static_partitioned_transform_exclusive_scan2 { } } else { // Make a vector to avoid the type of *_Dest is different with _Ty - _STD vector<_Ty> _Intermediate_result; + vector<_Ty> _Intermediate_result; // Calculate local sum and publish to other threads const auto _Last = _STD _Transform_exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, @@ -4967,7 +4967,7 @@ _FwdIt2 transform_exclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _L template _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, - _UnaryOp _Transform_op, _Ty& _Val, _Ty_fwd&& _Predecessor, _STD vector<_Ty>& _Intermediate_result) { + _UnaryOp _Transform_op, _Ty& _Val, _Ty_fwd&& _Predecessor, vector<_Ty>& _Intermediate_result) { // local-sum for parallel transform_inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall // sum in _Val // pre: _Val is *uninitialized* && _First != _Last @@ -5061,7 +5061,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { } // If the intermediate type is same with output type, then we can direct store the bop result in the output - if constexpr (_STD is_same_v<_Ty, typename std::iterator_traits<_FwdIt2>::value_type>) { + if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { // Calculate local sum and publish to other threads const auto _Last = _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}); @@ -5076,7 +5076,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { } } else { // Make a vector to avoid the type of *_Dest is different with _Ty - _STD vector<_Ty> _Intermediate_result; + vector<_Ty> _Intermediate_result; // Calculate local sum and publish to other threads const auto _Last = _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, From d4fe3c2606ae55ef3a1a58e59bdb33798d6c08b3 Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 04:32:06 -0700 Subject: [PATCH 06/17] Use emplace_back's returned reference. --- stl/inc/execution | 16 ++++++---------- 1 file changed, 6 insertions(+), 10 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 3faccfb687a..8500adf0ae5 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -4389,9 +4389,8 @@ _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ } _Ty _Tmp = _Reduce_op(_Val, *_First); // temp to enable _First == _Dest - _Intermediate_result.emplace_back(_STD move(_Val)); - *_Dest = _Intermediate_result.back(); - _Val = _STD move(_Tmp); + *_Dest = _Intermediate_result.emplace_back(_STD move(_Val)); + _Val = _STD move(_Tmp); } } @@ -4579,8 +4578,7 @@ _FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ return _Dest; } - _Intermediate_result.emplace_back(_STD move(_Val)); - _Val = _Reduce_op(_Intermediate_result.back(), *_First); + _Val = _Reduce_op(_Intermediate_result.emplace_back(_STD move(_Val)), *_First); } } @@ -4794,9 +4792,8 @@ _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, } _Ty _Tmp = _Reduce_op(_Val, _Transform_op(*_First)); // temp to enable _First == _Dest - _Intermediate_result.emplace_back(_STD move(_Val)); - *_Dest = _Intermediate_result.back(); - _Val = _STD move(_Tmp); + *_Dest = _Intermediate_result.emplace_back(_STD move(_Val)); + _Val = _STD move(_Tmp); } } @@ -4987,8 +4984,7 @@ _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, return _Dest; } - _Intermediate_result.emplace_back(_STD move(_Val)); - _Val = _Reduce_op(_Intermediate_result.back(), _Transform_op(*_First)); + _Val = _Reduce_op(_Intermediate_result.emplace_back(_STD move(_Val)), _Transform_op(*_First)); } } From 4c69b75669231b8479cb852bebb17d6548fb54e1 Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 06:46:33 -0700 Subject: [PATCH 07/17] `_FwdIt2 _First2` => `_Ty* _First2`; it's always `_Intermediate_result.data()`. --- stl/inc/execution | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 8500adf0ae5..52ece4cd7ac 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3619,9 +3619,9 @@ struct _Scan_decoupled_lookback { __std_execution_wake_by_address_all(&_State); } - template + template void _Apply_exclusive_predecessor( - _Ty& _Preceding, _FwdIt _First, _FwdIt2 _First2, const _FwdIt _Last, _BinOp _Reduce_op) { + _Ty& _Preceding, _FwdIt _First, _Ty* _First2, const _FwdIt _Last, _BinOp _Reduce_op) { // apply _Preceding to [_First, _Last) and _Sum._Ref(), using _Reduce_op _STD _Implicitly_construct_in_place_by_binary_op(_Sum._Ref(), _Reduce_op, _Preceding, _Local._Ref()); _State.store(_Local_available | _Sum_available); @@ -3647,9 +3647,9 @@ struct _Scan_decoupled_lookback { } } - template + template void _Apply_inclusive_predecessor( - _Ty& _Preceding, _FwdIt _First, _FwdIt2 _First2, const _FwdIt _Last, _BinOp _Reduce_op) { + _Ty& _Preceding, _FwdIt _First, _Ty* _First2, const _FwdIt _Last, _BinOp _Reduce_op) { // apply _Preceding to [_First, _Last) and _Sum._Ref(), using _Reduce_op _STD _Implicitly_construct_in_place_by_binary_op(_Sum._Ref(), _Reduce_op, _Preceding, _Local._Ref()); _State.store(_Local_available | _Sum_available); From 640c67e58a6488d33002d1f693a7c6c864271880 Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 08:18:52 -0700 Subject: [PATCH 08/17] Drop unnecessary comment in `_Transform_inclusive_scan_per_chunk`. --- stl/inc/execution | 1 - 1 file changed, 1 deletion(-) diff --git a/stl/inc/execution b/stl/inc/execution index 52ece4cd7ac..aa26bddbd5b 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -4980,7 +4980,6 @@ _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, ++_Dest; ++_First; if (_First == _Last) { - // The Last value is stored in the _Val return _Dest; } From 7c6aa0177180db470417b23993bf2702a3aed675 Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 07:27:38 -0700 Subject: [PATCH 09/17] Reduce code duplication. --- stl/inc/execution | 352 ++++++++++++++-------------------------------- 1 file changed, 103 insertions(+), 249 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index aa26bddbd5b..5d1d5739f77 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3629,21 +3629,13 @@ struct _Scan_decoupled_lookback { #pragma loop(ivdep) while (++_First != _Last) { - ++_First2; - *_First = _Reduce_op(_Preceding, _STD move(*_First2)); - } - } - - template - void _Apply_exclusive_predecessor_origin(_Ty& _Preceding, _FwdIt _First, const _FwdIt _Last, _BinOp _Reduce_op) { - // apply _Preceding to [_First, _Last) and _Sum._Ref(), using _Reduce_op - _STD _Implicitly_construct_in_place_by_binary_op(_Sum._Ref(), _Reduce_op, _Preceding, _Local._Ref()); - _State.store(_Local_available | _Sum_available); - *_First = _Preceding; - -#pragma loop(ivdep) - while (++_First != _Last) { - *_First = _Reduce_op(_Preceding, _STD move(*_First)); + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt>::value_type>) { + *_First = _Reduce_op(_Preceding, _STD move(*_First)); + } else { + ++_First2; + *_First = _Reduce_op(_Preceding, _STD move(*_First2)); + } } } @@ -3656,27 +3648,20 @@ struct _Scan_decoupled_lookback { #pragma loop(ivdep) for (; _First != _Last; ++_First) { - if (_STD next(_First) == _Last) { - *_First = _Reduce_op(_Preceding, _Local._Ref()); + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt>::value_type>) { + *_First = _Reduce_op(_Preceding, _STD move(*_First)); } else { - *_First = _Reduce_op(_Preceding, _STD move(*_First2)); - ++_First2; + if (_STD next(_First) == _Last) { + *_First = _Reduce_op(_Preceding, _Local._Ref()); + } else { + *_First = _Reduce_op(_Preceding, _STD move(*_First2)); + ++_First2; + } } } } - template - void _Apply_inclusive_predecessor_origin(_Ty& _Preceding, _FwdIt _First, const _FwdIt _Last, _BinOp _Reduce_op) { - // apply _Preceding to [_First, _Last) and _Sum._Ref(), using _Reduce_op - _STD _Implicitly_construct_in_place_by_binary_op(_Sum._Ref(), _Reduce_op, _Preceding, _Local._Ref()); - _State.store(_Local_available | _Sum_available); - -#pragma loop(ivdep) - for (; _First != _Last; ++_First) { - *_First = _Reduce_op(_Preceding, _STD move(*_First)); - } - } - ~_Scan_decoupled_lookback() { const auto _State_bits = _State.load(memory_order_relaxed); if (_State_bits & _Sum_available) { @@ -4389,28 +4374,13 @@ _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ } _Ty _Tmp = _Reduce_op(_Val, *_First); // temp to enable _First == _Dest - *_Dest = _Intermediate_result.emplace_back(_STD move(_Val)); - _Val = _STD move(_Tmp); - } -} - -template -_FwdIt2 _Exclusive_scan_per_chunk_origin( - _FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val) { - // local-sum for parallel exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and stores - // successor sum in _Val - // pre: _Val is *uninitialized* && _First != _Last - _STD _Construct_in_place(_Val, *_First); - for (;;) { - ++_First; - ++_Dest; - if (_First == _Last) { - return _Dest; + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { + *_Dest = _Val; + } else { + *_Dest = _Intermediate_result.emplace_back(_STD move(_Val)); } - - _Ty _Tmp = _Reduce_op(_Val, *_First); // temp to enable _First == _Dest - *_Dest = _Val; - _Val = _STD move(_Tmp); + _Val = _STD move(_Tmp); } } @@ -4474,36 +4444,20 @@ struct _Static_partitioned_exclusive_scan2 { return _Cancellation_status::_Running; } - // If the intermediate type is same with output type, then we can direct store the bop result in the output - if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { - // Calculate local sum and publish to other threads - const auto _Last = _STD _Exclusive_scan_per_chunk_origin( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref()); - _Chunk->_Store_available_state(_Local_available); - - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_exclusive_predecessor_origin(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); - } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); - } - } else { - vector<_Ty> _Intermediate_result; + vector<_Ty> _Intermediate_result; - // Calculate local sum and publish to other threads - const auto _Last = _STD _Exclusive_scan_per_chunk( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _Intermediate_result); - _Chunk->_Store_available_state(_Local_available); + // Calculate local sum and publish to other threads + const auto _Last = _STD _Exclusive_scan_per_chunk( + _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _Intermediate_result); + _Chunk->_Store_available_state(_Local_available); - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_exclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); - } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); - } + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_exclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + } else { + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); } return _Cancellation_status::_Running; @@ -4557,9 +4511,9 @@ _FwdIt2 exclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _Last, _FwdI return _Dest; } -template +template _FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, - _Ty_fwd&& _Predecessor, vector<_Ty>& _Intermediate_result) { + _Ty_fwd&& _Predecessor, [[maybe_unused]] _VectorTy&... _Intermediate_result) { // local-sum for parallel inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall sum in // _Val. // pre: _Val is *uninitialized* && _First != _Last @@ -4578,32 +4532,12 @@ _FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ return _Dest; } - _Val = _Reduce_op(_Intermediate_result.emplace_back(_STD move(_Val)), *_First); - } -} - -template -_FwdIt2 _Inclusive_scan_per_chunk_complete( - _FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, _Ty_fwd&& _Predecessor) { - // local-sum for parallel inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall sum in - // _Val. - // pre: _Val is *uninitialized* && _First != _Last - if constexpr (is_same_v<_No_init_tag, remove_const_t>>) { - _STD _Construct_in_place_by_deref(_Val, _First); - } else { - _STD _Implicitly_construct_in_place_by_binary_op_deref_rhs( - _Val, _Reduce_op, _STD forward<_Ty_fwd>(_Predecessor), _First); - } - - for (;;) { - *_Dest = _Val; - ++_Dest; - ++_First; - if (_First == _Last) { - return _Dest; + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (sizeof...(_VectorTy) == 0 || is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { + _Val = _Reduce_op(_STD move(_Val), *_First); + } else { + _Val = _Reduce_op(_Intermediate_result.emplace_back(_STD move(_Val))..., *_First); } - - _Val = _Reduce_op(_STD move(_Val), *_First); } } @@ -4634,7 +4568,7 @@ struct _Static_partitioned_inclusive_scan2 { // Run local inclusive_scan on this chunk const auto _Chunk = _Lookback.data() + static_cast(_Chunk_number); if (_Chunk_number == 0) { // chunk 0 is special as it has no predecessor; its local and total sums are the same - _STD _Inclusive_scan_per_chunk_complete( + _STD _Inclusive_scan_per_chunk( _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Sum._Ref(), _STD move(_Initial)); _Chunk->_Store_available_state(_Sum_available); return _Cancellation_status::_Running; @@ -4643,44 +4577,29 @@ struct _Static_partitioned_inclusive_scan2 { const auto _Prev_chunk = _STD _Prev_iter(_Chunk); if (_Prev_chunk->_State.load() & _Sum_available) { // if predecessor sum already complete, we can incorporate its value directly for 1 pass - _STD _Inclusive_scan_per_chunk_complete( + _STD _Inclusive_scan_per_chunk( _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Sum._Ref(), _Prev_chunk->_Sum._Ref()); _Chunk->_Store_available_state(_Sum_available); return _Cancellation_status::_Running; } - // If the intermediate type is same with output type, then we can direct store the bop result in the output - if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { - // Calculate local sum and publish to other threads - const auto _Last = _STD _Inclusive_scan_per_chunk_complete( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _No_init_tag{}); - _Chunk->_Store_available_state(_Local_available); - - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_inclusive_predecessor_origin(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); - } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); - } + // Make a vector to avoid the type of *_Dest is different with _Ty + vector<_Ty> _Intermediate_result; + + // Calculate local sum and publish to other threads + const auto _Last = _STD _Inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, + _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); + _Chunk->_Store_available_state(_Local_available); + + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_inclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); } else { - // Make a vector to avoid the type of *_Dest is different with _Ty - vector<_Ty> _Intermediate_result; - - // Calculate local sum and publish to other threads - const auto _Last = _STD _Inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, - _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); - _Chunk->_Store_available_state(_Local_available); - - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_inclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); - } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); - } + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); } + return _Cancellation_status::_Running; } @@ -4792,28 +4711,13 @@ _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, } _Ty _Tmp = _Reduce_op(_Val, _Transform_op(*_First)); // temp to enable _First == _Dest - *_Dest = _Intermediate_result.emplace_back(_STD move(_Val)); - _Val = _STD move(_Tmp); - } -} - -template -_FwdIt2 _Transform_exclusive_scan_per_chunk_origin( - _FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _UnaryOp _Transform_op, _Ty& _Val) { - // Local-sum for parallel transform_exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and - // stores successor sum in _Val. - // pre: _Val is *uninitialized* && _First != _Last - _STD _Construct_in_place_by_transform_deref(_Val, _Transform_op, _First); - for (;;) { - ++_First; - ++_Dest; - if (_First == _Last) { - return _Dest; + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { + *_Dest = _Val; + } else { + *_Dest = _Intermediate_result.emplace_back(_STD move(_Val)); } - - _Ty _Tmp = _Reduce_op(_Val, _Transform_op(*_First)); // temp to enable _First == _Dest - *_Dest = _Val; - _Val = _STD move(_Tmp); + _Val = _STD move(_Tmp); } } @@ -4879,38 +4783,23 @@ struct _Static_partitioned_transform_exclusive_scan2 { return _Cancellation_status::_Running; } - // If the intermediate type is same with output type, then we can direct store the bop result in the output - if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { - // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_exclusive_scan_per_chunk_origin( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Local._Ref()); - _Chunk->_Store_available_state(_Local_available); - - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_exclusive_predecessor_origin(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); - } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); - } + // Make a vector to avoid the type of *_Dest is different with _Ty + vector<_Ty> _Intermediate_result; + + // Calculate local sum and publish to other threads + const auto _Last = _STD _Transform_exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _Intermediate_result); + _Chunk->_Store_available_state(_Local_available); + + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_exclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); } else { - // Make a vector to avoid the type of *_Dest is different with _Ty - vector<_Ty> _Intermediate_result; - - // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, - _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _Intermediate_result); - _Chunk->_Store_available_state(_Local_available); - - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_exclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); - } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); - } + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); } + return _Cancellation_status::_Running; } @@ -4962,9 +4851,9 @@ _FwdIt2 transform_exclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _L return _Dest; } -template +template _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, - _UnaryOp _Transform_op, _Ty& _Val, _Ty_fwd&& _Predecessor, vector<_Ty>& _Intermediate_result) { + _UnaryOp _Transform_op, _Ty& _Val, _Ty_fwd&& _Predecessor, [[maybe_unused]] _VectorTy&... _Intermediate_result) { // local-sum for parallel transform_inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall // sum in _Val // pre: _Val is *uninitialized* && _First != _Last @@ -4983,32 +4872,12 @@ _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, return _Dest; } - _Val = _Reduce_op(_Intermediate_result.emplace_back(_STD move(_Val)), _Transform_op(*_First)); - } -} - -template -_FwdIt2 _Transform_inclusive_scan_per_chunk_complete(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, - _BinOp _Reduce_op, _UnaryOp _Transform_op, _Ty& _Val, _Ty_fwd&& _Predecessor) { - // local-sum for parallel transform_inclusive_scan; writes local inclusive prefix sums into _Dest and stores overall - // sum in _Val - // pre: _Val is *uninitialized* && _First != _Last - if constexpr (is_same_v<_No_init_tag, remove_const_t>>) { - _STD _Construct_in_place_by_transform_deref(_Val, _Transform_op, _First); - } else { - _STD _Implicitly_construct_in_place_by_binary_op_transform_deref_rhs( - _Val, _Reduce_op, _Transform_op, _STD forward<_Ty_fwd>(_Predecessor), _First); - } - - for (;;) { - *_Dest = _Val; - ++_Dest; - ++_First; - if (_First == _Last) { - return _Dest; + // If the intermediate type is same with output type, then we can direct store the bop result in the output + if constexpr (sizeof...(_VectorTy) == 0 || is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { + _Val = _Reduce_op(_STD move(_Val), _Transform_op(*_First)); + } else { + _Val = _Reduce_op(_Intermediate_result.emplace_back(_STD move(_Val))..., _Transform_op(*_First)); } - - _Val = _Reduce_op(_STD move(_Val), _Transform_op(*_First)); } } @@ -5040,7 +4909,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { // Run local transform_inclusive_scan on this chunk const auto _Chunk = _Lookback.data() + static_cast(_Chunk_number); if (_Chunk_number == 0) { // chunk 0 is special as it has no predecessor; its local and total sums are the same - _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, _Dest, _Reduce_op, + _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Sum._Ref(), _STD move(_Initial)); _Chunk->_Store_available_state(_Sum_available); return _Cancellation_status::_Running; @@ -5049,44 +4918,29 @@ struct _Static_partitioned_transform_inclusive_scan2 { const auto _Prev_chunk = _STD _Prev_iter(_Chunk); if (_Prev_chunk->_State.load() & _Sum_available) { // if predecessor sum already complete, we can incorporate its value directly for 1 pass - _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, _Dest, _Reduce_op, + _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, _Transform_op, _Chunk->_Sum._Ref(), _Prev_chunk->_Sum._Ref()); _Chunk->_Store_available_state(_Sum_available); return _Cancellation_status::_Running; } - // If the intermediate type is same with output type, then we can direct store the bop result in the output - if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { - // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_inclusive_scan_per_chunk_complete(_In_range._First, _In_range._Last, - _Dest, _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}); - _Chunk->_Store_available_state(_Local_available); - - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_inclusive_predecessor_origin(_Prev_chunk->_Sum._Ref(), _Dest, _Last, _Reduce_op); - } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor_origin(_Tmp, _Dest, _Last, _Reduce_op); - } + // Make a vector to avoid the type of *_Dest is different with _Ty + vector<_Ty> _Intermediate_result; + + // Calculate local sum and publish to other threads + const auto _Last = _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); + _Chunk->_Store_available_state(_Local_available); + + // Apply the predecessor overall sum to current overall sum and elements + if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly + _Chunk->_Apply_inclusive_predecessor( + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); } else { - // Make a vector to avoid the type of *_Dest is different with _Ty - vector<_Ty> _Intermediate_result; - - // Calculate local sum and publish to other threads - const auto _Last = _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, - _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); - _Chunk->_Store_available_state(_Local_available); - - // Apply the predecessor overall sum to current overall sum and elements - if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly - _Chunk->_Apply_inclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); - } else { - auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); - } + auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); } + return _Cancellation_status::_Running; } From 0b751e8b1c9c780db4b577f036cd4e0d7f42277e Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 09:08:27 -0700 Subject: [PATCH 10/17] Remove mysterious runtime branch. Nobody will notice, right? --- stl/inc/execution | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 5d1d5739f77..f7cede24198 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3652,12 +3652,8 @@ struct _Scan_decoupled_lookback { if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt>::value_type>) { *_First = _Reduce_op(_Preceding, _STD move(*_First)); } else { - if (_STD next(_First) == _Last) { - *_First = _Reduce_op(_Preceding, _Local._Ref()); - } else { - *_First = _Reduce_op(_Preceding, _STD move(*_First2)); - ++_First2; - } + *_First = _Reduce_op(_Preceding, _STD move(*_First2)); + ++_First2; } } } From e35ff3a0201d3cc1b51f915ca24abd692233b3a8 Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 09:38:05 -0700 Subject: [PATCH 11/17] Improve comment grammar. --- stl/inc/execution | 19 ++++++++++--------- 1 file changed, 10 insertions(+), 9 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index f7cede24198..3405a6c2ff2 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3629,7 +3629,7 @@ struct _Scan_decoupled_lookback { #pragma loop(ivdep) while (++_First != _Last) { - // If the intermediate type is same with output type, then we can direct store the bop result in the output + // If the intermediate and output types are the same, we can directly store the bop result in the output if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt>::value_type>) { *_First = _Reduce_op(_Preceding, _STD move(*_First)); } else { @@ -3648,7 +3648,7 @@ struct _Scan_decoupled_lookback { #pragma loop(ivdep) for (; _First != _Last; ++_First) { - // If the intermediate type is same with output type, then we can direct store the bop result in the output + // If the intermediate and output types are the same, we can directly store the bop result in the output if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt>::value_type>) { *_First = _Reduce_op(_Preceding, _STD move(*_First)); } else { @@ -4370,7 +4370,7 @@ _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ } _Ty _Tmp = _Reduce_op(_Val, *_First); // temp to enable _First == _Dest - // If the intermediate type is same with output type, then we can direct store the bop result in the output + // If the intermediate and output types are the same, we can directly store the bop result in the output if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { *_Dest = _Val; } else { @@ -4440,6 +4440,7 @@ struct _Static_partitioned_exclusive_scan2 { return _Cancellation_status::_Running; } + // Make a vector to handle the intermediate and output types being different vector<_Ty> _Intermediate_result; // Calculate local sum and publish to other threads @@ -4528,7 +4529,7 @@ _FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ return _Dest; } - // If the intermediate type is same with output type, then we can direct store the bop result in the output + // If the intermediate and output types are the same, we can directly store the bop result in the output if constexpr (sizeof...(_VectorTy) == 0 || is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { _Val = _Reduce_op(_STD move(_Val), *_First); } else { @@ -4579,7 +4580,7 @@ struct _Static_partitioned_inclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to avoid the type of *_Dest is different with _Ty + // Make a vector to handle the intermediate and output types being different vector<_Ty> _Intermediate_result; // Calculate local sum and publish to other threads @@ -4707,7 +4708,7 @@ _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, } _Ty _Tmp = _Reduce_op(_Val, _Transform_op(*_First)); // temp to enable _First == _Dest - // If the intermediate type is same with output type, then we can direct store the bop result in the output + // If the intermediate and output types are the same, we can directly store the bop result in the output if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { *_Dest = _Val; } else { @@ -4779,7 +4780,7 @@ struct _Static_partitioned_transform_exclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to avoid the type of *_Dest is different with _Ty + // Make a vector to handle the intermediate and output types being different vector<_Ty> _Intermediate_result; // Calculate local sum and publish to other threads @@ -4868,7 +4869,7 @@ _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, return _Dest; } - // If the intermediate type is same with output type, then we can direct store the bop result in the output + // If the intermediate and output types are the same, we can directly store the bop result in the output if constexpr (sizeof...(_VectorTy) == 0 || is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { _Val = _Reduce_op(_STD move(_Val), _Transform_op(*_First)); } else { @@ -4920,7 +4921,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to avoid the type of *_Dest is different with _Ty + // Make a vector to handle the intermediate and output types being different vector<_Ty> _Intermediate_result; // Calculate local sum and publish to other threads From 8830be2690abc244c5f5b5f7bbf158cfb9bc7166 Mon Sep 17 00:00:00 2001 From: Casey Carter Date: Mon, 12 Aug 2024 15:05:26 -0700 Subject: [PATCH 12/17] Pre-allocate storage for intermediates --- stl/inc/execution | 85 ++++++++++++++++++++++++++++++----------------- 1 file changed, 54 insertions(+), 31 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 3405a6c2ff2..7ac3da6d322 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -4357,7 +4357,7 @@ struct _No_init_tag { template _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, _Ty& _Val, - vector<_Ty>& _Intermediate_result) { + _Parallel_vector<_Ty>& _Intermediate_result) { // local-sum for parallel exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and stores // successor sum in _Val // pre: _Val is *uninitialized* && _First != _Last @@ -4374,6 +4374,7 @@ _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { *_Dest = _Val; } else { + _STL_INTERNAL_CHECK(_Intermediate_result.size() < _Intermediate_result.capacity()); *_Dest = _Intermediate_result.emplace_back(_STD move(_Val)); } _Val = _STD move(_Tmp); @@ -4396,6 +4397,24 @@ void _Exclusive_scan_per_chunk_complete( } } +template +struct _Scan_intermediate_storage { + _Parallel_vector<_Parallel_vector<_Ty>> _Data; + + template + _Scan_intermediate_storage(const _Static_partition_team<_Diff>& _Team) : _Data(_Team._Chunks) { + const auto _Chunk_size = static_cast(_Team._Chunk_size); + for (auto& _Vec : _Data) { + _Vec.reserve(_Chunk_size); + } + } + + _Parallel_vector<_Ty>& operator[](const size_t _Idx) noexcept { + _STL_INTERNAL_CHECK(_Idx < _Data.size()); + return _Data[_Idx]; + } +}; + template struct _Static_partitioned_exclusive_scan2 { using _Diff = _Common_diff_t<_FwdIt1, _FwdIt2>; @@ -4403,13 +4422,14 @@ struct _Static_partitioned_exclusive_scan2 { _Static_partition_range<_FwdIt1, _Diff> _Basis1; _Static_partition_range<_FwdIt2, _Diff> _Basis2; _Parallel_vector<_Scan_decoupled_lookback<_Ty>> _Lookback; + _Scan_intermediate_storage<_Ty> _Intermediate_result; _Ty& _Initial; _BinOp _Reduce_op; _Static_partitioned_exclusive_scan2(const size_t _Hw_threads, const _Diff _Count, const _FwdIt1 _First, _Ty& _Initial_, _BinOp _Reduce_op_, const _FwdIt2&) : _Team{_Count, _Get_chunked_work_chunk_count(_Hw_threads, _Count)}, _Basis1{}, _Basis2{}, - _Lookback(_Team._Chunks), _Initial(_Initial_), _Reduce_op(_Reduce_op_) { + _Lookback(_Team._Chunks), _Intermediate_result{_Team}, _Initial(_Initial_), _Reduce_op(_Reduce_op_) { _Basis1._Populate(_Team, _First); } @@ -4440,23 +4460,22 @@ struct _Static_partitioned_exclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to handle the intermediate and output types being different - vector<_Ty> _Intermediate_result; - // Calculate local sum and publish to other threads - const auto _Last = _STD _Exclusive_scan_per_chunk( - _In_range._First, _In_range._Last, _Dest, _Reduce_op, _Chunk->_Local._Ref(), _Intermediate_result); + const auto _Last = _STD _Exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, + _Chunk->_Local._Ref(), _Intermediate_result[_Chunk_number]); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly _Chunk->_Apply_exclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor( + _Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } + _Intermediate_result[_Chunk_number].clear(); return _Cancellation_status::_Running; } @@ -4533,6 +4552,7 @@ _FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ if constexpr (sizeof...(_VectorTy) == 0 || is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { _Val = _Reduce_op(_STD move(_Val), *_First); } else { + _STL_INTERNAL_CHECK(((_Intermediate_result.size() < _Intermediate_result.capacity()) && ...)); _Val = _Reduce_op(_Intermediate_result.emplace_back(_STD move(_Val))..., *_First); } } @@ -4545,13 +4565,14 @@ struct _Static_partitioned_inclusive_scan2 { _Static_partition_range<_FwdIt1, _Diff> _Basis1; _Static_partition_range<_FwdIt2, _Diff> _Basis2; _Parallel_vector<_Scan_decoupled_lookback<_Ty>> _Lookback; + _Scan_intermediate_storage<_Ty> _Intermediate_result; _BinOp _Reduce_op; _Init_ty& _Initial; _Static_partitioned_inclusive_scan2( const size_t _Hw_threads, const _Diff _Count, _BinOp _Reduce_op_, _Init_ty& _Initial_) : _Team{_Count, _Get_chunked_work_chunk_count(_Hw_threads, _Count)}, _Basis1{}, _Basis2{}, - _Lookback(_Team._Chunks), _Reduce_op(_Reduce_op_), _Initial(_Initial_) {} + _Lookback(_Team._Chunks), _Intermediate_result{_Team}, _Reduce_op(_Reduce_op_), _Initial(_Initial_) {} _Cancellation_status _Process_chunk() { const auto _Key = _Team._Get_next_key(); @@ -4580,23 +4601,22 @@ struct _Static_partitioned_inclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to handle the intermediate and output types being different - vector<_Ty> _Intermediate_result; - // Calculate local sum and publish to other threads const auto _Last = _STD _Inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, _Reduce_op, - _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); + _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result[_Chunk_number]); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly _Chunk->_Apply_inclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor( + _Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } + _Intermediate_result[_Chunk_number].clear(); return _Cancellation_status::_Running; } @@ -4695,7 +4715,7 @@ _FwdIt2 inclusive_scan(_ExPo&&, _FwdIt1 _First, _FwdIt1 _Last, _FwdIt2 _Dest, _B template _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _Dest, _BinOp _Reduce_op, - _UnaryOp _Transform_op, _Ty& _Val, vector<_Ty>& _Intermediate_result) { + _UnaryOp _Transform_op, _Ty& _Val, _Parallel_vector<_Ty>& _Intermediate_result) { // Local-sum for parallel transform_exclusive_scan; writes local sums into [_Dest + 1, _Dest + (_Last - _First)) and // stores successor sum in _Val. // pre: _Val is *uninitialized* && _First != _Last @@ -4712,6 +4732,7 @@ _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { *_Dest = _Val; } else { + _STL_INTERNAL_CHECK(_Intermediate_result.size() < _Intermediate_result.capacity()); *_Dest = _Intermediate_result.emplace_back(_STD move(_Val)); } _Val = _STD move(_Tmp); @@ -4742,6 +4763,7 @@ struct _Static_partitioned_transform_exclusive_scan2 { _Static_partition_range<_FwdIt1, _Diff> _Basis1; _Static_partition_range<_FwdIt2, _Diff> _Basis2; _Parallel_vector<_Scan_decoupled_lookback<_Ty>> _Lookback; + _Scan_intermediate_storage<_Ty> _Intermediate_result; _Ty& _Initial; _BinOp _Reduce_op; _UnaryOp _Transform_op; @@ -4749,7 +4771,8 @@ struct _Static_partitioned_transform_exclusive_scan2 { _Static_partitioned_transform_exclusive_scan2(const size_t _Hw_threads, const _Diff _Count, const _FwdIt1 _First, _Ty& _Initial_, _BinOp _Reduce_op_, _UnaryOp _Transform_op_, const _FwdIt2&) : _Team{_Count, _Get_chunked_work_chunk_count(_Hw_threads, _Count)}, _Basis1{}, _Basis2{}, - _Lookback(_Team._Chunks), _Initial(_Initial_), _Reduce_op(_Reduce_op_), _Transform_op(_Transform_op_) { + _Lookback(_Team._Chunks), _Intermediate_result{_Team}, _Initial(_Initial_), _Reduce_op(_Reduce_op_), + _Transform_op(_Transform_op_) { _Basis1._Populate(_Team, _First); } @@ -4780,23 +4803,21 @@ struct _Static_partitioned_transform_exclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to handle the intermediate and output types being different - vector<_Ty> _Intermediate_result; - // Calculate local sum and publish to other threads const auto _Last = _STD _Transform_exclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, - _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _Intermediate_result); + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _Intermediate_result[_Chunk_number]); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly _Chunk->_Apply_exclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } + _Intermediate_result[_Chunk_number].clear(); return _Cancellation_status::_Running; } @@ -4873,6 +4894,7 @@ _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, if constexpr (sizeof...(_VectorTy) == 0 || is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { _Val = _Reduce_op(_STD move(_Val), _Transform_op(*_First)); } else { + _STL_INTERNAL_CHECK(((_Intermediate_result.size() < _Intermediate_result.capacity()) && ...)); _Val = _Reduce_op(_Intermediate_result.emplace_back(_STD move(_Val))..., _Transform_op(*_First)); } } @@ -4885,6 +4907,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { _Static_partition_range<_FwdIt1, _Diff> _Basis1; _Static_partition_range<_FwdIt2, _Diff> _Basis2; _Parallel_vector<_Scan_decoupled_lookback<_Ty>> _Lookback; + _Scan_intermediate_storage<_Ty> _Intermediate_result; _BinOp _Reduce_op; _UnaryOp _Transform_op; _Init_ty& _Initial; @@ -4892,7 +4915,8 @@ struct _Static_partitioned_transform_inclusive_scan2 { _Static_partitioned_transform_inclusive_scan2( const size_t _Hw_threads, const _Diff _Count, _BinOp _Reduce_op_, _UnaryOp _Transform_op_, _Init_ty& _Initial_) : _Team{_Count, _Get_chunked_work_chunk_count(_Hw_threads, _Count)}, _Basis1{}, _Basis2{}, - _Lookback(_Team._Chunks), _Reduce_op(_Reduce_op_), _Transform_op(_Transform_op_), _Initial(_Initial_) {} + _Lookback(_Team._Chunks), _Intermediate_result{_Team}, _Reduce_op(_Reduce_op_), _Transform_op(_Transform_op_), + _Initial(_Initial_) {} _Cancellation_status _Process_chunk() { const auto _Key = _Team._Get_next_key(); @@ -4921,23 +4945,22 @@ struct _Static_partitioned_transform_inclusive_scan2 { return _Cancellation_status::_Running; } - // Make a vector to handle the intermediate and output types being different - vector<_Ty> _Intermediate_result; - // Calculate local sum and publish to other threads const auto _Last = _STD _Transform_inclusive_scan_per_chunk(_In_range._First, _In_range._Last, _Dest, - _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result); + _Reduce_op, _Transform_op, _Chunk->_Local._Ref(), _No_init_tag{}, _Intermediate_result[_Chunk_number]); _Chunk->_Store_available_state(_Local_available); // Apply the predecessor overall sum to current overall sum and elements if (_Prev_chunk->_Get_available_state() & _Sum_available) { // predecessor overall sum done, use directly _Chunk->_Apply_inclusive_predecessor( - _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_inclusive_predecessor(_Tmp, _Dest, _Intermediate_result.data(), _Last, _Reduce_op); + _Chunk->_Apply_inclusive_predecessor( + _Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } + _Intermediate_result[_Chunk_number].clear(); return _Cancellation_status::_Running; } From f65e3bc613d8eb0c31c37fe958de50367073ab64 Mon Sep 17 00:00:00 2001 From: Casey Carter Date: Mon, 12 Aug 2024 16:14:34 -0700 Subject: [PATCH 13/17] Don't allocate space we won't use --- stl/inc/execution | 102 ++++++++++++++++++++++++++++------------------ 1 file changed, 63 insertions(+), 39 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 7ac3da6d322..cb6d3e555e1 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3594,6 +3594,45 @@ _FwdIt partition(_ExPo&&, _FwdIt _First, const _FwdIt _Last, _Pr _Pred) noexcept inline constexpr unsigned char _Local_available = 1; inline constexpr unsigned char _Sum_available = 2; +// For scan algorithms: if the intermediate and output types are the same, we can directly store the binary operation +// result in the output. IF not, then we need a place to store intermediate results. +template +constexpr bool _Scan_reuse_output = is_same_v<_Ty, typename iterator_traits<_Iter>::value_type>; + +template > +class _Scan_intermediate_storage { +private: + _Parallel_vector<_Ty> _Data; + +public: + template + _Scan_intermediate_storage(const _Static_partition_team<_Diff>&) noexcept {} + + _Parallel_vector<_Ty>& operator[](size_t) noexcept { + return _Data; + } +}; + +template +class _Scan_intermediate_storage<_Ty, _Iter, false> { +private: + _Parallel_vector<_Parallel_vector<_Ty>> _Data; + +public: + template + _Scan_intermediate_storage(const _Static_partition_team<_Diff>& _Team) : _Data(_Team._Chunks) { + const auto _Chunk_size = static_cast(_Team._Chunk_size); + for (auto& _Vec : _Data) { + _Vec.reserve(_Chunk_size); + } + } + + _Parallel_vector<_Ty>& operator[](const size_t _Idx) noexcept { + _STL_INTERNAL_CHECK(_Idx < _Data.size()); + return _Data[_Idx]; + } +}; + template struct _Scan_decoupled_lookback { // inter-chunk communication block in "Single-pass Parallel Prefix Scan with Decoupled Look-back" by Merrill and @@ -3629,8 +3668,7 @@ struct _Scan_decoupled_lookback { #pragma loop(ivdep) while (++_First != _Last) { - // If the intermediate and output types are the same, we can directly store the bop result in the output - if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt>::value_type>) { + if constexpr (_Scan_reuse_output<_Ty, _FwdIt>) { *_First = _Reduce_op(_Preceding, _STD move(*_First)); } else { ++_First2; @@ -3648,8 +3686,7 @@ struct _Scan_decoupled_lookback { #pragma loop(ivdep) for (; _First != _Last; ++_First) { - // If the intermediate and output types are the same, we can directly store the bop result in the output - if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt>::value_type>) { + if constexpr (_Scan_reuse_output<_Ty, _FwdIt>) { *_First = _Reduce_op(_Preceding, _STD move(*_First)); } else { *_First = _Reduce_op(_Preceding, _STD move(*_First2)); @@ -4370,8 +4407,7 @@ _FwdIt2 _Exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ } _Ty _Tmp = _Reduce_op(_Val, *_First); // temp to enable _First == _Dest - // If the intermediate and output types are the same, we can directly store the bop result in the output - if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { + if constexpr (_Scan_reuse_output<_Ty, _FwdIt2>) { *_Dest = _Val; } else { _STL_INTERNAL_CHECK(_Intermediate_result.size() < _Intermediate_result.capacity()); @@ -4397,24 +4433,6 @@ void _Exclusive_scan_per_chunk_complete( } } -template -struct _Scan_intermediate_storage { - _Parallel_vector<_Parallel_vector<_Ty>> _Data; - - template - _Scan_intermediate_storage(const _Static_partition_team<_Diff>& _Team) : _Data(_Team._Chunks) { - const auto _Chunk_size = static_cast(_Team._Chunk_size); - for (auto& _Vec : _Data) { - _Vec.reserve(_Chunk_size); - } - } - - _Parallel_vector<_Ty>& operator[](const size_t _Idx) noexcept { - _STL_INTERNAL_CHECK(_Idx < _Data.size()); - return _Data[_Idx]; - } -}; - template struct _Static_partitioned_exclusive_scan2 { using _Diff = _Common_diff_t<_FwdIt1, _FwdIt2>; @@ -4422,7 +4440,7 @@ struct _Static_partitioned_exclusive_scan2 { _Static_partition_range<_FwdIt1, _Diff> _Basis1; _Static_partition_range<_FwdIt2, _Diff> _Basis2; _Parallel_vector<_Scan_decoupled_lookback<_Ty>> _Lookback; - _Scan_intermediate_storage<_Ty> _Intermediate_result; + _Scan_intermediate_storage<_Ty, _FwdIt2> _Intermediate_result; _Ty& _Initial; _BinOp _Reduce_op; @@ -4475,7 +4493,9 @@ struct _Static_partitioned_exclusive_scan2 { _Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } - _Intermediate_result[_Chunk_number].clear(); + if constexpr (!_Scan_reuse_output<_Ty, _FwdIt2>) { + _Intermediate_result[_Chunk_number].clear(); + } return _Cancellation_status::_Running; } @@ -4548,8 +4568,7 @@ _FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ return _Dest; } - // If the intermediate and output types are the same, we can directly store the bop result in the output - if constexpr (sizeof...(_VectorTy) == 0 || is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { + if constexpr (sizeof...(_VectorTy) == 0 || _Scan_reuse_output<_Ty, _FwdIt2>) { _Val = _Reduce_op(_STD move(_Val), *_First); } else { _STL_INTERNAL_CHECK(((_Intermediate_result.size() < _Intermediate_result.capacity()) && ...)); @@ -4565,7 +4584,7 @@ struct _Static_partitioned_inclusive_scan2 { _Static_partition_range<_FwdIt1, _Diff> _Basis1; _Static_partition_range<_FwdIt2, _Diff> _Basis2; _Parallel_vector<_Scan_decoupled_lookback<_Ty>> _Lookback; - _Scan_intermediate_storage<_Ty> _Intermediate_result; + _Scan_intermediate_storage<_Ty, _FwdIt2> _Intermediate_result; _BinOp _Reduce_op; _Init_ty& _Initial; @@ -4616,7 +4635,9 @@ struct _Static_partitioned_inclusive_scan2 { _Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } - _Intermediate_result[_Chunk_number].clear(); + if constexpr (!_Scan_reuse_output<_Ty, _FwdIt2>) { + _Intermediate_result[_Chunk_number].clear(); + } return _Cancellation_status::_Running; } @@ -4728,8 +4749,7 @@ _FwdIt2 _Transform_exclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, } _Ty _Tmp = _Reduce_op(_Val, _Transform_op(*_First)); // temp to enable _First == _Dest - // If the intermediate and output types are the same, we can directly store the bop result in the output - if constexpr (is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { + if constexpr (_Scan_reuse_output<_Ty, _FwdIt2>) { *_Dest = _Val; } else { _STL_INTERNAL_CHECK(_Intermediate_result.size() < _Intermediate_result.capacity()); @@ -4763,7 +4783,7 @@ struct _Static_partitioned_transform_exclusive_scan2 { _Static_partition_range<_FwdIt1, _Diff> _Basis1; _Static_partition_range<_FwdIt2, _Diff> _Basis2; _Parallel_vector<_Scan_decoupled_lookback<_Ty>> _Lookback; - _Scan_intermediate_storage<_Ty> _Intermediate_result; + _Scan_intermediate_storage<_Ty, _FwdIt2> _Intermediate_result; _Ty& _Initial; _BinOp _Reduce_op; _UnaryOp _Transform_op; @@ -4814,10 +4834,13 @@ struct _Static_partitioned_transform_exclusive_scan2 { _Prev_chunk->_Sum._Ref(), _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } else { auto _Tmp = _STD _Get_lookback_sum(_Prev_chunk, _Reduce_op); - _Chunk->_Apply_exclusive_predecessor(_Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); + _Chunk->_Apply_exclusive_predecessor( + _Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } - _Intermediate_result[_Chunk_number].clear(); + if constexpr (!_Scan_reuse_output<_Ty, _FwdIt2>) { + _Intermediate_result[_Chunk_number].clear(); + } return _Cancellation_status::_Running; } @@ -4890,8 +4913,7 @@ _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, return _Dest; } - // If the intermediate and output types are the same, we can directly store the bop result in the output - if constexpr (sizeof...(_VectorTy) == 0 || is_same_v<_Ty, typename iterator_traits<_FwdIt2>::value_type>) { + if constexpr (sizeof...(_VectorTy) == 0 || _Scan_reuse_output<_Ty, _FwdIt2>) { _Val = _Reduce_op(_STD move(_Val), _Transform_op(*_First)); } else { _STL_INTERNAL_CHECK(((_Intermediate_result.size() < _Intermediate_result.capacity()) && ...)); @@ -4907,7 +4929,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { _Static_partition_range<_FwdIt1, _Diff> _Basis1; _Static_partition_range<_FwdIt2, _Diff> _Basis2; _Parallel_vector<_Scan_decoupled_lookback<_Ty>> _Lookback; - _Scan_intermediate_storage<_Ty> _Intermediate_result; + _Scan_intermediate_storage<_Ty, _FwdIt2> _Intermediate_result; _BinOp _Reduce_op; _UnaryOp _Transform_op; _Init_ty& _Initial; @@ -4960,7 +4982,9 @@ struct _Static_partitioned_transform_inclusive_scan2 { _Tmp, _Dest, _Intermediate_result[_Chunk_number].data(), _Last, _Reduce_op); } - _Intermediate_result[_Chunk_number].clear(); + if constexpr (!_Scan_reuse_output<_Ty, _FwdIt2>) { + _Intermediate_result[_Chunk_number].clear(); + } return _Cancellation_status::_Running; } From c2fd1909b3690be6bef3cb395cde523997e053a4 Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 23:16:20 -0700 Subject: [PATCH 14/17] Update comment. --- stl/inc/execution | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/stl/inc/execution b/stl/inc/execution index cb6d3e555e1..e6d854e91d1 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3595,7 +3595,7 @@ inline constexpr unsigned char _Local_available = 1; inline constexpr unsigned char _Sum_available = 2; // For scan algorithms: if the intermediate and output types are the same, we can directly store the binary operation -// result in the output. IF not, then we need a place to store intermediate results. +// result in the output. If not, then we need a place to store intermediate results. template constexpr bool _Scan_reuse_output = is_same_v<_Ty, typename iterator_traits<_Iter>::value_type>; From d8703eec48ffdf1f5c20e8689ab0d2a080abedae Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 23:31:38 -0700 Subject: [PATCH 15/17] All shall love `_NODISCARD` and despair. --- stl/inc/execution | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index e6d854e91d1..b22ba24959e 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3608,7 +3608,7 @@ public: template _Scan_intermediate_storage(const _Static_partition_team<_Diff>&) noexcept {} - _Parallel_vector<_Ty>& operator[](size_t) noexcept { + _NODISCARD _Parallel_vector<_Ty>& operator[](size_t) noexcept { return _Data; } }; @@ -3627,7 +3627,7 @@ public: } } - _Parallel_vector<_Ty>& operator[](const size_t _Idx) noexcept { + _NODISCARD _Parallel_vector<_Ty>& operator[](const size_t _Idx) noexcept { _STL_INTERNAL_CHECK(_Idx < _Data.size()); return _Data[_Idx]; } From 28174fef1b4bb54b6b408717ba5bedfee06e4c88 Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 23:39:48 -0700 Subject: [PATCH 16/17] All shall love `explicit` and despair. --- stl/inc/execution | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index b22ba24959e..4f1c7b7a3bc 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -3606,7 +3606,7 @@ private: public: template - _Scan_intermediate_storage(const _Static_partition_team<_Diff>&) noexcept {} + explicit _Scan_intermediate_storage(const _Static_partition_team<_Diff>&) noexcept {} _NODISCARD _Parallel_vector<_Ty>& operator[](size_t) noexcept { return _Data; @@ -3620,7 +3620,7 @@ private: public: template - _Scan_intermediate_storage(const _Static_partition_team<_Diff>& _Team) : _Data(_Team._Chunks) { + explicit _Scan_intermediate_storage(const _Static_partition_team<_Diff>& _Team) : _Data(_Team._Chunks) { const auto _Chunk_size = static_cast(_Team._Chunk_size); for (auto& _Vec : _Data) { _Vec.reserve(_Chunk_size); From 7d072260d7e3b019b9cf09289340e17a40ccd787 Mon Sep 17 00:00:00 2001 From: "Stephan T. Lavavej" Date: Mon, 12 Aug 2024 23:26:01 -0700 Subject: [PATCH 17/17] Rename to avoid ODR violations: `(_Static_partitioned_(transform_)?(ex|in)clusive)_scan2` => `$1_scan3` --- stl/inc/execution | 36 ++++++++++++++++++------------------ 1 file changed, 18 insertions(+), 18 deletions(-) diff --git a/stl/inc/execution b/stl/inc/execution index 4f1c7b7a3bc..d331de71d60 100644 --- a/stl/inc/execution +++ b/stl/inc/execution @@ -4434,7 +4434,7 @@ void _Exclusive_scan_per_chunk_complete( } template -struct _Static_partitioned_exclusive_scan2 { +struct _Static_partitioned_exclusive_scan3 { using _Diff = _Common_diff_t<_FwdIt1, _FwdIt2>; _Static_partition_team<_Diff> _Team; _Static_partition_range<_FwdIt1, _Diff> _Basis1; @@ -4444,7 +4444,7 @@ struct _Static_partitioned_exclusive_scan2 { _Ty& _Initial; _BinOp _Reduce_op; - _Static_partitioned_exclusive_scan2(const size_t _Hw_threads, const _Diff _Count, const _FwdIt1 _First, + _Static_partitioned_exclusive_scan3(const size_t _Hw_threads, const _Diff _Count, const _FwdIt1 _First, _Ty& _Initial_, _BinOp _Reduce_op_, const _FwdIt2&) : _Team{_Count, _Get_chunked_work_chunk_count(_Hw_threads, _Count)}, _Basis1{}, _Basis2{}, _Lookback(_Team._Chunks), _Intermediate_result{_Team}, _Initial(_Initial_), _Reduce_op(_Reduce_op_) { @@ -4501,7 +4501,7 @@ struct _Static_partitioned_exclusive_scan2 { static void __stdcall _Threadpool_callback( __std_PTP_CALLBACK_INSTANCE, void* const _Context, __std_PTP_WORK) noexcept /* terminates */ { - _STD _Run_available_chunked_work(*static_cast<_Static_partitioned_exclusive_scan2*>(_Context)); + _STD _Run_available_chunked_work(*static_cast<_Static_partitioned_exclusive_scan3*>(_Context)); } }; @@ -4522,7 +4522,7 @@ _FwdIt2 exclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _Last, _FwdI const auto _UDest = _STD _Get_unwrapped_n(_Dest, _Count); if (_Count >= 2) { // ... with at least 2 elements _TRY_BEGIN - _Static_partitioned_exclusive_scan2 _Operation{ + _Static_partitioned_exclusive_scan3 _Operation{ _Hw_threads, _Count, _UFirst, _Val, _STD _Pass_fn(_Reduce_op), _UDest}; _STD _Seek_wrapped(_Dest, _Operation._Basis2._Populate(_Operation._Team, _UDest)); // Note that _Val is used as temporary storage by whichever thread runs the first chunk. @@ -4578,7 +4578,7 @@ _FwdIt2 _Inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, _FwdIt2 _ } template -struct _Static_partitioned_inclusive_scan2 { +struct _Static_partitioned_inclusive_scan3 { using _Diff = _Common_diff_t<_FwdIt1, _FwdIt2>; _Static_partition_team<_Diff> _Team; _Static_partition_range<_FwdIt1, _Diff> _Basis1; @@ -4588,7 +4588,7 @@ struct _Static_partitioned_inclusive_scan2 { _BinOp _Reduce_op; _Init_ty& _Initial; - _Static_partitioned_inclusive_scan2( + _Static_partitioned_inclusive_scan3( const size_t _Hw_threads, const _Diff _Count, _BinOp _Reduce_op_, _Init_ty& _Initial_) : _Team{_Count, _Get_chunked_work_chunk_count(_Hw_threads, _Count)}, _Basis1{}, _Basis2{}, _Lookback(_Team._Chunks), _Intermediate_result{_Team}, _Reduce_op(_Reduce_op_), _Initial(_Initial_) {} @@ -4643,7 +4643,7 @@ struct _Static_partitioned_inclusive_scan2 { static void __stdcall _Threadpool_callback( __std_PTP_CALLBACK_INSTANCE, void* const _Context, __std_PTP_WORK) noexcept /* terminates */ { - _STD _Run_available_chunked_work(*static_cast<_Static_partitioned_inclusive_scan2*>(_Context)); + _STD _Run_available_chunked_work(*static_cast<_Static_partitioned_inclusive_scan3*>(_Context)); } }; @@ -4665,7 +4665,7 @@ _FwdIt2 inclusive_scan(_ExPo&&, _FwdIt1 _First, _FwdIt1 _Last, _FwdIt2 _Dest, _B if (_Count >= 2) { // ... with at least 2 elements _TRY_BEGIN auto _Passed_op = _STD _Pass_fn(_Reduce_op); - _Static_partitioned_inclusive_scan2<_Ty, _Ty, _Unwrapped_t, decltype(_UDest), + _Static_partitioned_inclusive_scan3<_Ty, _Ty, _Unwrapped_t, decltype(_UDest), decltype(_Passed_op)> _Operation{_Hw_threads, _Count, _Passed_op, _Val}; _Operation._Basis1._Populate(_Operation._Team, _UFirst); @@ -4711,7 +4711,7 @@ _FwdIt2 inclusive_scan(_ExPo&&, _FwdIt1 _First, _FwdIt1 _Last, _FwdIt2 _Dest, _B _TRY_BEGIN _No_init_tag _Tag; auto _Passed_op = _STD _Pass_fn(_Reduce_op); - _Static_partitioned_inclusive_scan2<_Iter_value_t<_FwdIt1>, _No_init_tag, _Unwrapped_t, + _Static_partitioned_inclusive_scan3<_Iter_value_t<_FwdIt1>, _No_init_tag, _Unwrapped_t, decltype(_UDest), decltype(_Passed_op)> _Operation{_Hw_threads, _Count, _Passed_op, _Tag}; _Operation._Basis1._Populate(_Operation._Team, _UFirst); @@ -4777,7 +4777,7 @@ void _Transform_exclusive_scan_per_chunk_complete(_FwdIt1 _First, const _FwdIt1 } template -struct _Static_partitioned_transform_exclusive_scan2 { +struct _Static_partitioned_transform_exclusive_scan3 { using _Diff = _Common_diff_t<_FwdIt1, _FwdIt2>; _Static_partition_team<_Diff> _Team; _Static_partition_range<_FwdIt1, _Diff> _Basis1; @@ -4788,7 +4788,7 @@ struct _Static_partitioned_transform_exclusive_scan2 { _BinOp _Reduce_op; _UnaryOp _Transform_op; - _Static_partitioned_transform_exclusive_scan2(const size_t _Hw_threads, const _Diff _Count, const _FwdIt1 _First, + _Static_partitioned_transform_exclusive_scan3(const size_t _Hw_threads, const _Diff _Count, const _FwdIt1 _First, _Ty& _Initial_, _BinOp _Reduce_op_, _UnaryOp _Transform_op_, const _FwdIt2&) : _Team{_Count, _Get_chunked_work_chunk_count(_Hw_threads, _Count)}, _Basis1{}, _Basis2{}, _Lookback(_Team._Chunks), _Intermediate_result{_Team}, _Initial(_Initial_), _Reduce_op(_Reduce_op_), @@ -4846,7 +4846,7 @@ struct _Static_partitioned_transform_exclusive_scan2 { static void __stdcall _Threadpool_callback( __std_PTP_CALLBACK_INSTANCE, void* const _Context, __std_PTP_WORK) noexcept /* terminates */ { - _STD _Run_available_chunked_work(*static_cast<_Static_partitioned_transform_exclusive_scan2*>(_Context)); + _STD _Run_available_chunked_work(*static_cast<_Static_partitioned_transform_exclusive_scan3*>(_Context)); } }; @@ -4867,7 +4867,7 @@ _FwdIt2 transform_exclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _L const auto _UDest = _STD _Get_unwrapped_n(_Dest, _Count); if (_Count >= 2) { // ... with at least 2 elements _TRY_BEGIN - _Static_partitioned_transform_exclusive_scan2 _Operation{_Hw_threads, _Count, _UFirst, _Val, + _Static_partitioned_transform_exclusive_scan3 _Operation{_Hw_threads, _Count, _UFirst, _Val, _STD _Pass_fn(_Reduce_op), _STD _Pass_fn(_Transform_op), _UDest}; _STD _Seek_wrapped(_Dest, _Operation._Basis2._Populate(_Operation._Team, _UDest)); // Note that _Val is used as temporary storage by whichever thread runs the first chunk. @@ -4923,7 +4923,7 @@ _FwdIt2 _Transform_inclusive_scan_per_chunk(_FwdIt1 _First, const _FwdIt1 _Last, } template -struct _Static_partitioned_transform_inclusive_scan2 { +struct _Static_partitioned_transform_inclusive_scan3 { using _Diff = _Common_diff_t<_FwdIt1, _FwdIt2>; _Static_partition_team<_Diff> _Team; _Static_partition_range<_FwdIt1, _Diff> _Basis1; @@ -4934,7 +4934,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { _UnaryOp _Transform_op; _Init_ty& _Initial; - _Static_partitioned_transform_inclusive_scan2( + _Static_partitioned_transform_inclusive_scan3( const size_t _Hw_threads, const _Diff _Count, _BinOp _Reduce_op_, _UnaryOp _Transform_op_, _Init_ty& _Initial_) : _Team{_Count, _Get_chunked_work_chunk_count(_Hw_threads, _Count)}, _Basis1{}, _Basis2{}, _Lookback(_Team._Chunks), _Intermediate_result{_Team}, _Reduce_op(_Reduce_op_), _Transform_op(_Transform_op_), @@ -4990,7 +4990,7 @@ struct _Static_partitioned_transform_inclusive_scan2 { static void __stdcall _Threadpool_callback( __std_PTP_CALLBACK_INSTANCE, void* const _Context, __std_PTP_WORK) noexcept /* terminates */ { - _STD _Run_available_chunked_work(*static_cast<_Static_partitioned_transform_inclusive_scan2*>(_Context)); + _STD _Run_available_chunked_work(*static_cast<_Static_partitioned_transform_inclusive_scan3*>(_Context)); } }; @@ -5013,7 +5013,7 @@ _FwdIt2 transform_inclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _L _TRY_BEGIN auto _Passed_reduce = _STD _Pass_fn(_Reduce_op); auto _Passed_transform = _STD _Pass_fn(_Transform_op); - _Static_partitioned_transform_inclusive_scan2<_Ty, _Ty, _Unwrapped_t, decltype(_UDest), + _Static_partitioned_transform_inclusive_scan3<_Ty, _Ty, _Unwrapped_t, decltype(_UDest), decltype(_Passed_reduce), decltype(_Passed_transform)> _Operation{_Hw_threads, _Count, _Passed_reduce, _Passed_transform, _Val}; _Operation._Basis1._Populate(_Operation._Team, _UFirst); @@ -5061,7 +5061,7 @@ _FwdIt2 transform_inclusive_scan(_ExPo&&, const _FwdIt1 _First, const _FwdIt1 _L auto _Passed_reduce = _STD _Pass_fn(_Reduce_op); auto _Passed_transform = _STD _Pass_fn(_Transform_op); using _Intermediate_t = decay_t; - _Static_partitioned_transform_inclusive_scan2<_Intermediate_t, _No_init_tag, + _Static_partitioned_transform_inclusive_scan3<_Intermediate_t, _No_init_tag, _Unwrapped_t, decltype(_UDest), decltype(_Passed_reduce), decltype(_Passed_transform)> _Operation{_Hw_threads, _Count, _Passed_reduce, _Passed_transform, _Tag};