SWDEV-420237 - Fix reduce sync operations when masks are divergent (#181)
Do not use __ockl_activelane_u32() to calculate the index of the lane within the mask, as that would not work with divergent masks that have other bits on before the associated lane.
This commit is contained in:
committed by
GitHub
parent
bf28bbd9ab
commit
1a8d766836
@@ -373,6 +373,7 @@ __device__ inline T __reduce_op_sync(MaskT mask, T val, BinaryOp op, WfReduce wf
|
||||
// the number of bits in the mask is a power of 2
|
||||
numIterations -= 1;
|
||||
|
||||
maskIdx = __popcll(((1ul << laneId) - 1) & mask);
|
||||
mask >>= laneId;
|
||||
mask >>= 1ul;
|
||||
|
||||
@@ -381,8 +382,6 @@ __device__ inline T __reduce_op_sync(MaskT mask, T val, BinaryOp op, WfReduce wf
|
||||
else
|
||||
__builtin_memcpy(&result, &val, sizeof(T));
|
||||
|
||||
maskIdx = __ockl_activelane_u32();
|
||||
|
||||
// add the values from the lanes using a reduction tree (first the threads with even-numbered
|
||||
// lanes, then multiples of 4, then 8, ...
|
||||
while (numIterations) {
|
||||
|
||||
Reference in New Issue
Block a user