unsigned int tid = threadIdx.x;
unsigned int i = blockIdx.x * blockDim.x * 4 + tid;
__shared__ float sharedData[BLOCK_SIZE];
sharedData[tid] = 0.0f;
// Phase 1: Global-to-Shared Accumulation
for (int j = 0; j < 4; ++j) {
sharedData[tid] += input[i + j * blockDim.x];
}
__syncthreads();
// Phase 2: Tree Reduction (Shrinking Stride)
for (int stride = blockDim.x / 2; stride > 0; stride /= 2) {
if (tid < stride) {
sharedData[tid] += sharedData[tid + stride];
}
__syncthreads();
}
if (tid == 0) partialSums[blockIdx.x] = sharedData[0];