#include #include #include #include #include #include #include namespace cg = cooperative_groups; __global__ void random_and_atomic(float* values, unsigned long long seed) { const int i = blockIdx.x * blockDim.x + threadIdx.x; curandStatePhilox4_32_10_t state; curand_init(seed, i, 0, &state); values[i] = curand_uniform(&state); cg::thread_block block = cg::this_thread_block(); block.sync(); cuda::atomic_ref first(values[0]); if (i != 0) { first.fetch_add(values[i], cuda::memory_order_relaxed); } } int main() { constexpr int n = 256; thrust::device_vector values(n, 0.0F); random_and_atomic<<<1, n>>>(thrust::raw_pointer_cast(values.data()), 7); const float thrust_sum = thrust::reduce(values.begin(), values.end()); std::printf("thrust_sum_with_atomic_first=%f\n", thrust_sum); // CUB is included and compiled here; its device-wide primitives should be // exercised in a dedicated reduction receipt rather than conflated with // the Thrust result above. }