#include #include #include #define CUDA_CHECK(call) \ do { \ const cudaError_t status = (call); \ if (status != cudaSuccess) { \ std::fprintf(stderr, "%s:%d CUDA error: %s\n", __FILE__, __LINE__, \ cudaGetErrorString(status)); \ std::exit(EXIT_FAILURE); \ } \ } while (0) __global__ void add_one(float* values, int n) { const int i = blockIdx.x * blockDim.x + threadIdx.x; if (i < n) { values[i] += 1.0F; } } int main() { constexpr int n = 1 << 20; constexpr size_t bytes = n * sizeof(float); int device = 0; cudaDeviceProp properties{}; CUDA_CHECK(cudaGetDevice(&device)); CUDA_CHECK(cudaGetDeviceProperties(&properties, device)); cudaStream_t stream{}; cudaEvent_t start{}, stop{}; CUDA_CHECK(cudaStreamCreateWithFlags(&stream, cudaStreamNonBlocking)); CUDA_CHECK(cudaEventCreate(&start)); CUDA_CHECK(cudaEventCreate(&stop)); // cudaMallocAsync uses the device's default stream-ordered memory pool. float* values = nullptr; CUDA_CHECK(cudaMallocAsync(&values, bytes, stream)); CUDA_CHECK(cudaMemsetAsync(values, 0, bytes, stream)); cudaGraph_t graph{}; cudaGraphExec_t executable{}; CUDA_CHECK(cudaStreamBeginCapture(stream, cudaStreamCaptureModeGlobal)); add_one<<<(n + 255) / 256, 256, 0, stream>>>(values, n); CUDA_CHECK(cudaGetLastError()); CUDA_CHECK(cudaStreamEndCapture(stream, &graph)); CUDA_CHECK(cudaGraphInstantiate(&executable, graph, nullptr, nullptr, 0)); CUDA_CHECK(cudaEventRecord(start, stream)); CUDA_CHECK(cudaGraphLaunch(executable, stream)); CUDA_CHECK(cudaEventRecord(stop, stream)); CUDA_CHECK(cudaEventSynchronize(stop)); float elapsed_ms = 0.0F; float first = 0.0F; CUDA_CHECK(cudaEventElapsedTime(&elapsed_ms, start, stop)); CUDA_CHECK(cudaMemcpyAsync(&first, values, sizeof(first), cudaMemcpyDeviceToHost, stream)); CUDA_CHECK(cudaStreamSynchronize(stream)); std::printf( "gpu=%s compute_capability=%d.%d graph_kernel_ms=%.6f first=%.1f\n", properties.name, properties.major, properties.minor, elapsed_ms, first); CUDA_CHECK(cudaFreeAsync(values, stream)); CUDA_CHECK(cudaStreamSynchronize(stream)); CUDA_CHECK(cudaGraphExecDestroy(executable)); CUDA_CHECK(cudaGraphDestroy(graph)); CUDA_CHECK(cudaEventDestroy(stop)); CUDA_CHECK(cudaEventDestroy(start)); CUDA_CHECK(cudaStreamDestroy(stream)); return first == 1.0F ? EXIT_SUCCESS : EXIT_FAILURE; }