Sync HIP documentation 2025-10-20 (#1258)
* Add examples to tools folder * Correct P2P memory access section * Sync poriting guide * Add HIP Graph tutorial * Add hint about using amdgpu-dkms for IPC API * Add a few more env variables
This commit is contained in:
committed by
GitHub
orang tua
8e98b80deb
melakukan
197f73dac9
@@ -207,319 +207,24 @@ The example codes
|
||||
|
||||
.. tab-item:: Sequential
|
||||
|
||||
.. code-block:: cpp
|
||||
|
||||
#include <hip/hip_runtime.h>
|
||||
#include <vector>
|
||||
#include <iostream>
|
||||
|
||||
#define HIP_CHECK(expression) \
|
||||
{ \
|
||||
const hipError_t status = expression; \
|
||||
if(status != hipSuccess){ \
|
||||
std::cerr << "HIP error " \
|
||||
<< status << ": " \
|
||||
<< hipGetErrorString(status) \
|
||||
<< " at " << __FILE__ << ":" \
|
||||
<< __LINE__ << std::endl; \
|
||||
} \
|
||||
}
|
||||
|
||||
// GPU Kernels
|
||||
__global__ void kernelA(double* arrayA, size_t size){
|
||||
const size_t x = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if(x < size){arrayA[x] += 1.0;}
|
||||
};
|
||||
__global__ void kernelB(double* arrayA, double* arrayB, size_t size){
|
||||
const size_t x = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if(x < size){arrayB[x] += arrayA[x] + 3.0;}
|
||||
};
|
||||
|
||||
int main()
|
||||
{
|
||||
constexpr int numOfBlocks = 1 << 20;
|
||||
constexpr int threadsPerBlock = 1024;
|
||||
constexpr int numberOfIterations = 50;
|
||||
// The array size smaller to avoid the relatively short kernel launch compared to memory copies
|
||||
constexpr size_t arraySize = 1U << 25;
|
||||
double *d_dataA;
|
||||
double *d_dataB;
|
||||
|
||||
double initValueA = 0.0;
|
||||
double initValueB = 2.0;
|
||||
|
||||
std::vector<double> vectorA(arraySize, initValueA);
|
||||
std::vector<double> vectorB(arraySize, initValueB);
|
||||
// Allocate device memory
|
||||
HIP_CHECK(hipMalloc(&d_dataA, arraySize * sizeof(*d_dataA)));
|
||||
HIP_CHECK(hipMalloc(&d_dataB, arraySize * sizeof(*d_dataB)));
|
||||
for(int iteration = 0; iteration < numberOfIterations; iteration++)
|
||||
{
|
||||
// Host to Device copies
|
||||
HIP_CHECK(hipMemcpy(d_dataA, vectorA.data(), arraySize * sizeof(*d_dataA), hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(d_dataB, vectorB.data(), arraySize * sizeof(*d_dataB), hipMemcpyHostToDevice));
|
||||
// Launch the GPU kernels
|
||||
hipLaunchKernelGGL(kernelA, dim3(numOfBlocks), dim3(threadsPerBlock), 0, 0, d_dataA, arraySize);
|
||||
hipLaunchKernelGGL(kernelB, dim3(numOfBlocks), dim3(threadsPerBlock), 0, 0, d_dataA, d_dataB, arraySize);
|
||||
// Device to Host copies
|
||||
HIP_CHECK(hipMemcpy(vectorA.data(), d_dataA, arraySize * sizeof(*vectorA.data()), hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipMemcpy(vectorB.data(), d_dataB, arraySize * sizeof(*vectorB.data()), hipMemcpyDeviceToHost));
|
||||
}
|
||||
// Wait for all operations to complete
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
// Verify results
|
||||
const double expectedA = (double)numberOfIterations;
|
||||
const double expectedB =
|
||||
initValueB + (3.0 * numberOfIterations) +
|
||||
(expectedA * (expectedA + 1.0)) / 2.0;
|
||||
bool passed = true;
|
||||
for(size_t i = 0; i < arraySize; ++i){
|
||||
if(vectorA[i] != expectedA){
|
||||
passed = false;
|
||||
std::cerr << "Validation failed! Expected " << expectedA << " got " << vectorA[i] << " at index: " << i << std::endl;
|
||||
break;
|
||||
}
|
||||
if(vectorB[i] != expectedB){
|
||||
passed = false;
|
||||
std::cerr << "Validation failed! Expected " << expectedB << " got " << vectorB[i] << " at index: " << i << std::endl;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if(passed){
|
||||
std::cout << "Sequential execution completed successfully." << std::endl;
|
||||
}else{
|
||||
std::cerr << "Sequential execution failed." << std::endl;
|
||||
}
|
||||
|
||||
// Cleanup
|
||||
HIP_CHECK(hipFree(d_dataA));
|
||||
HIP_CHECK(hipFree(d_dataB));
|
||||
|
||||
return 0;
|
||||
}
|
||||
.. literalinclude:: ../../tools/example_codes/sequential_kernel_execution.hip
|
||||
:start-after: // [sphinx-start]
|
||||
:end-before: // [sphinx-end]
|
||||
:language: cpp
|
||||
|
||||
.. tab-item:: Asynchronous
|
||||
|
||||
.. code-block:: cpp
|
||||
|
||||
#include <hip/hip_runtime.h>
|
||||
#include <vector>
|
||||
#include <iostream>
|
||||
|
||||
#define HIP_CHECK(expression) \
|
||||
{ \
|
||||
const hipError_t status = expression; \
|
||||
if(status != hipSuccess){ \
|
||||
std::cerr << "HIP error " \
|
||||
<< status << ": " \
|
||||
<< hipGetErrorString(status) \
|
||||
<< " at " << __FILE__ << ":" \
|
||||
<< __LINE__ << std::endl; \
|
||||
} \
|
||||
}
|
||||
|
||||
// GPU Kernels
|
||||
__global__ void kernelA(double* arrayA, size_t size){
|
||||
const size_t x = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if(x < size){arrayA[x] += 1.0;}
|
||||
};
|
||||
__global__ void kernelB(double* arrayA, double* arrayB, size_t size){
|
||||
const size_t x = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if(x < size){arrayB[x] += arrayA[x] + 3.0;}
|
||||
};
|
||||
|
||||
int main()
|
||||
{
|
||||
constexpr int numOfBlocks = 1 << 20;
|
||||
constexpr int threadsPerBlock = 1024;
|
||||
constexpr int numberOfIterations = 50;
|
||||
// The array size smaller to avoid the relatively short kernel launch compared to memory copies
|
||||
constexpr size_t arraySize = 1U << 25;
|
||||
double *d_dataA;
|
||||
double *d_dataB;
|
||||
|
||||
double initValueA = 0.0;
|
||||
double initValueB = 2.0;
|
||||
|
||||
std::vector<double> vectorA(arraySize, initValueA);
|
||||
std::vector<double> vectorB(arraySize, initValueB);
|
||||
// Allocate device memory
|
||||
HIP_CHECK(hipMalloc(&d_dataA, arraySize * sizeof(*d_dataA)));
|
||||
HIP_CHECK(hipMalloc(&d_dataB, arraySize * sizeof(*d_dataB)));
|
||||
// Create streams
|
||||
hipStream_t streamA, streamB;
|
||||
HIP_CHECK(hipStreamCreate(&streamA));
|
||||
HIP_CHECK(hipStreamCreate(&streamB));
|
||||
for(unsigned int iteration = 0; iteration < numberOfIterations; iteration++)
|
||||
{
|
||||
// Stream 1: Host to Device 1
|
||||
HIP_CHECK(hipMemcpyAsync(d_dataA, vectorA.data(), arraySize * sizeof(*d_dataA), hipMemcpyHostToDevice, streamA));
|
||||
// Stream 2: Host to Device 2
|
||||
HIP_CHECK(hipMemcpyAsync(d_dataB, vectorB.data(), arraySize * sizeof(*d_dataB), hipMemcpyHostToDevice, streamB));
|
||||
// Stream 1: Kernel 1
|
||||
hipLaunchKernelGGL(kernelA, dim3(numOfBlocks), dim3(threadsPerBlock), 0, streamA, d_dataA, arraySize);
|
||||
// Wait for streamA finish
|
||||
HIP_CHECK(hipStreamSynchronize(streamA));
|
||||
// Stream 2: Kernel 2
|
||||
hipLaunchKernelGGL(kernelB, dim3(numOfBlocks), dim3(threadsPerBlock), 0, streamB, d_dataA, d_dataB, arraySize);
|
||||
// Stream 1: Device to Host 2 (after Kernel 1)
|
||||
HIP_CHECK(hipMemcpyAsync(vectorA.data(), d_dataA, arraySize * sizeof(*vectorA.data()), hipMemcpyDeviceToHost, streamA));
|
||||
// Stream 2: Device to Host 2 (after Kernel 2)
|
||||
HIP_CHECK(hipMemcpyAsync(vectorB.data(), d_dataB, arraySize * sizeof(*vectorB.data()), hipMemcpyDeviceToHost, streamB));
|
||||
}
|
||||
// Wait for all operations in both streams to complete
|
||||
HIP_CHECK(hipStreamSynchronize(streamA));
|
||||
HIP_CHECK(hipStreamSynchronize(streamB));
|
||||
// Verify results
|
||||
double expectedA = (double)numberOfIterations;
|
||||
double expectedB =
|
||||
initValueB + (3.0 * numberOfIterations) +
|
||||
(expectedA * (expectedA + 1.0)) / 2.0;
|
||||
bool passed = true;
|
||||
for(size_t i = 0; i < arraySize; ++i){
|
||||
if(vectorA[i] != expectedA){
|
||||
passed = false;
|
||||
std::cerr << "Validation failed! Expected " << expectedA << " got " << vectorA[i] << " at index: " << i << std::endl;
|
||||
break;
|
||||
}
|
||||
if(vectorB[i] != expectedB){
|
||||
passed = false;
|
||||
std::cerr << "Validation failed! Expected " << expectedB << " got " << vectorB[i] << " at index: " << i << std::endl;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if(passed){
|
||||
std::cout << "Asynchronous execution completed successfully." << std::endl;
|
||||
}else{
|
||||
std::cerr << "Asynchronous execution failed." << std::endl;
|
||||
}
|
||||
|
||||
// Cleanup
|
||||
HIP_CHECK(hipStreamDestroy(streamA));
|
||||
HIP_CHECK(hipStreamDestroy(streamB));
|
||||
HIP_CHECK(hipFree(d_dataA));
|
||||
HIP_CHECK(hipFree(d_dataB));
|
||||
|
||||
return 0;
|
||||
}
|
||||
.. literalinclude:: ../../tools/example_codes/async_kernel_execution.hip
|
||||
:start-after: // [sphinx-start]
|
||||
:end-before: // [sphinx-end]
|
||||
:language: cpp
|
||||
|
||||
.. tab-item:: hipStreamWaitEvent
|
||||
|
||||
.. code-block:: cpp
|
||||
|
||||
#include <hip/hip_runtime.h>
|
||||
#include <vector>
|
||||
#include <iostream>
|
||||
|
||||
#define HIP_CHECK(expression) \
|
||||
{ \
|
||||
const hipError_t status = expression; \
|
||||
if(status != hipSuccess){ \
|
||||
std::cerr << "HIP error " \
|
||||
<< status << ": " \
|
||||
<< hipGetErrorString(status) \
|
||||
<< " at " << __FILE__ << ":" \
|
||||
<< __LINE__ << std::endl; \
|
||||
} \
|
||||
}
|
||||
|
||||
// GPU Kernels
|
||||
__global__ void kernelA(double* arrayA, size_t size){
|
||||
const size_t x = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if(x < size){arrayA[x] += 1.0;}
|
||||
};
|
||||
__global__ void kernelB(double* arrayA, double* arrayB, size_t size){
|
||||
const size_t x = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
if(x < size){arrayB[x] += arrayA[x] + 3.0;}
|
||||
};
|
||||
|
||||
int main()
|
||||
{
|
||||
constexpr int numOfBlocks = 1 << 20;
|
||||
constexpr int threadsPerBlock = 1024;
|
||||
constexpr int numberOfIterations = 50;
|
||||
// The array size smaller to avoid the relatively short kernel launch compared to memory copies
|
||||
constexpr size_t arraySize = 1U << 25;
|
||||
double *d_dataA;
|
||||
double *d_dataB;
|
||||
double initValueA = 0.0;
|
||||
double initValueB = 2.0;
|
||||
|
||||
std::vector<double> vectorA(arraySize, initValueA);
|
||||
std::vector<double> vectorB(arraySize, initValueB);
|
||||
// Allocate device memory
|
||||
HIP_CHECK(hipMalloc(&d_dataA, arraySize * sizeof(*d_dataA)));
|
||||
HIP_CHECK(hipMalloc(&d_dataB, arraySize * sizeof(*d_dataB)));
|
||||
// Create streams
|
||||
hipStream_t streamA, streamB;
|
||||
HIP_CHECK(hipStreamCreate(&streamA));
|
||||
HIP_CHECK(hipStreamCreate(&streamB));
|
||||
// Create events
|
||||
hipEvent_t event, eventA, eventB;
|
||||
HIP_CHECK(hipEventCreate(&event));
|
||||
HIP_CHECK(hipEventCreate(&eventA));
|
||||
HIP_CHECK(hipEventCreate(&eventB));
|
||||
for(unsigned int iteration = 0; iteration < numberOfIterations; iteration++)
|
||||
{
|
||||
// Stream 1: Host to Device 1
|
||||
HIP_CHECK(hipMemcpyAsync(d_dataA, vectorA.data(), arraySize * sizeof(*d_dataA), hipMemcpyHostToDevice, streamA));
|
||||
// Stream 2: Host to Device 2
|
||||
HIP_CHECK(hipMemcpyAsync(d_dataB, vectorB.data(), arraySize * sizeof(*d_dataB), hipMemcpyHostToDevice, streamB));
|
||||
// Stream 1: Kernel 1
|
||||
hipLaunchKernelGGL(kernelA, dim3(numOfBlocks), dim3(threadsPerBlock), 0, streamA, d_dataA, arraySize);
|
||||
// Record event after the GPU kernel in Stream 1
|
||||
HIP_CHECK(hipEventRecord(event, streamA));
|
||||
// Stream 2: Wait for event before starting Kernel 2
|
||||
HIP_CHECK(hipStreamWaitEvent(streamB, event, 0));
|
||||
// Stream 2: Kernel 2
|
||||
hipLaunchKernelGGL(kernelB, dim3(numOfBlocks), dim3(threadsPerBlock), 0, streamB, d_dataA, d_dataB, arraySize);
|
||||
// Stream 1: Device to Host 2 (after Kernel 1)
|
||||
HIP_CHECK(hipMemcpyAsync(vectorA.data(), d_dataA, arraySize * sizeof(*vectorA.data()), hipMemcpyDeviceToHost, streamA));
|
||||
// Stream 2: Device to Host 2 (after Kernel 2)
|
||||
HIP_CHECK(hipMemcpyAsync(vectorB.data(), d_dataB, arraySize * sizeof(*vectorB.data()), hipMemcpyDeviceToHost, streamB));
|
||||
// Wait for all operations in both streams to complete
|
||||
HIP_CHECK(hipEventRecord(eventA, streamA));
|
||||
HIP_CHECK(hipEventRecord(eventB, streamB));
|
||||
HIP_CHECK(hipStreamWaitEvent(streamA, eventA, 0));
|
||||
HIP_CHECK(hipStreamWaitEvent(streamB, eventB, 0));
|
||||
}
|
||||
// Verify results
|
||||
double expectedA = (double)numberOfIterations;
|
||||
double expectedB =
|
||||
initValueB + (3.0 * numberOfIterations) +
|
||||
(expectedA * (expectedA + 1.0)) / 2.0;
|
||||
bool passed = true;
|
||||
for(size_t i = 0; i < arraySize; ++i){
|
||||
if(vectorA[i] != expectedA){
|
||||
passed = false;
|
||||
std::cerr << "Validation failed! Expected " << expectedA << " got " << vectorA[i] << std::endl;
|
||||
break;
|
||||
}
|
||||
if(vectorB[i] != expectedB){
|
||||
passed = false;
|
||||
std::cerr << "Validation failed! Expected " << expectedB << " got " << vectorB[i] << std::endl;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if(passed){
|
||||
std::cout << "Asynchronous execution with events completed successfully." << std::endl;
|
||||
}else{
|
||||
std::cerr << "Asynchronous execution with events failed." << std::endl;
|
||||
}
|
||||
|
||||
// Cleanup
|
||||
HIP_CHECK(hipEventDestroy(event));
|
||||
HIP_CHECK(hipEventDestroy(eventA));
|
||||
HIP_CHECK(hipEventDestroy(eventB));
|
||||
HIP_CHECK(hipStreamDestroy(streamA));
|
||||
HIP_CHECK(hipStreamDestroy(streamB));
|
||||
HIP_CHECK(hipFree(d_dataA));
|
||||
HIP_CHECK(hipFree(d_dataB));
|
||||
|
||||
return 0;
|
||||
}
|
||||
.. literalinclude:: ../../tools/example_codes/event_based_synchronization.hip
|
||||
:start-after: // [sphinx-start]
|
||||
:end-before: // [sphinx-end]
|
||||
:language: cpp
|
||||
|
||||
HIP Graphs
|
||||
===============================================================================
|
||||
|
||||
Reference in New Issue
Block a user