Apply .clangformat to all repo source files
Change-Id: I7e79c6058f0303f9a98911e3b7dd2e8596079344
[ROCm/hip commit: 1ba06f63c4]
This commit is contained in:
@@ -26,8 +26,8 @@ THE SOFTWARE.
|
||||
// will automatically copy data to and from the host, without the user needing
|
||||
// to manually perform such copies. This is an excellent mode for developers
|
||||
// new to GPU programming and matches the memory models provided by recent systems where
|
||||
// CPU and GPU share the same memory pool. Advanced programmers may prefer
|
||||
// more explicit control over the data movement - shown in the other vadd_hc_array and
|
||||
// CPU and GPU share the same memory pool. Advanced programmers may prefer
|
||||
// more explicit control over the data movement - shown in the other vadd_hc_array and
|
||||
// vadd_hc_am examples.
|
||||
// This example shows the similarity between C++AMP and and HC for simple cases where
|
||||
// implicit data transfer is used - really the only difference is the namespace.
|
||||
@@ -35,8 +35,7 @@ THE SOFTWARE.
|
||||
|
||||
#include <amp.h>
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
int main(int argc, char* argv[]) {
|
||||
int sizeElements = 1000000;
|
||||
bool pass = true;
|
||||
|
||||
@@ -46,28 +45,30 @@ int main(int argc, char *argv[])
|
||||
concurrency::array_view<float> C(sizeElements);
|
||||
|
||||
// Initialize host data
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
A[i] = 1.618f * i;
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
A[i] = 1.618f * i;
|
||||
B[i] = 3.142f * i;
|
||||
}
|
||||
C.discard_data(); // tell runtime not to copy CPU host data.
|
||||
C.discard_data(); // tell runtime not to copy CPU host data.
|
||||
|
||||
|
||||
// Launch kernel onto default accelerator
|
||||
// The HCC runtime will ensure that A and B are available on the accelerator before launching the kernel.
|
||||
concurrency::parallel_for_each(concurrency::extent<1> (sizeElements),
|
||||
[=] (concurrency::index<1> idx) restrict(amp) {
|
||||
int i = idx[0];
|
||||
C[i] = A[i] + B[i];
|
||||
});
|
||||
// The HCC runtime will ensure that A and B are available on the accelerator before launching
|
||||
// the kernel.
|
||||
concurrency::parallel_for_each(concurrency::extent<1>(sizeElements),
|
||||
[=](concurrency::index<1> idx) restrict(amp) {
|
||||
int i = idx[0];
|
||||
C[i] = A[i] + B[i];
|
||||
});
|
||||
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
float ref= 1.618f * i + 3.142f * i;
|
||||
// Because C is an array_view, the HCC runtime will copy C back to host at first access here:
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
float ref = 1.618f * i + 3.142f * i;
|
||||
// Because C is an array_view, the HCC runtime will copy C back to host at first access
|
||||
// here:
|
||||
if (C[i] != ref) {
|
||||
printf ("error:%d computed=%6.2f, reference=%6.2f\n", i, C[i], ref);
|
||||
printf("error:%d computed=%6.2f, reference=%6.2f\n", i, C[i], ref);
|
||||
pass = false;
|
||||
}
|
||||
};
|
||||
if (pass) printf ("PASSED!\n");
|
||||
if (pass) printf("PASSED!\n");
|
||||
}
|
||||
|
||||
@@ -24,21 +24,20 @@ THE SOFTWARE.
|
||||
// AM provides a set of c-style memory management routines for allocating,
|
||||
// freeing, and copying memory. am_alloc returns a device pointer
|
||||
// which can only be used on the device. The programmer has full control
|
||||
// over when data is copied.
|
||||
// over when data is copied.
|
||||
|
||||
#include <hc.hpp>
|
||||
#include <hc_am.hpp>
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
int main(int argc, char* argv[]) {
|
||||
int sizeElements = 1000000;
|
||||
size_t sizeBytes = sizeElements * sizeof(float);
|
||||
bool pass = true;
|
||||
|
||||
// Allocate host memory
|
||||
float *A_h = (float*)malloc(sizeBytes);
|
||||
float *B_h = (float*)malloc(sizeBytes);
|
||||
float *C_h = (float*)malloc(sizeBytes);
|
||||
float* A_h = (float*)malloc(sizeBytes);
|
||||
float* B_h = (float*)malloc(sizeBytes);
|
||||
float* C_h = (float*)malloc(sizeBytes);
|
||||
|
||||
// Allocate device pointers:
|
||||
// Unlike array_view, these must be explicitly managed by user:
|
||||
@@ -51,36 +50,37 @@ int main(int argc, char *argv[])
|
||||
C_d = hc::am_alloc(sizeBytes, acc, 0);
|
||||
|
||||
// Initialize host data
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
A_h[i] = 1.618f * i;
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
A_h[i] = 1.618f * i;
|
||||
B_h[i] = 3.142f * i;
|
||||
C_h[i] = 0;
|
||||
}
|
||||
|
||||
av.copy(A_h, A_d, sizeBytes); // C++ copy H2D
|
||||
av.copy(B_h, B_d, sizeBytes); // C++ copy H2D
|
||||
av.copy(A_h, A_d, sizeBytes); // C++ copy H2D
|
||||
av.copy(B_h, B_d, sizeBytes); // C++ copy H2D
|
||||
|
||||
// Launch kernel onto AV.
|
||||
// Launch kernel onto AV.
|
||||
// Because the kernel PFE and the copies are submitted to same AV, they will execute in order
|
||||
// and we don't need additional synchronization to ensure the copies complete before the PFE begins.
|
||||
hc::completion_future cf=
|
||||
hc::parallel_for_each(av, hc::extent<1> (sizeElements),
|
||||
[=] (hc::index<1> idx) [[hc]] {
|
||||
int i = idx[0];
|
||||
C_d[i] = A_d[i] + B_d[i];
|
||||
});
|
||||
|
||||
|
||||
// This copy is in same AV as the kernel and thus will wait for the kernel to finish before executing.
|
||||
av.copy(C_d, C_h, sizeBytes); // C++ copy D2H
|
||||
// and we don't need additional synchronization to ensure the copies complete before the PFE
|
||||
// begins.
|
||||
hc::completion_future cf =
|
||||
hc::parallel_for_each(av, hc::extent<1>(sizeElements), [=](hc::index<1> idx)[[hc]] {
|
||||
int i = idx[0];
|
||||
C_d[i] = A_d[i] + B_d[i];
|
||||
});
|
||||
|
||||
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
float ref= 1.618f * i + 3.142f * i;
|
||||
// This copy is in same AV as the kernel and thus will wait for the kernel to finish before
|
||||
// executing.
|
||||
av.copy(C_d, C_h, sizeBytes); // C++ copy D2H
|
||||
|
||||
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
float ref = 1.618f * i + 3.142f * i;
|
||||
if (C_h[i] != ref) {
|
||||
printf ("error:%d computed=%6.2f, reference=%6.2f\n", i, C_h[i], ref);
|
||||
printf("error:%d computed=%6.2f, reference=%6.2f\n", i, C_h[i], ref);
|
||||
pass = false;
|
||||
}
|
||||
};
|
||||
if (pass) printf ("PASSED!\n");
|
||||
if (pass) printf("PASSED!\n");
|
||||
}
|
||||
|
||||
@@ -21,7 +21,7 @@ THE SOFTWARE.
|
||||
*/
|
||||
|
||||
// Simple test showing how to use HC syntax with array.
|
||||
// Array provides a type-safe C++ mechanism to allocate accelerator memory.
|
||||
// Array provides a type-safe C++ mechanism to allocate accelerator memory.
|
||||
// Like array_view, hc::array provides multi-dimensional indexing capability,
|
||||
// and is typed. However, unlike array_view, hc::array does not provide
|
||||
// automatic data management capabilities - instead the programmer
|
||||
@@ -29,16 +29,15 @@ THE SOFTWARE.
|
||||
|
||||
#include <hc.hpp>
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
int main(int argc, char* argv[]) {
|
||||
int sizeElements = 1000000;
|
||||
size_t sizeBytes = sizeElements * sizeof(float);
|
||||
bool pass = true;
|
||||
|
||||
// Allocate host memory
|
||||
float *A_h = (float*)malloc(sizeBytes);
|
||||
float *B_h = (float*)malloc(sizeBytes);
|
||||
float *C_h = (float*)malloc(sizeBytes);
|
||||
float* A_h = (float*)malloc(sizeBytes);
|
||||
float* B_h = (float*)malloc(sizeBytes);
|
||||
float* C_h = (float*)malloc(sizeBytes);
|
||||
|
||||
// Allocate device arrays<>
|
||||
// Unlike array_view, these must be explicitly managed by user:
|
||||
@@ -47,32 +46,32 @@ int main(int argc, char *argv[])
|
||||
hc::array<float> C_d(sizeElements);
|
||||
|
||||
// Initialize host data
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
A_h[i] = 1.618f * i;
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
A_h[i] = 1.618f * i;
|
||||
B_h[i] = 3.142f * i;
|
||||
}
|
||||
|
||||
hc::copy(A_h, A_d); // C++ copy H2D
|
||||
hc::copy(B_h, B_d); // C++ copy H2D
|
||||
hc::copy(A_h, A_d); // C++ copy H2D
|
||||
hc::copy(B_h, B_d); // C++ copy H2D
|
||||
|
||||
// Launch kernel onto default accelerator:
|
||||
// array<> types are not implicitly copied, so we performed copies above.
|
||||
hc::parallel_for_each(hc::extent<1> (sizeElements),
|
||||
[&] (hc::index<1> idx) [[hc]] {
|
||||
hc::parallel_for_each(hc::extent<1>(sizeElements), [&](hc::index<1> idx)[[hc]] {
|
||||
int i = idx[0];
|
||||
C_d[i] = A_d[i] + B_d[i];
|
||||
});
|
||||
|
||||
// HCC runtime knows that C_d depends on previous PFE and will force the copy to wait for the PFE to complte.
|
||||
hc::copy(C_d, C_h); // C++ copy D2H
|
||||
// HCC runtime knows that C_d depends on previous PFE and will force the copy to wait for the
|
||||
// PFE to complte.
|
||||
hc::copy(C_d, C_h); // C++ copy D2H
|
||||
|
||||
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
float ref= 1.618f * i + 3.142f * i;
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
float ref = 1.618f * i + 3.142f * i;
|
||||
if (C_h[i] != ref) {
|
||||
printf ("error:%d computed=%6.2f, reference=%6.2f\n", i, C_h[i], ref);
|
||||
printf("error:%d computed=%6.2f, reference=%6.2f\n", i, C_h[i], ref);
|
||||
pass = false;
|
||||
}
|
||||
};
|
||||
if (pass) printf ("PASSED!\n");
|
||||
if (pass) printf("PASSED!\n");
|
||||
}
|
||||
|
||||
@@ -26,8 +26,8 @@ THE SOFTWARE.
|
||||
// will automatically copy data to and from the host, without the user needing
|
||||
// to manually perform such copies. This is an excellent mode for developers
|
||||
// new to GPU programming and matches the memory models provided by recent systems where
|
||||
// CPU and GPU share the same memory pool. Advanced programmers may prefer
|
||||
// more explicit control over the data movement - shown in the other vadd_hc_array and
|
||||
// CPU and GPU share the same memory pool. Advanced programmers may prefer
|
||||
// more explicit control over the data movement - shown in the other vadd_hc_array and
|
||||
// vadd_hc_am examples.
|
||||
// This example shows the similarity between C++AMP and and HC for simple cases where
|
||||
// implicit data transfer is used - really the only difference is the namespace.
|
||||
@@ -35,8 +35,7 @@ THE SOFTWARE.
|
||||
|
||||
#include <hc.hpp>
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
int main(int argc, char* argv[]) {
|
||||
int sizeElements = 1000000;
|
||||
bool pass = true;
|
||||
|
||||
@@ -46,28 +45,29 @@ int main(int argc, char *argv[])
|
||||
hc::array_view<float> C(sizeElements);
|
||||
|
||||
// Initialize host data
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
A[i] = 1.618f * i;
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
A[i] = 1.618f * i;
|
||||
B[i] = 3.142f * i;
|
||||
}
|
||||
C.discard_data(); // tell runtime not to copy CPU host data.
|
||||
C.discard_data(); // tell runtime not to copy CPU host data.
|
||||
|
||||
|
||||
// Launch kernel onto default accelerator:
|
||||
// The HCC runtime will ensure that A and B are available on the accelerator before launching the kernel.
|
||||
hc::parallel_for_each(hc::extent<1> (sizeElements),
|
||||
[=] (hc::index<1> idx) [[hc]] {
|
||||
// The HCC runtime will ensure that A and B are available on the accelerator before launching
|
||||
// the kernel.
|
||||
hc::parallel_for_each(hc::extent<1>(sizeElements), [=](hc::index<1> idx)[[hc]] {
|
||||
int i = idx[0];
|
||||
C[i] = A[i] + B[i];
|
||||
});
|
||||
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
float ref= 1.618f * i + 3.142f * i;
|
||||
// Because C is an array_view, the HCC runtime will copy C back to host at first access here:
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
float ref = 1.618f * i + 3.142f * i;
|
||||
// Because C is an array_view, the HCC runtime will copy C back to host at first access
|
||||
// here:
|
||||
if (C[i] != ref) {
|
||||
printf ("error:%d computed=%6.2f, reference=%6.2f\n", i, C[i], ref);
|
||||
printf("error:%d computed=%6.2f, reference=%6.2f\n", i, C[i], ref);
|
||||
pass = false;
|
||||
}
|
||||
};
|
||||
if (pass) printf ("PASSED!\n");
|
||||
if (pass) printf("PASSED!\n");
|
||||
}
|
||||
|
||||
@@ -22,8 +22,7 @@ THE SOFTWARE.
|
||||
|
||||
#include "hip/hip_runtime.h"
|
||||
|
||||
__global__ void vadd_hip(hipLaunchParm lp, const float *a, const float *b, float *c, int N)
|
||||
{
|
||||
__global__ void vadd_hip(hipLaunchParm lp, const float* a, const float* b, float* c, int N) {
|
||||
int idx = (hipBlockIdx_x * hipBlockDim_x + hipThreadIdx_x);
|
||||
|
||||
if (idx < N) {
|
||||
@@ -32,16 +31,15 @@ __global__ void vadd_hip(hipLaunchParm lp, const float *a, const float *b, float
|
||||
}
|
||||
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
int main(int argc, char* argv[]) {
|
||||
int sizeElements = 1000000;
|
||||
size_t sizeBytes = sizeElements * sizeof(float);
|
||||
bool pass = true;
|
||||
|
||||
// Allocate host memory
|
||||
float *A_h = (float*)malloc(sizeBytes);
|
||||
float *B_h = (float*)malloc(sizeBytes);
|
||||
float *C_h = (float*)malloc(sizeBytes);
|
||||
float* A_h = (float*)malloc(sizeBytes);
|
||||
float* B_h = (float*)malloc(sizeBytes);
|
||||
float* C_h = (float*)malloc(sizeBytes);
|
||||
|
||||
// Allocate device memory:
|
||||
float *A_d, *B_d, *C_d;
|
||||
@@ -50,8 +48,8 @@ int main(int argc, char *argv[])
|
||||
hipMalloc(&C_d, sizeBytes);
|
||||
|
||||
// Initialize host memory
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
A_h[i] = 1.618f * i;
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
A_h[i] = 1.618f * i;
|
||||
B_h[i] = 3.142f * i;
|
||||
}
|
||||
|
||||
@@ -60,20 +58,20 @@ int main(int argc, char *argv[])
|
||||
hipMemcpy(B_d, B_h, sizeBytes, hipMemcpyHostToDevice);
|
||||
|
||||
// Launch kernel onto default accelerator
|
||||
int blockSize = 256; // pick arbitrary block size
|
||||
int blocks = (sizeElements+blockSize-1)/blockSize; // round up to launch enough blocks
|
||||
int blockSize = 256; // pick arbitrary block size
|
||||
int blocks = (sizeElements + blockSize - 1) / blockSize; // round up to launch enough blocks
|
||||
hipLaunchKernel(vadd_hip, dim3(blocks), dim3(blockSize), 0, 0, A_d, B_d, C_d, sizeElements);
|
||||
|
||||
// D2H Copy
|
||||
hipMemcpy(C_h, C_d, sizeBytes, hipMemcpyDeviceToHost);
|
||||
|
||||
// Verify
|
||||
for (int i=0; i<sizeElements; i++) {
|
||||
float ref= 1.618f * i + 3.142f * i;
|
||||
for (int i = 0; i < sizeElements; i++) {
|
||||
float ref = 1.618f * i + 3.142f * i;
|
||||
if (C_h[i] != ref) {
|
||||
printf ("error:%d computed=%6.2f, reference=%6.2f\n", i, C_h[i], ref);
|
||||
printf("error:%d computed=%6.2f, reference=%6.2f\n", i, C_h[i], ref);
|
||||
pass = false;
|
||||
}
|
||||
};
|
||||
if (pass) printf ("PASSED!\n");
|
||||
if (pass) printf("PASSED!\n");
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user