SWDEV-470698 - fix formatting, add format check workflow (#657)

This commit is contained in:
Danylo Lytovchenko
2025-08-20 16:28:06 +02:00
committed by GitHub
orang tua 5840940caa
melakukan f7338717ae
1574 mengubah file dengan 162972 tambahan dan 199346 penghapusan
@@ -23,36 +23,31 @@ THE SOFTWARE.
// Helper function to spin on address until address equals value.
// If the address holds the value of -1, abort because the other thread failed.
__device__ int
gpu_spin_loop_or_abort_on_negative_one(unsigned int* address,
unsigned int value) {
__device__ int gpu_spin_loop_or_abort_on_negative_one(unsigned int* address, unsigned int value) {
unsigned int compare;
bool check = false;
do {
compare = value;
check = __hip_atomic_compare_exchange_strong(
address, /*expected=*/ &compare,
/*desired=*/ value, __ATOMIC_ACQUIRE, __ATOMIC_ACQUIRE,
/*scope=*/ __HIP_MEMORY_SCOPE_SYSTEM);
if (compare == -1)
return -1;
check =
__hip_atomic_compare_exchange_strong(address, /*expected=*/&compare,
/*desired=*/value, __ATOMIC_ACQUIRE, __ATOMIC_ACQUIRE,
/*scope=*/__HIP_MEMORY_SCOPE_SYSTEM);
if (compare == -1) return -1;
} while (!check);
return 0;
}
// This kernel requires a single block, single thread dispatch.
__global__ void
gpu_kernel(int *A, int *B, int *X, int *Y, size_t N,
unsigned int *AA1, unsigned int *AA2,
unsigned int *BA1, unsigned int *BA2, unsigned int *dresult) {
__global__ void gpu_kernel(int* A, int* B, int* X, int* Y, size_t N, unsigned int* AA1,
unsigned int* AA2, unsigned int* BA1, unsigned int* BA2,
unsigned int* dresult) {
for (size_t i = 0; i < N; i++) {
// Store data into A, system fence, and atomically mark flag.
// This guarantees this global write is visible by device 1.
A[i] = X[i];
__hip_atomic_fetch_add(AA1, 1,
__ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
__hip_atomic_fetch_add(AA1, 1, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
// Wait on device 1's global write to B.
if (gpu_spin_loop_or_abort_on_negative_one(BA1, i+1) == -1) {
if (gpu_spin_loop_or_abort_on_negative_one(BA1, i + 1) == -1) {
*dresult = -1;
break;
}
@@ -61,17 +56,14 @@ gpu_kernel(int *A, int *B, int *X, int *Y, size_t N,
bool stored_data_matches = (B[i] == Y[i]);
if (!stored_data_matches) {
// If the data does not match, alert other thread and abort.
printf("FAIL: at i=%zu, B[i]=%d, which does not match Y[i]=%d.\n",
i, B[i], Y[i]);
__hip_atomic_exchange(AA2, -1,
__ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
printf("FAIL: at i=%zu, B[i]=%d, which does not match Y[i]=%d.\n", i, B[i], Y[i]);
__hip_atomic_exchange(AA2, -1, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
*dresult = -1;
}
// Otherwise tell the other thread to continue.
__hip_atomic_fetch_add(AA2, 1,
__ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
__hip_atomic_fetch_add(AA2, 1, __ATOMIC_RELEASE, __HIP_MEMORY_SCOPE_SYSTEM);
// Wait on kernel gpu_cache1 to finish checking X is stored in A.
if (gpu_spin_loop_or_abort_on_negative_one(BA2, i+1) == -1) {
if (gpu_spin_loop_or_abort_on_negative_one(BA2, i + 1) == -1) {
*dresult = -1;
break;
}
@@ -79,45 +71,39 @@ gpu_kernel(int *A, int *B, int *X, int *Y, size_t N,
*dresult = 0;
}
__host__ int
cpu_spin_loop_or_abort_on_negative_one(unsigned int* address,
unsigned int value) {
__host__ int cpu_spin_loop_or_abort_on_negative_one(unsigned int* address, unsigned int value) {
unsigned int compare;
bool check = false;
do {
compare = value;
check = __atomic_compare_exchange_n(
address, /*expected=*/ &compare, /*desired=*/ value,
/*weak=*/ false, __ATOMIC_ACQUIRE, __ATOMIC_ACQUIRE);
if (compare == -1)
return -1;
check = __atomic_compare_exchange_n(address, /*expected=*/&compare, /*desired=*/value,
/*weak=*/false, __ATOMIC_ACQUIRE, __ATOMIC_ACQUIRE);
if (compare == -1) return -1;
} while (!check);
return 0;
}
// This host thread runs only on a single CPU thread.
__host__ void
cpu_thread(int *A, int *B, int *X, int *Y, size_t N,
unsigned int *AA1, unsigned int *AA2,
unsigned int *BA1, unsigned int *BA2, unsigned int *hresult) {
__host__ void cpu_thread(int* A, int* B, int* X, int* Y, size_t N, unsigned int* AA1,
unsigned int* AA2, unsigned int* BA1, unsigned int* BA2,
unsigned int* hresult) {
for (size_t i = 0; i < N; i++) {
B[i] = Y[i];
__atomic_fetch_add(BA1, 1, __ATOMIC_RELEASE);
if (cpu_spin_loop_or_abort_on_negative_one(AA1, i+1) == -1) {
if (cpu_spin_loop_or_abort_on_negative_one(AA1, i + 1) == -1) {
*hresult = -1;
break;
}
bool stored_data_matches = (A[i] == X[i]);
if (!stored_data_matches) {
printf("FAIL: at i=%zu, A[i]=%d, which does not match X[i]=%d.\n",
i, A[i], X[i]);
printf("FAIL: at i=%zu, A[i]=%d, which does not match X[i]=%d.\n", i, A[i], X[i]);
__atomic_exchange_n(BA2, -1, __ATOMIC_RELEASE);
*hresult = -1;
break;
}
__atomic_fetch_add(BA2, 1, __ATOMIC_RELEASE);
if (cpu_spin_loop_or_abort_on_negative_one(AA2, i+1) == -1) {
if (cpu_spin_loop_or_abort_on_negative_one(AA2, i + 1) == -1) {
*hresult = -1;
break;
}
@@ -129,7 +115,7 @@ static bool cpu_to_gpu_coherency() {
int *A_d, *B_d, *X_d, *Y_d;
int *A_res, *A_h, *B_h, *X_h, *Y_h;
unsigned int hresult = 0;
unsigned int *dresult = nullptr;
unsigned int* dresult = nullptr;
size_t N = 1024;
size_t Nbytes = N * sizeof(int);
int numDevices = 0;
@@ -148,20 +134,17 @@ static bool cpu_to_gpu_coherency() {
return true;
}
fprintf(stderr, "info: allocate device mem (%zu bytes) on device 0\n", Nbytes);
HIP_CHECK(hipExtMallocWithFlags(reinterpret_cast<void**>(&A_d),
Nbytes, hipDeviceMallocFinegrained));
}
SECTION("With host(SVM) fine grained buffer") {
HIP_CHECK(hipHostMalloc(&A_d, Nbytes));
HIP_CHECK(
hipExtMallocWithFlags(reinterpret_cast<void**>(&A_d), Nbytes, hipDeviceMallocFinegrained));
}
SECTION("With host(SVM) fine grained buffer") { HIP_CHECK(hipHostMalloc(&A_d, Nbytes)); }
A_h = A_d;
HIP_CHECK(hipHostMalloc(&dresult, sizeof(unsigned int)));
*dresult = 0;
// Allocate Host Side Memory. Coherent Fine-grained Memory for array B.
fprintf(stderr, "info: allocate host mem (%zu bytes)\n", Nbytes);
HIP_CHECK(hipHostMalloc(&B_h, Nbytes,
(hipHostMallocCoherent | hipHostMallocMapped)));
HIP_CHECK(hipHostMalloc(&B_h, Nbytes, (hipHostMallocCoherent | hipHostMallocMapped)));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&B_d), B_h, 0));
X_h = reinterpret_cast<int*>(malloc(Nbytes));
HIP_CHECK(X_h == 0 ? hipErrorOutOfMemory : hipSuccess);
@@ -178,20 +161,16 @@ static bool cpu_to_gpu_coherency() {
unsigned int *AA1_h, *AA2_h, *BA1_h, *BA2_h;
unsigned int *AA1_d, *AA2_d, *BA1_d, *BA2_d;
HIP_CHECK(hipHostMalloc(&AA1_h, sizeof(unsigned int), hipHostMallocCoherent));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&AA1_d),
AA1_h, 0));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&AA1_d), AA1_h, 0));
*AA1_h = 0;
HIP_CHECK(hipHostMalloc(&AA2_h, sizeof(unsigned int), hipHostMallocCoherent));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&AA2_d),
AA2_h, 0));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&AA2_d), AA2_h, 0));
*AA2_h = 0;
HIP_CHECK(hipHostMalloc(&BA1_h, sizeof(unsigned int), hipHostMallocCoherent));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&BA1_d),
BA1_h, 0));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&BA1_d), BA1_h, 0));
*BA1_h = 0;
HIP_CHECK(hipHostMalloc(&BA2_h, sizeof(unsigned int), hipHostMallocCoherent));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&BA2_d),
BA2_h, 0));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&BA2_d), BA2_h, 0));
*BA2_h = 0;
// Skip the first stream, ensure stream is non-blocking.
@@ -208,17 +187,13 @@ static bool cpu_to_gpu_coherency() {
// Launch the GPU kernel.
const unsigned blocks = 1;
const unsigned threadsPerBlock = 1;
hipLaunchKernelGGL(gpu_kernel, dim3(blocks), dim3(threadsPerBlock),
0, stream,
A_d, B_d, X_d, Y_d, N,
AA1_d, AA2_d, BA1_d, BA2_d, dresult);
hipLaunchKernelGGL(gpu_kernel, dim3(blocks), dim3(threadsPerBlock), 0, stream, A_d, B_d, X_d, Y_d,
N, AA1_d, AA2_d, BA1_d, BA2_d, dresult);
// Check if launch failed.
HIP_CHECK(hipGetLastError());
// Do not sync the launched stream, instead run the cpu_thread.
std::thread host_thread(cpu_thread,
A_h, B_h, X_h, Y_h, N,
AA1_h, AA2_h, BA1_h, BA2_h, &hresult);
std::thread host_thread(cpu_thread, A_h, B_h, X_h, Y_h, N, AA1_h, AA2_h, BA1_h, BA2_h, &hresult);
// Wait for Device side to finish.
HIP_CHECK(hipStreamSynchronize(stream));
host_thread.join();
@@ -230,7 +205,7 @@ static bool cpu_to_gpu_coherency() {
HIP_CHECK(A_res == 0 ? hipErrorOutOfMemory : hipSuccess);
HIP_CHECK(hipMemcpy(A_res, A_d, Nbytes, hipMemcpyDeviceToHost));
for (size_t i = 0; i < N; i++) {
for (size_t i = 0; i < N; i++) {
REQUIRE(A_res[i] == (100000000 + i));
REQUIRE(B_h[i] == (300000000 + i));
}