Refactor IPC test files

Change-Id: I879656b9e99f5cffb6adf16e0fea4e75220cd272
This commit is contained in:
Ramesh Errabolu
2020-06-17 12:43:18 -05:00
rodzic e9a4eff8a1
commit 17bd3b7da5
2 zmienionych plików z 374 dodań i 234 usunięć
Executable → Regular
+340 -231
Wyświetl plik
@@ -50,7 +50,44 @@
// Otherwise, the boilerplate code can be either supplemented or replaced
// by your own code in your example, as necessary.
//
// The comments provided are focused more on the use of the common rocrtst
// The comments below capture the high-level flow of the two processes that
// exercise ROCr Api for IPC. The test consists of two processes a result of
// forking: One creates a buffer and signal which it shares with the second
// process via shared data/ structure. The interaction between the two process
// is given below.
//
// Parent Process
// Allocate a block of gpu-local memory
// Print log message about allocation
// Acquire access to gpu-local memory
// This step may not be needed
// Obtain a IPC handle for gpu-local memory
// Print log message about getting IPC handle
// Initialize DWords of gpu-local memory with 0x01
// Print log message about updating gpu-local memory
// Create a Signal that is capable of IPC
// Obtain a IPC handle to signal
// Print log message about signalling Child process
// Signal Child process that it can proceed
// Print log message about waiting for signal from Child process
// Wait for Child processes signal
// Verify Child has updated DWords of gpu-local memory to 0x02
// Print log message about validation of gpu-local memory
// Print log message that IPC test passed
//
// Child Process
// Print log message about waiting for signal from Parent process
// Wait/Yield for Parent process signal
// Validate Parent process signal is per expectation
// Attach to IPC memory handle shared by Parent process
// Print log message about successful acquisition of IPC memory handle
// Print log message about successful acquisition of IPC signal handle
// Verify Parent process has updated every DWord of Gpu buffer to 0x01
// Update every DWord of Gpu buffer with 0x02 value
// Print log message about validation of Gpu buffer state i.e every DWord has 0x01
//
//
// The comments provided below are focused more on the use of common rocrtst
// utilities and boilerplate code, rather than the example app. itself.
//
// The boilerplate code includes code for:
@@ -99,7 +136,6 @@
#include <sys/mman.h>
#include <algorithm>
#include <iostream>
#include <vector>
#include <atomic>
@@ -123,20 +159,35 @@ struct callback_args {
// Wrap printf to add first or second process indicator
#define PROCESS_LOG(format, ...) { \
if (verbosity() >= VERBOSE_STANDARD || !processOne_) { \
if (verbosity() >= VERBOSE_STANDARD || !parentProcess_) { \
fprintf(stdout, "line:%d P%u: " format, \
__LINE__, static_cast<int>(!processOne_), ##__VA_ARGS__); \
__LINE__, static_cast<int>(!parentProcess_), ##__VA_ARGS__); \
} \
}
// Fork safe ASSERT_EQ.
#define MSG(y, msg, ...) msg
#define Y(y, ...) y
#define FORK_ASSERT_EQ(x, ...) \
if ((x) != (Y(__VA_ARGS__))) { \
EXPECT_EQ(x, Y(__VA_ARGS__)) << MSG(__VA_ARGS__, ""); \
if (!processOne_) exit(0); \
return; \
#define FORK_ASSERT_EQ(x, ...) \
if ((x) != (Y(__VA_ARGS__))) { \
if ((x) != (Y(__VA_ARGS__))) { \
std::cout << MSG(__VA_ARGS__, ""); \
if (parentProcess_) { \
shared_->parent_status = -1; \
} else { \
shared_->child_status = -1; \
} \
ASSERT_EQ(x, Y(__VA_ARGS__)); \
} \
}
#define USR_TRIGGERED_FAILURE(x, y, z) \
if (usr_fail_val_ == (z)) { \
std::cout << "Env value is: " << z << std::endl; \
std::cout << "Return value before: " << x << std::endl; \
std::cout << "Return value after: " << y << std::endl << std::flush; \
(x) = (y); \
}
IPCTest::IPCTest(void) :
@@ -171,7 +222,12 @@ static int CheckAndSetToken(std::atomic<int> *token, int newVal) {
void IPCTest::SetUp(void) {
hsa_status_t err;
int ret;
// Allow user to trigger a failure
const char* env_val = getenv("ROCR_IPC_FAIL_KEY");
if (env_val != NULL) {
usr_fail_val_ = atoi(env_val);
}
// We must fork process before doing HSA stuff, specifically, hsa_init, as
// each process needs to do this.
// Allocate linux shared_ memory.
@@ -180,16 +236,17 @@ void IPCTest::SetUp(void) {
MAP_SHARED | MAP_ANONYMOUS, -1, 0));
ASSERT_NE(shared_, MAP_FAILED) << "mmap failed to allocated shared_ memory";
// "token" is used to signal state changes between the 2 processes.
std::atomic<int> * token = &shared_->token;
*token = 0;
// Initialize shared control block to zeros. The field "token"
// is used to signal state changes between the 2 processes.
memset(shared_, 0, sizeof(Shared));
// Spawn second process and verify communication
child_ = 0;
child_ = fork();
ASSERT_NE(-1, child_) << "fork failed";
std::atomic<int> * token = &shared_->token;
if (child_ != 0) {
processOne_ = true;
parentProcess_ = true;
// Signal to other process we are waiting, and then wait...
*token = 1;
@@ -204,7 +261,7 @@ void IPCTest::SetUp(void) {
}
} else {
processOne_ = false;
parentProcess_ = false;
set_verbosity(0);
PROCESS_LOG("Second process running.\n");
@@ -212,14 +269,15 @@ void IPCTest::SetUp(void) {
sched_yield();
}
int ret;
ret = CheckAndSetToken(token, 0);
ASSERT_EQ(0, ret) << "Error detected in child process";
ASSERT_EQ(0, ret) << "Error detected in child process\n";
// Wait for handshake
while (*token == 0) {
sched_yield();
}
ret = CheckAndSetToken(token, 0);
ASSERT_EQ(0, ret) << "Error detected in child process";
ASSERT_EQ(0, ret) << "Error detected in child process\n";
}
// TestBase::SetUp() will set HSA_ENABLE_INTERRUPT if enable_interrupt() is
// true, and call hsa_init(). It also prints the SetUp header.
@@ -239,6 +297,14 @@ void IPCTest::SetUp(void) {
err = rocrtst::SetPoolsTypical(this);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Update the size granularity for allocations
#ifdef ROCRTST_EMULATOR_BUILD
gpu_mem_granule = 4;
#else
err = hsa_amd_memory_pool_get_info(device_pool(), HSA_AMD_MEMORY_POOL_INFO_RUNTIME_ALLOC_GRANULE,
&gpu_mem_granule);
#endif
return;
}
@@ -248,238 +314,281 @@ uint32_t IPCTest::RealIterationNum(void) {
return num_iteration() * 1.2 + 1;
}
void IPCTest::Run(void) {
hsa_status_t err;
std::atomic<int> *token = &shared_->token;
TestBase::Run();
void IPCTest::ChildProcessImpl() {
// Print out name of the device.
// Yield until shared token value changes i.e. is updated by parent.
// Validate parent's update is per expectation
PROCESS_LOG("Child: Waiting for parent process to signal\n");
while (shared_->token == 0) {
sched_yield();
}
if (shared_->token != 1) {
shared_->token = -1;
}
FORK_ASSERT_EQ(1, shared_->token, "Child: Error detected in signaling token\n");
PROCESS_LOG("Child: Waking upon signal from parent process\n");
// List of devices involved in test. Gpu device is used
// to allocate buffer and signal that are part of an IPC
// transaction. Cpu is used in support of initialization
// of Gpu buffer
hsa_agent_t ag_list[2] = {*gpu_device1(), *cpu_device()};
// Attach to IPC memory handle shared by parent process
void* ipc_ptr;
hsa_status_t err;
err = hsa_amd_ipc_memory_attach(const_cast<hsa_amd_ipc_memory_t*>(&shared_->handle),
shared_->size, 1, ag_list, &ipc_ptr);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 200);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Child: Failure in attaching to IPC memory handle\n");
PROCESS_LOG("Child: Attached to IPC buffer shared by parent process\n");
PROCESS_LOG("Child: Address of buffer enabled for IPC: %p\n", ipc_ptr);
// Attach to IPC signal handle shared by parent process
hsa_signal_t ipc_signal;
err = hsa_amd_ipc_signal_attach(const_cast<hsa_amd_ipc_signal_t*>(&shared_->signal_handle),
&ipc_signal);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 201);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Child: Failure in attaching to IPC signal handle\n");
PROCESS_LOG("Child: Attached to IPC signal shared by parent process\n");
// Validate Gpu buffer is filled per expectation i.e. if so update
// per previously agreed upon value (first_val_ and second_val_)
CheckAndFillBuffer(reinterpret_cast<uint32_t*>(ipc_ptr), first_val_, second_val_);
PROCESS_LOG("Child: Confirmed DWord's of IPC buffer has: %d\n", first_val_);
PROCESS_LOG("Child: Updated DWord's of IPC buffer to: %d\n", second_val_);
// Detach IPC memory that was used to test
err = hsa_amd_ipc_memory_detach(ipc_ptr);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 202);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Child: Failure in detaching IPC memory handle\n");
PROCESS_LOG("Child: Detached IPC memory handle\n");
// Signal parent process to wake up and continue
hsa_signal_store_release(ipc_signal, 2);
// Wait for signal from parent to continue, expected value is zero
hsa_signal_value_t ret = 2;
while(true) {
ret = hsa_signal_wait_relaxed(ipc_signal, HSA_SIGNAL_CONDITION_NE, 2, timeout_, HSA_WAIT_STATE_BLOCKED);
if (shared_->parent_status == -1) {
exit(0);
}
if (ret == 0) {
break;
}
}
USR_TRIGGERED_FAILURE(ret, HSA_STATUS_ERROR, 203);
FORK_ASSERT_EQ(0, ret, "Child: Expected signal value of 0, but got " << ret << "\n");
// Reset the signal object and release acquired resources
err = hsa_signal_destroy(ipc_signal);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 204);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Child: Failure in destroying IPC signal handle\n");
PROCESS_LOG("Child: IPC test PASSED\n");
}
void IPCTest::ParentProcessImpl() {
// Ignoring the first allocation to exercise fragment allocation.
hsa_status_t err;
uint32_t* discard = NULL;
err = hsa_amd_memory_pool_allocate(device_pool(), gpu_mem_granule, 0,
reinterpret_cast<void**>(&discard));
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 100);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failed to allocate gpu memory\n");
// Allocate some VRAM that is used to test IPC
uint32_t* gpuBuf = NULL;
err = hsa_amd_memory_pool_allocate(device_pool(), gpu_mem_granule, 0,
reinterpret_cast<void**>(&gpuBuf));
PROCESS_LOG("Parent: Allocated framebuffer of size: %zu\n", gpu_mem_granule);
PROCESS_LOG("Parent: Address of allocated framebuffer: %p\n", gpuBuf);
// Free the test allocation of memory block
err = hsa_amd_memory_pool_free(discard);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 101);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failed to free gpu memory\n");
// List of devices involved in test. Gpu device is used
// to allocate buffer and signal that are part of an IPC
// transaction. Cpu is used in support of initialization
// of Gpu buffer
hsa_agent_t ag_list[2] = {*gpu_device1(), *cpu_device()};
// Grant access to buffer to participating devices
err = hsa_amd_agents_allow_access(2, ag_list, NULL, gpuBuf);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 102);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failed to get access to gpu memory\n");
// Update shared data structure's buffer related parameters
shared_->size = gpu_mem_granule;
shared_->count = gpu_mem_granule / sizeof(uint32_t);
// Initialize every DWord of IPC buffer with a value per previous
// agreement i.e. first_val_
err = hsa_amd_memory_fill(gpuBuf, first_val_, shared_->count);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 103);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failed to initialize gpu memory\n");
PROCESS_LOG("Parent: Initialized Dword's of framebuffer with: %d\n", first_val_);
// Create an IPC memory handle. IPC handle value is shared with
// child process via a shared data structure
err = hsa_amd_ipc_memory_create(gpuBuf, gpu_mem_granule,
const_cast<hsa_amd_ipc_memory_t*>(&shared_->handle));
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 104);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failed to create IPC memory handle\n");
PROCESS_LOG("Parent: Created IPC handle for framebuffer: %p\n", gpuBuf);
// Create a signal that is capable of IPC. Also obtain a IPC handle
// which is shared with child process via a shared data structure
hsa_signal_t ipc_signal;
err = hsa_amd_signal_create(1, 0, NULL, HSA_AMD_SIGNAL_IPC, &ipc_signal);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 105);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failed to create IPC signal\n");
err = hsa_amd_ipc_signal_create(ipc_signal,
const_cast<hsa_amd_ipc_signal_t*>(&shared_->signal_handle));
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 106);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failed to create IPC signal handle\n");
PROCESS_LOG("Parent: Created IPC handle associated with ipc_signal\n");
// Signal child process that the gpu buffer is ready to read.
PROCESS_LOG("Parent: Signalling child proces process\n");
CheckAndSetToken(&shared_->token, 1);
PROCESS_LOG("Parent: Waiting for signal from child process\n");
// Wait for child processs to signal. Child will update signal object
// value to TWO (2). Check signal value is per expectation
hsa_signal_value_t ret = 1;
while(true) {
ret = hsa_signal_wait_acquire(ipc_signal, HSA_SIGNAL_CONDITION_NE, 1, timeout_, HSA_WAIT_STATE_BLOCKED);
if (shared_->child_status == -1) {
exit(0);
}
if (ret == 2) {
break;
}
}
USR_TRIGGERED_FAILURE(ret, HSA_STATUS_ERROR, 107);
FORK_ASSERT_EQ(2, ret, "Parent: Expected signal value of 2, but got " << ret << "\n");
// Verify child process has updated all DWords of buffer per
// previously agreed upon values (second_val_ and third_val_)
CheckAndFillBuffer(gpuBuf, second_val_, third_val_);
PROCESS_LOG("Parent: Confirmed DWord's of frambuffer has: %d\n", second_val_);
PROCESS_LOG("Parent: Updated DWord's of framebuffer to: %d\n", third_val_);
// Reset the signal object and release acquired resources
hsa_signal_store_relaxed(ipc_signal, 0);
err = hsa_signal_destroy(ipc_signal);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 108);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failure in destroying IPC signal\n");
err = hsa_amd_memory_pool_free(gpuBuf);
USR_TRIGGERED_FAILURE(err, HSA_STATUS_ERROR, 109);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Parent: Failed to free gpu memory\n");
PROCESS_LOG("Parent: IPC test PASSED\n");
// Wait for child process to terminate before exiting
int exit_status = 0;
waitpid(child_, &exit_status, 0);
munmap(shared_, sizeof(Shared));
}
void IPCTest::PrintVerboseMesg(void) {
// Collect names of GPU's
hsa_status_t err;
char name1[64] = {0};
char name2[64] = {0};
err = hsa_agent_get_info(*cpu_device(), HSA_AGENT_INFO_NAME, name1);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "hsa_agent_get_info() failed");
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "hsa_agent_get_info() failed\n");
err = hsa_agent_get_info(*gpu_device1(), HSA_AGENT_INFO_NAME, name2);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "hsa_agent_get_info() failed");
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "hsa_agent_get_info() failed\n");
// Collect BDF information of GPU's
uint16_t loc1, loc2;
err = hsa_agent_get_info(*cpu_device(),
(hsa_agent_info_t)HSA_AMD_AGENT_INFO_BDFID, &loc1);
err = hsa_agent_get_info(*cpu_device(), (hsa_agent_info_t)HSA_AMD_AGENT_INFO_BDFID, &loc1);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
err = hsa_agent_get_info(*gpu_device1(),
(hsa_agent_info_t)HSA_AMD_AGENT_INFO_BDFID, &loc2);
err = hsa_agent_get_info(*gpu_device1(), (hsa_agent_info_t)HSA_AMD_AGENT_INFO_BDFID, &loc2);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
size_t gpu_mem_granule;
#ifdef ROCRTST_EMULATOR_BUILD
gpu_mem_granule = 4;
#else
err = hsa_amd_memory_pool_get_info(device_pool(),
HSA_AMD_MEMORY_POOL_INFO_RUNTIME_ALLOC_GRANULE, &gpu_mem_granule);
#endif
// Print the name and BDF info about the devices
fprintf(stdout, "Using: %s (%d) and %s (%d)\n", name1, loc1, name2, loc2);
}
if (verbosity() >= VERBOSE_STANDARD) {
fprintf(stdout, "Using: %s (%d) and %s (%d)\n", name1, loc1, name2, loc2);
void IPCTest::CheckAndFillBuffer(void* gpu_src_ptr, uint32_t exp_cur_val, uint32_t new_val) {
uint32_t* sysBuf;
hsa_status_t err;
hsa_signal_value_t sig;
hsa_signal_t copy_signal;
// Bind the size granularity of allocation
size_t sz = gpu_mem_granule;
// Allocate a signal to track copy progress
err = hsa_signal_create(1, 0, NULL, &copy_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Allocate buffer in system memory to validate
err = hsa_amd_memory_pool_allocate(cpu_pool(), sz, 0, reinterpret_cast<void**>(&sysBuf));
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Enable access to buffer in system memory
hsa_agent_t ag_list[2] = {*gpu_device1(), *cpu_device()};
err = hsa_amd_agents_allow_access(2, ag_list, NULL, sysBuf);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Copy data to buffer in system memory
err = hsa_amd_memory_async_copy(sysBuf, *cpu_device(), gpu_src_ptr, *gpu_device1(), sz, 0, NULL,
copy_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Wait for copy to complete
sig = hsa_signal_wait_relaxed(copy_signal,
HSA_SIGNAL_CONDITION_LT, 1, -1, HSA_WAIT_STATE_BLOCKED);
FORK_ASSERT_EQ(0, sig, "Expected signal 0, but got " << sig << "\n");
// Validate buffer has expected data
uint32_t count = sz / sizeof(uint32_t);
for (uint32_t idx = 0; idx < count; idx++) {
if (exp_cur_val != sysBuf[idx]) {
PROCESS_LOG("Validation failed: expected: %d observed: %d at index: %d\n",
exp_cur_val, sysBuf[idx], idx);
FORK_ASSERT_EQ(exp_cur_val, sysBuf[idx]);
}
sysBuf[idx] = new_val;
}
hsa_agent_t ag_list[2] = {*gpu_device1(), *cpu_device()};
// Reset copy signal and update buffer in Gpu with new value
hsa_signal_store_relaxed(copy_signal, 1);
err = hsa_amd_memory_async_copy(gpu_src_ptr, *gpu_device1(), sysBuf, *cpu_device(), sz, 0, NULL,
copy_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
auto CheckAndFillBuffer = [&](void *gpu_src_ptr, uint32_t exp_cur_val,
uint32_t new_val) -> void {
hsa_signal_t copy_signal;
size_t sz = gpu_mem_granule;
hsa_status_t err;
hsa_signal_value_t sig;
// Wait for copy to complete
sig = hsa_signal_wait_relaxed(copy_signal, HSA_SIGNAL_CONDITION_LT, 1, -1, HSA_WAIT_STATE_BLOCKED);
FORK_ASSERT_EQ(sig, 0, "Expected signal 0, but got " << sig << "\n");
err = hsa_signal_create(1, 0, NULL, &copy_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Release resources allocated by this method
err = hsa_signal_destroy(copy_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
err = hsa_amd_memory_pool_free(sysBuf);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
}
uint32_t *sysBuf;
void IPCTest::Run(void) {
TestBase::Run();
err = hsa_amd_memory_pool_allocate(cpu_pool(), sz, 0,
reinterpret_cast<void **>(&sysBuf));
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Collect and print debug information
if (verbosity() >= VERBOSE_STANDARD) {
PrintVerboseMesg();
}
hsa_agent_t ag_list[2] = {*gpu_device1(), *cpu_device()};
err = hsa_amd_agents_allow_access(2, ag_list, NULL, sysBuf);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
err = hsa_amd_memory_async_copy(sysBuf, *cpu_device(), gpu_src_ptr,
*gpu_device1(), sz, 0, NULL, copy_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
sig = hsa_signal_wait_relaxed(copy_signal, HSA_SIGNAL_CONDITION_LT,
1, -1, HSA_WAIT_STATE_BLOCKED);
FORK_ASSERT_EQ(0, sig, "Expected signal 0, but got " << sig);
uint32_t count = sz/sizeof(uint32_t);
for (uint32_t i = 0; i < count; ++i) {
FORK_ASSERT_EQ(exp_cur_val, sysBuf[i]);
sysBuf[i] = new_val;
}
hsa_signal_store_relaxed(copy_signal, 1);
err = hsa_amd_memory_async_copy(gpu_src_ptr, *gpu_device1(), sysBuf,
*cpu_device(), sz, 0, NULL, copy_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
sig = hsa_signal_wait_relaxed(copy_signal, HSA_SIGNAL_CONDITION_LT,
1, -1, HSA_WAIT_STATE_BLOCKED);
FORK_ASSERT_EQ(sig, 0, "Expected signal 0, but got " << sig);
err = hsa_signal_destroy(copy_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
err = hsa_amd_memory_pool_free(sysBuf);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
};
if (processOne_) {
// Ignoring the first allocation to exercise fragment allocation.
uint32_t* discard = NULL;
err = hsa_amd_memory_pool_allocate(device_pool(), gpu_mem_granule, 0,
reinterpret_cast<void**>(&discard));
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Failed to allocate memory from gpu pool");
// Allocate some VRAM and fill it with 1's
uint32_t* gpuBuf = NULL;
err = hsa_amd_memory_pool_allocate(device_pool(), gpu_mem_granule, 0,
reinterpret_cast<void**>(&gpuBuf));
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Failed to allocate memory from gpu pool");
err = hsa_amd_memory_pool_free(discard);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "Failed to free GPU memory");
PROCESS_LOG("Allocated local memory buffer at %p\n", gpuBuf);
err = hsa_amd_agents_allow_access(2, ag_list, NULL, gpuBuf);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
err = hsa_amd_ipc_memory_create(gpuBuf, gpu_mem_granule,
const_cast<hsa_amd_ipc_memory_t*>(&shared_->handle));
PROCESS_LOG(
"Created IPC handle associated with gpu-local buffer at P0 address %p\n",
gpuBuf);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
uint32_t count = gpu_mem_granule/sizeof(uint32_t);
shared_->size = gpu_mem_granule;
shared_->count = count;
err = hsa_amd_memory_fill(gpuBuf, 1, count);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Get IPC capable signal
hsa_signal_t ipc_signal;
err = hsa_amd_signal_create(1, 0, NULL, HSA_AMD_SIGNAL_IPC, &ipc_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
err = hsa_amd_ipc_signal_create(ipc_signal,
const_cast<hsa_amd_ipc_signal_t*>(&shared_->signal_handle));
PROCESS_LOG("Created IPC handle associated with ipc_signal\n");
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
// Signal Process 2 that the gpu buffer is ready to read.
CheckAndSetToken(token, 1);
PROCESS_LOG("Allocated buffer and filled it with 1's. Wait for P1...\n");
hsa_signal_value_t ret =
hsa_signal_wait_acquire(ipc_signal, HSA_SIGNAL_CONDITION_NE, 1, -1,
HSA_WAIT_STATE_BLOCKED);
FORK_ASSERT_EQ(2, ret, "Expected signal value of 2, but got " << ret);
CheckAndFillBuffer(gpuBuf, 2, 0);
PROCESS_LOG("Confirmed P1 filled buffer with 2\n")
PROCESS_LOG("PASSED on P0\n");
hsa_signal_store_relaxed(ipc_signal, 0);
err = hsa_signal_destroy(ipc_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
err = hsa_amd_memory_pool_free(gpuBuf);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
waitpid(child_, nullptr, 0);
munmap(shared_, sizeof(Shared));
// Note: Close() (and hsa_shut_down()) will be called from main() in
// RunCustomEpilog()
} else { // "ProcessTwo"
PROCESS_LOG("Waiting for process 0 to write 1 to token...\n");
while (*token == 0) {
sched_yield();
}
if (*token != 1) {
*token = -1;
}
FORK_ASSERT_EQ(1, *token, "Error detected in signaling token");
PROCESS_LOG("Process 0 wrote 1 to token...\n");
// Attach shared_ VRAM
void* ptr;
err = hsa_amd_ipc_memory_attach(
const_cast<hsa_amd_ipc_memory_t*>(&shared_->handle), shared_->size, 1,
ag_list, &ptr);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
PROCESS_LOG(
"Attached to IPC handle; P1 buffer address gpu-local memory is %p\n",
ptr);
// Attach shared_ signal
hsa_signal_t ipc_signal;
err = hsa_amd_ipc_signal_attach(
const_cast<hsa_amd_ipc_signal_t*>(&shared_->signal_handle), &ipc_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
PROCESS_LOG("Attached to signal IPC handle\n");
CheckAndFillBuffer(reinterpret_cast<uint32_t *>(ptr), 1, 2);
PROCESS_LOG(
"Confirmed P0 filled buffer with 1; P1 re-filled buffer with 2\n");
PROCESS_LOG("PASSED on P1\n");
hsa_signal_store_release(ipc_signal, 2);
// Test ptrinfo - allocation is a single granule so this tests imported
// fragment info.
err = hsa_amd_pointer_info_set_userdata(ptr,
reinterpret_cast<void*>(0xDEADBEEF));
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err, "hsa_amd_pointer_info_set_userdata() failed");
hsa_amd_pointer_info_t info;
info.size = sizeof(info);
err = hsa_amd_pointer_info(
reinterpret_cast<uint8_t*>(ptr) + shared_->size / 2, &info,
nullptr, nullptr, nullptr);
if ((info.sizeInBytes != shared_->size) ||
(info.userData != reinterpret_cast<void*>(0xDEADBEEF)) ||
(info.agentBaseAddress != ptr)) {
PROCESS_LOG("Pointer Info check failed.\n");
} else {
PROCESS_LOG("Pointer Info check PASSED.\n");
}
err = hsa_amd_ipc_memory_detach(ptr);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
hsa_signal_wait_relaxed(ipc_signal, HSA_SIGNAL_CONDITION_NE, 2, -1,
HSA_WAIT_STATE_BLOCKED);
err = hsa_signal_destroy(ipc_signal);
FORK_ASSERT_EQ(HSA_STATUS_SUCCESS, err);
Close();
// exit(0);
// hwloc hack - OpenCL registers exit handlers that can't execute in this process.
// Skip them.
abort();
// Note: Close() (and hsa_shut_down()) will be called from main()
// processOne is true for parent process, false for child process
if (parentProcess_) {
ParentProcessImpl();
} else {
ChildProcessImpl();
}
return;
Executable → Regular
+34 -3
Wyświetl plik
@@ -46,6 +46,8 @@
#ifndef ROCRTST_SUITES_FUNCTIONAL_IPC_H_
#define ROCRTST_SUITES_FUNCTIONAL_IPC_H_
#include <sys/types.h>
#include <unistd.h>
#include <atomic>
#include "common/base_rocr.h"
@@ -55,7 +57,9 @@
struct Shared {
std::atomic<int> token;
std::atomic<int> count;
std::atomic<size_t> size;
std::atomic<size_t> size;
std::atomic<int> child_status;
std::atomic<int> parent_status;
hsa_amd_ipc_memory_t handle;
hsa_amd_ipc_signal_t signal_handle;
};
@@ -82,11 +86,38 @@ class IPCTest : public TestBase {
// @Brief: Display information about what this test does
virtual void DisplayTestInfo(void);
// @Brief: Implements child process exclusive logic
void ChildProcessImpl();
// @Brief: Implements parent process exclusive logic
void ParentProcessImpl();
private:
// @Brief: Bind number of iterations to run per user specification
uint32_t RealIterationNum(void);
Shared* shared_;
bool processOne_;
// @Brief: Collect and print verbose messages to enable debugging
void PrintVerboseMesg(void);
// @Brief: Implements the check to see if buffer has expected
// value if so updates it with new values
void CheckAndFillBuffer(void* gpu_src_ptr, uint32_t exp_cur_val, uint32_t new_val);
// @Brief: Values used to initialize framebuffer that is shared
uint32_t first_val_ = 0x01;
uint32_t second_val_ = 0x02;
uint32_t third_val_ = 0x03;
int child_;
Shared* shared_;
bool parentProcess_;
size_t gpu_mem_granule;
// Supports user triggered failure
int32_t usr_fail_val_ = 0xFFFFFFFF;
// Specifies timeout period for parent/child processes
int32_t timeout_ = 0x20000;
};
#endif // ROCRTST_SUITES_FUNCTIONAL_IPC_H_