SWDEV-470698 - fix formatting, add format check workflow (#657)

This commit is contained in:
Danylo Lytovchenko
2025-08-20 16:28:06 +02:00
committed by GitHub
parent 5840940caa
commit f7338717ae
1574 changed files with 162972 additions and 199346 deletions
@@ -25,20 +25,17 @@
#include <hip_test_checkers.hh>
#define N 1048576
__managed__ float A[N]; // Accessible by ALL CPU and GPU functions !!!
__managed__ float A[N]; // Accessible by ALL CPU and GPU functions !!!
__managed__ float B[N];
__managed__ int x = 0;
__managed__ int x = 0;
__global__ void add(const float *A, float *B) {
__global__ void add(const float* A, float* B) {
int index = blockIdx.x * blockDim.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
for (int i = index; i < N; i += stride)
B[i] = A[i] + B[i];
for (int i = index; i < N; i += stride) B[i] = A[i] + B[i];
}
__global__ void GPU_func() {
x++;
}
__global__ void GPU_func() { x++; }
TEST_CASE("Unit_hipManagedKeyword_SingleGpu") {
for (int i = 0; i < N; i++) {
@@ -57,8 +54,7 @@ TEST_CASE("Unit_hipManagedKeyword_SingleGpu") {
HIP_CHECK(hipDeviceSynchronize());
float maxError = 0.0f;
for (int i = 0; i < N; i++)
maxError = fmax(maxError, fabs(B[i]-3.0f));
for (int i = 0; i < N; i++) maxError = fmax(maxError, fabs(B[i] - 3.0f));
REQUIRE(maxError == 0.0f);
}
@@ -67,11 +63,9 @@ TEST_CASE("Unit_hipManagedKeyword_MultiGpu") {
int numDevices = 0;
HIP_CHECK(hipGetDeviceCount(&numDevices));
for (int i = 0; i < numDevices; i++){
for (int i = 0; i < numDevices; i++) {
int managed_memory = 0;
HIPCHECK(hipDeviceGetAttribute(&managed_memory,
hipDeviceAttributeManagedMemory,
i));
HIPCHECK(hipDeviceGetAttribute(&managed_memory, hipDeviceAttributeManagedMemory, i));
if (!managed_memory) {
HipTest::HIP_SKIP_TEST("managed memory access not supported on device");
return;
@@ -80,7 +74,7 @@ TEST_CASE("Unit_hipManagedKeyword_MultiGpu") {
for (int i = 0; i < numDevices; i++) {
HIP_CHECK(hipSetDevice(i));
GPU_func<<< 1, 1 >>>();
GPU_func<<<1, 1>>>();
HIP_CHECK(hipDeviceSynchronize());
}
REQUIRE(x == numDevices);
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -164,11 +164,10 @@ void TestContext::getConfigFiles() {
}
std::string env_config = TestContext::getEnvVar("HIP_CATCH_EXCLUDE_FILE");
LogPrintf("Env Config file: %s",
(!env_config.empty()) ? env_config.c_str() : "Not found");
LogPrintf("Env Config file: %s", (!env_config.empty()) ? env_config.c_str() : "Not found");
// HIP_CATCH_EXCLUDE_FILE is set for custom file path
if (!env_config.empty()) {
if(fs::exists(env_config)) {
if (fs::exists(env_config)) {
config_.json_files.push_back(env_config);
}
} else {
@@ -236,7 +235,8 @@ bool TestContext::parseJsonFiles() {
}
// Open the file
std::ifstream js_file(fl);
std::string json_str((std::istreambuf_iterator<char>(js_file)), std::istreambuf_iterator<char>());
std::string json_str((std::istreambuf_iterator<char>(js_file)),
std::istreambuf_iterator<char>());
LogPrintf("Json contents:: %s", json_str.data());
picojson::value v;
@@ -6,9 +6,9 @@
#include "hip_test_context.hh"
std::vector<std::unordered_set<std::string>> GCNArchFeatMap = {
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_FINEGRAIN_HWSUPPORT
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_HMM
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_TEXTURES_NOT_SUPPORTED
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_FINEGRAIN_HWSUPPORT
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_HMM
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_TEXTURES_NOT_SUPPORTED
};
#if HT_AMD
@@ -24,7 +24,7 @@ std::string TrimAndGetGFXName(const std::string& full_gfx_name) {
gfx_name = full_gfx_name.substr(0, pos);
}
assert(gfx_name.substr(0,3) == "gfx");
assert(gfx_name.substr(0, 3) == "gfx");
return gfx_name;
}
#endif
@@ -32,14 +32,14 @@ std::string TrimAndGetGFXName(const std::string& full_gfx_name) {
// Check if the GCN Maps
bool CheckIfFeatSupported(enum CTFeatures test_feat, std::string gcn_arch) {
#if HT_NVIDIA
return true; // returning true since feature check does not exist for NV.
return true; // returning true since feature check does not exist for NV.
#elif HT_AMD
assert(test_feat >= 0 && test_feat < CTFeatures::CT_FEATURE_LAST);
gcn_arch = TrimAndGetGFXName(gcn_arch);
assert(gcn_arch != "");
return (GCNArchFeatMap[test_feat].find(gcn_arch) != GCNArchFeatMap[test_feat].cend());
#else
std::cout<<"Platform has to be either AMD or NVIDIA, asserting..."<<std::endl;
std::cout << "Platform has to be either AMD or NVIDIA, asserting..." << std::endl;
assert(false);
#endif
}
+4 -4
View File
@@ -122,8 +122,8 @@ inline dim3 GenerateThreadDimensions() {
map([max = props.maxThreadsDim[2], warp_size = props.warpSize](
double i) { return dim3(1, 1, std::min(static_cast<int>(i * warp_size), max)); },
values(multipliers)),
dim3(16, 8, 8), dim3(32, 32, 1), dim3(64, 8, 2), dim3(16, 16, 3), dim3(props.warpSize - 1, 3, 3),
dim3(props.warpSize + 1, 3, 3));
dim3(16, 8, 8), dim3(32, 32, 1), dim3(64, 8, 2), dim3(16, 16, 3),
dim3(props.warpSize - 1, 3, 3), dim3(props.warpSize + 1, 3, 3));
}
/* Generate dimensions for 1D, 2D and 3D grids of blocks */
@@ -161,8 +161,8 @@ inline dim3 GenerateThreadDimensionsForShuffle() {
map([max = props.maxThreadsDim[2], warp_size = props.warpSize](
double i) { return dim3(1, 1, std::min(static_cast<int>(i * warp_size), max)); },
values(multipliers)),
dim3(16, 8, 8), dim3(32, 32, 1), dim3(64, 8, 2), dim3(16, 16, 3), dim3(props.warpSize - 1, 3, 3),
dim3(props.warpSize + 1, 3, 3));
dim3(16, 8, 8), dim3(32, 32, 1), dim3(64, 8, 2), dim3(16, 16, 3),
dim3(props.warpSize - 1, 3, 3), dim3(props.warpSize + 1, 3, 3));
}
/* Generate dimensions for 1D, 2D and 3D grids of blocks - reduced set */
+2 -2
View File
@@ -1,6 +1,6 @@
#include <kernels.hh>
__global__ void Set(int* Ad, int val) {
int tx = threadIdx.x + blockIdx.x * blockDim.x;
Ad[tx] = val;
int tx = threadIdx.x + blockIdx.x * blockDim.x;
Ad[tx] = val;
}
@@ -41,7 +41,7 @@ static __global__ void kerTestDeviceMalloc(size_t size) {
int myId = threadIdx.x + blockDim.x * blockIdx.x;
// Allocate
if (myId == 0) {
dev_common_ptr = reinterpret_cast<char*> (malloc(size));
dev_common_ptr = reinterpret_cast<char*>(malloc(size));
if (dev_common_ptr == nullptr) {
printf("Device Allocation Failed! \n");
return;
@@ -67,13 +67,13 @@ static __global__ void kerTestDeviceWrite() {
* This kernel frees the memory chunk allocated in kernel
* kerTestDeviceMalloc using free().
*/
static __global__ void kerTestDeviceFree(int *result) {
static __global__ void kerTestDeviceFree(int* result) {
int myId = threadIdx.x + blockDim.x * blockIdx.x;
// Allocate
if (myId == 0) {
if (dev_common_ptr != nullptr) {
*result = 1;
for (int idx = 0; idx < (BLOCKSIZE*GRIDSIZE); idx++) {
for (int idx = 0; idx < (BLOCKSIZE * GRIDSIZE); idx++) {
if (*(dev_common_ptr + myId) != SCHAR_MAX) {
*result = 0;
break;
@@ -105,13 +105,13 @@ static __global__ void kerTestDeviceNew(size_t size) {
* This kernel frees the memory chunk allocated in kernel
* kerTestDeviceNew using delete operator.
*/
static __global__ void kerTestDeviceDelete(int *result) {
static __global__ void kerTestDeviceDelete(int* result) {
int myId = threadIdx.x + blockDim.x * blockIdx.x;
// Allocate
if (myId == 0) {
if (dev_common_ptr != nullptr) {
*result = 1;
for (int idx = 0; idx < (BLOCKSIZE*GRIDSIZE); idx++) {
for (int idx = 0; idx < (BLOCKSIZE * GRIDSIZE); idx++) {
if (*(dev_common_ptr + myId) != SCHAR_MAX) {
*result = 0;
break;
@@ -140,7 +140,7 @@ static bool testDeviceAllocMulProc(bool testmalloc) {
childpid = fork();
if (childpid > 0) { // Parent
close(fd[1]);
int *result_d{nullptr};
int* result_d{nullptr};
HIP_CHECK(hipMalloc(&result_d, sizeof(int)));
// Allocate in parent
if (testmalloc) {
@@ -185,7 +185,7 @@ static bool testDeviceAllocMulProc(bool testmalloc) {
HIP_CHECK(hipFree(result_d));
} else if (!childpid) { // Child
// Wait for hipDeviceSetLimit() completion in parent.
int *result_d{nullptr};
int* result_d{nullptr};
HIP_CHECK(hipMalloc(&result_d, sizeof(int)));
close(fd[0]);
// Allocate in child
@@ -230,7 +230,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
bool testResult = false;
pid_t childpid;
int testResultChild = 0;
size_t size = BLOCKSIZE*GRIDSIZE;
size_t size = BLOCKSIZE * GRIDSIZE;
// create pipe descriptors
pipe(fd);
// fork process
@@ -239,7 +239,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
close(fd[1]);
int *result_d{nullptr}, *result_h{nullptr};
HIP_CHECK(hipMalloc(&result_d, sizeof(int)));
result_h = reinterpret_cast<int*> (malloc(sizeof(int)));
result_h = reinterpret_cast<int*>(malloc(sizeof(int)));
REQUIRE(result_h != nullptr);
// Allocate in parent
if (testmalloc) {
@@ -257,8 +257,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
}
HIP_CHECK(hipDeviceSynchronize());
*result_h = 0;
HIP_CHECK(hipMemcpy(result_h, result_d, sizeof(int),
hipMemcpyDefault));
HIP_CHECK(hipMemcpy(result_h, result_d, sizeof(int), hipMemcpyDefault));
if (*result_h == 0) {
testResult = false;
} else {
@@ -282,7 +281,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
close(fd[0]);
int *result_d{nullptr}, *result_h{nullptr};
HIP_CHECK(hipMalloc(&result_d, sizeof(int)));
result_h = reinterpret_cast<int*> (malloc(sizeof(int)));
result_h = reinterpret_cast<int*>(malloc(sizeof(int)));
REQUIRE(result_h != nullptr);
// Allocate in child
if (testmalloc) {
@@ -300,8 +299,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
}
HIP_CHECK(hipDeviceSynchronize());
*result_h = 0;
HIP_CHECK(hipMemcpy(result_h, result_d, sizeof(int),
hipMemcpyDefault));
HIP_CHECK(hipMemcpy(result_h, result_d, sizeof(int), hipMemcpyDefault));
// send the value on the write-descriptor:
write(fd[1], result_h, sizeof(int));
// close the write descriptor:
@@ -22,5 +22,4 @@ THE SOFTWARE.
#include "hip/hip_runtime.h"
extern "C" __global__ void dummy_ker() {
}
extern "C" __global__ void dummy_ker() {}
@@ -34,7 +34,7 @@ THE SOFTWARE.
/**
* Fetches Gpu device count
*/
static void getDeviceCount(int *pdevCnt) {
static void getDeviceCount(int* pdevCnt) {
int fd[2], val = 0;
pid_t childpid;
@@ -107,15 +107,14 @@ bool runMaskedDeviceTest(int actualNumGPUs) {
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
#endif
for (int count = 1;
count < actualNumGPUs; count++) {
for (int count = 1; count < actualNumGPUs; count++) {
int major, minor;
err = hipDeviceComputeCapability(&major, &minor, count);
if (err == hipSuccess) {
testResult = false;
} else {
printf("hipDeviceComputeCapability: Error Code Returned: '%s'(%d)\n",
hipGetErrorString(err), err);
hipGetErrorString(err), err);
}
}
close(fd[0]);
@@ -38,7 +38,7 @@ namespace hipDeviceGetPCIBusIdTests {
/**
* Fetches Gpu device count
*/
void getDeviceCount(int *pdevCnt) {
void getDeviceCount(int* pdevCnt) {
int fd[2], val = 0;
pid_t childpid;
@@ -112,14 +112,13 @@ bool testWithMaskedDevices(int actualNumGPUs) {
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
#endif
for (int count = 1;
count < actualNumGPUs; count++) {
for (int count = 1; count < actualNumGPUs; count++) {
err = hipDeviceGetPCIBusId(pciBusId, MAX_DEVICE_LENGTH, count);
if (err == hipSuccess) {
testResult &= false;
} else {
printf("hipGetDeviceProperties: Error Code Returned: '%s'(%d)\n",
hipGetErrorString(err), err);
printf("hipGetDeviceProperties: Error Code Returned: '%s'(%d)\n", hipGetErrorString(err),
err);
}
}
close(fd[0]);
@@ -143,8 +142,7 @@ bool testWithMaskedDevices(int actualNumGPUs) {
}
bool getPciBusId(int deviceCount,
char **hipDeviceList) {
bool getPciBusId(int deviceCount, char** hipDeviceList) {
for (int i = 0; i < deviceCount; i++) {
HIP_CHECK(hipDeviceGetPCIBusId(hipDeviceList[i], MAX_DEVICE_LENGTH, i));
}
@@ -175,11 +173,11 @@ TEST_CASE("Unit_hipDeviceGetPCIBusId_MaskedDevices") {
* hipDeviceGetPCIBusId vs lspci
*/
TEST_CASE("Unit_hipDeviceGetPCIBusId_CheckPciBusIDWithLspci") {
FILE *fpipe;
FILE* fpipe;
{
// Check if lspci is installed, if not, don't proceed
char const *cmd = "lspci --version";
char *lspciCheck{nullptr};
char const* cmd = "lspci --version";
char* lspciCheck{nullptr};
constexpr auto MaxLen = 50;
char temp[MaxLen]{};
@@ -199,9 +197,9 @@ TEST_CASE("Unit_hipDeviceGetPCIBusId_CheckPciBusIDWithLspci") {
HIP_CHECK(hipGetDeviceCount(&deviceCount));
REQUIRE_FALSE(deviceCount == 0);
// Allocate an array of pointer to characters
char **hipDeviceList = new char*[deviceCount];
char** hipDeviceList = new char*[deviceCount];
REQUIRE_FALSE(hipDeviceList == nullptr);
char **pciDeviceList = new char*[deviceCount];
char** pciDeviceList = new char*[deviceCount];
REQUIRE_FALSE(pciDeviceList == nullptr);
for (int i = 0; i < deviceCount; i++) {
hipDeviceList[i] = new char[MAX_DEVICE_LENGTH];
@@ -211,14 +209,16 @@ TEST_CASE("Unit_hipDeviceGetPCIBusId_CheckPciBusIDWithLspci") {
}
hipDeviceGetPCIBusIdTests::getPciBusId(deviceCount, hipDeviceList);
char const *command = nullptr;
char const* command = nullptr;
// Get lspci device list and compare with hip device list
if ((TestContext::get()).isNvidia()) {
command = "lspci -D | grep controller | grep NVIDIA | "
"cut -d ' ' -f 1";
command =
"lspci -D | grep controller | grep NVIDIA | "
"cut -d ' ' -f 1";
} else {
command = "lspci -D | grep -e controller -e accelerator | grep AMD/ATI | "
"cut -d ' ' -f 1";
command =
"lspci -D | grep -e controller -e accelerator | grep AMD/ATI | "
"cut -d ' ' -f 1";
}
fpipe = popen(command, "r");
REQUIRE_FALSE(fpipe == nullptr);
@@ -229,15 +229,13 @@ TEST_CASE("Unit_hipDeviceGetPCIBusId_CheckPciBusIDWithLspci") {
while (fgets(pciDeviceList[index], MAX_DEVICE_LENGTH, fpipe)) {
bool bMatchFound = false;
for (int deviceNo = 0; deviceNo < deviceCount; deviceNo++) {
if (!strncasecmp(pciDeviceList[index], hipDeviceList[deviceNo],
cmpLen)) {
if (!strncasecmp(pciDeviceList[index], hipDeviceList[deviceNo], cmpLen)) {
deviceMatchCount++;
bMatchFound = true;
}
}
if (bMatchFound == false) {
printf("PCI device: %s is not reported by HIP\n",
pciDeviceList[index]);
printf("PCI device: %s is not reported by HIP\n", pciDeviceList[index]);
}
index++;
if (index >= deviceCount) break;
@@ -34,7 +34,7 @@ THE SOFTWARE.
/**
* Fetches Gpu device count
*/
static void getDeviceCount(int *pdevCnt) {
static void getDeviceCount(int* pdevCnt) {
int fd[2], val = 0;
pid_t childpid;
@@ -109,15 +109,13 @@ static bool getTotalMemoryOfMaskedDevices(int actualNumGPUs) {
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
#endif
for (int count = 1;
count < actualNumGPUs; count++) {
for (int count = 1; count < actualNumGPUs; count++) {
size_t totMem;
err = hipDeviceTotalMem(&totMem, count);
if (err == hipSuccess) {
testResult &= false;
} else {
printf("hipDeviceTotalMem: Error Code Returned: '%s'(%d)\n",
hipGetErrorString(err), err);
printf("hipDeviceTotalMem: Error Code Returned: '%s'(%d)\n", hipGetErrorString(err), err);
}
}
close(fd[0]);
@@ -37,7 +37,7 @@ THE SOFTWARE.
/**
* Fetches Gpu device count
*/
static void getDeviceCount(int *pdevCnt) {
static void getDeviceCount(int* pdevCnt) {
int fd[2], val = 0;
pid_t childpid;
@@ -112,15 +112,14 @@ static bool validateGetAttributeOfMaskedDevices(int actualNumGPUs) {
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
#endif
for (int count = 1;
count < actualNumGPUs; count++) {
for (int count = 1; count < actualNumGPUs; count++) {
int pi = -1;
err = hipDeviceGetAttribute(&pi, hipDeviceAttributePciBusId, count);
if (err == hipSuccess) {
testResult &= false;
} else {
printf("hipDeviceGetAttribute: Error Code Returned: '%s'(%d)\n",
hipGetErrorString(err), err);
printf("hipDeviceGetAttribute: Error Code Returned: '%s'(%d)\n", hipGetErrorString(err),
err);
}
}
close(fd[0]);
@@ -36,7 +36,7 @@ THE SOFTWARE.
/**
* Fetches Gpu device count
*/
static void getDeviceCount(int *pdevCnt) {
static void getDeviceCount(int* pdevCnt) {
int fd[2], val = 0;
pid_t childpid;
@@ -84,7 +84,6 @@ static void getDeviceCount(int *pdevCnt) {
}
/**
* Tries to fetch device properties of masked devices and returns pass/fail.
*/
@@ -112,15 +111,14 @@ static bool validateGetPropsOfMaskedDevices(int actualNumGPUs) {
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
#endif
for (int count = 1;
count < actualNumGPUs; count++) {
for (int count = 1; count < actualNumGPUs; count++) {
hipDeviceProp_t prop;
err = hipGetDeviceProperties(&prop, count);
if (err == hipSuccess) {
testResult &= false;
} else {
printf("hipGetDeviceProperties: Error Code Returned: '%s'(%d)\n",
hipGetErrorString(err), err);
printf("hipGetDeviceProperties: Error Code Returned: '%s'(%d)\n", hipGetErrorString(err),
err);
}
}
close(fd[0]);
@@ -144,7 +142,6 @@ static bool validateGetPropsOfMaskedDevices(int actualNumGPUs) {
}
/**
* Scenario: Validate behavior of hipGetDeviceProperties for masked devices.
*/
@@ -34,8 +34,8 @@ THE SOFTWARE.
* This opaque handle may be copied into other processes and opened with hipIpcOpenEventHandle.
*/
#define BUF_SIZE 4096
#define MAX_DEVICES 16
#define BUF_SIZE 4096
#define MAX_DEVICES 16
typedef struct ipcEventInfo {
@@ -60,7 +60,7 @@ typedef struct ipcBarrier {
Get device count and list down devices with
P2P access with Device 0.
*/
void getDevices(ipcDevices_t *devices) {
void getDevices(ipcDevices_t* devices) {
pid_t pid = fork();
if (!pid) {
@@ -70,9 +70,9 @@ void getDevices(ipcDevices_t *devices) {
HIP_CHECK(hipGetDeviceCount(&devCnt));
if (devCnt < 2) {
devices->count = 0;
WARN("Count less than expected number of devices");
exit(EXIT_SUCCESS);
devices->count = 0;
WARN("Count less than expected number of devices");
exit(EXIT_SUCCESS);
}
// Device 0
@@ -85,27 +85,26 @@ void getDevices(ipcDevices_t *devices) {
int canPeerAccess_0i, canPeerAccess_i0;
for (i = 1; i < devCnt; i++) {
HIP_CHECK(hipDeviceCanAccessPeer(&canPeerAccess_0i, 0, i));
HIP_CHECK(hipDeviceCanAccessPeer(&canPeerAccess_i0, i, 0));
HIP_CHECK(hipDeviceCanAccessPeer(&canPeerAccess_0i, 0, i));
HIP_CHECK(hipDeviceCanAccessPeer(&canPeerAccess_i0, i, 0));
if (canPeerAccess_0i * canPeerAccess_i0) {
devices->ordinals[i] = i;
INFO("Two-way peer access is available between GPU"
<< devices->ordinals[0] <<" and GPU"
<< devices->ordinals[devices->count]);
devices->count += 1;
}
if (canPeerAccess_0i * canPeerAccess_i0) {
devices->ordinals[i] = i;
INFO("Two-way peer access is available between GPU" << devices->ordinals[0] << " and GPU"
<< devices->ordinals[devices->count]);
devices->count += 1;
}
}
exit(EXIT_SUCCESS);
} else {
int status;
waitpid(pid, &status, 0);
HIP_ASSERT(!status);
int status;
waitpid(pid, &status, 0);
HIP_ASSERT(!status);
}
}
static ipcBarrier_t *g_Barrier{};
static ipcBarrier_t* g_Barrier{};
static bool g_procSense;
static int g_processCnt;
@@ -121,11 +120,11 @@ void processBarrier() {
} else {
while (g_Barrier->sense == g_procSense) {
if (!g_Barrier->allExit) {
sched_yield();
} else {
exit(EXIT_FAILURE);
}
if (!g_Barrier->allExit) {
sched_yield();
} else {
exit(EXIT_FAILURE);
}
}
}
@@ -133,9 +132,9 @@ void processBarrier() {
}
__global__ void computeKernel(int *dst, int *src, int num) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
dst[idx] = src[idx] / num;
__global__ void computeKernel(int* dst, int* src, int num) {
int idx = blockIdx.x * blockDim.x + threadIdx.x;
dst[idx] = src[idx] / num;
}
/*
@@ -144,14 +143,14 @@ __global__ void computeKernel(int *dst, int *src, int num) {
* and records event.
* 3) Process 0 synchronizes event and validates the resulting buffer.
*/
void runMultiProcKernel(ipcEventInfo_t *shmEventInfo, int index) {
int *d_ptr;
void runMultiProcKernel(ipcEventInfo_t* shmEventInfo, int index) {
int* d_ptr;
int hData[BUF_SIZE]{};
unsigned int seed = time(nullptr);
// Randomize data before computation
for (int i = 0; i < BUF_SIZE; i++) {
hData[i] = rand_r(&seed);
hData[i] = rand_r(&seed);
}
HIP_CHECK(hipSetDevice(shmEventInfo[index].device));
@@ -162,8 +161,7 @@ void runMultiProcKernel(ipcEventInfo_t *shmEventInfo, int index) {
HIP_CHECK(hipMalloc(&d_ptr, BUF_SIZE * g_processCnt * sizeof(int)));
HIP_CHECK(hipIpcGetMemHandle(&shmEventInfo[0].memHandle, d_ptr));
HIP_CHECK(hipMemcpy(d_ptr, hData,
BUF_SIZE * sizeof(int), hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(d_ptr, hData, BUF_SIZE * sizeof(int), hipMemcpyHostToDevice));
// Barrier 1: Process0 will wait for all processes to create event handles,
// signals device memory creation.
@@ -181,40 +179,38 @@ void runMultiProcKernel(ipcEventInfo_t *shmEventInfo, int index) {
HIP_CHECK(hipEventSynchronize(event[i]));
}
HIP_CHECK(hipMemcpy(h_results, d_ptr + BUF_SIZE,
BUF_SIZE * (g_processCnt - 1) * sizeof(int), hipMemcpyDeviceToHost));
HIP_CHECK(hipMemcpy(h_results, d_ptr + BUF_SIZE, BUF_SIZE * (g_processCnt - 1) * sizeof(int),
hipMemcpyDeviceToHost));
// Barrier 3: Process0 signals event usage is done.
processBarrier();
HIP_CHECK(hipFree(d_ptr));
for (int n = 1; n < g_processCnt; n++) {
for (int i = 0; i < BUF_SIZE; i++) {
if (hData[i]/(n + 1) != h_results[(n-1) * BUF_SIZE + i]) {
WARN("Data validation error at index " << i << " n" << n);
g_Barrier->allExit = true;
exit(EXIT_FAILURE);
}
for (int i = 0; i < BUF_SIZE; i++) {
if (hData[i] / (n + 1) != h_results[(n - 1) * BUF_SIZE + i]) {
WARN("Data validation error at index " << i << " n" << n);
g_Barrier->allExit = true;
exit(EXIT_FAILURE);
}
}
}
for (int i = 1; i < g_processCnt; i++) {
HIP_CHECK(hipEventDestroy(event[i]));
}
} else {
hipEvent_t event;
HIP_CHECK(hipEventCreateWithFlags(&event,
hipEventDisableTiming | hipEventInterprocess));
HIP_CHECK(hipEventCreateWithFlags(&event, hipEventDisableTiming | hipEventInterprocess));
HIP_CHECK(hipIpcGetEventHandle(&shmEventInfo[index].eventHandle, event));
// Barrier 1 : wait until proc 0 initializes device memory,
// signals event creation.
processBarrier();
HIP_CHECK(hipIpcOpenMemHandle(reinterpret_cast<void **>(&d_ptr),
shmEventInfo[0].memHandle,
hipIpcMemLazyEnablePeerAccess));
HIP_CHECK(hipIpcOpenMemHandle(reinterpret_cast<void**>(&d_ptr), shmEventInfo[0].memHandle,
hipIpcMemLazyEnablePeerAccess));
const dim3 threads(512, 1);
const dim3 blocks(BUF_SIZE / threads.x, 1);
hipLaunchKernelGGL(computeKernel, dim3(blocks), dim3(threads), 0, 0,
d_ptr + index *BUF_SIZE, d_ptr, index + 1);
hipLaunchKernelGGL(computeKernel, dim3(blocks), dim3(threads), 0, 0, d_ptr + index * BUF_SIZE,
d_ptr, index + 1);
HIP_CHECK(hipGetLastError());
HIP_CHECK(hipEventRecord(event));
@@ -243,10 +239,10 @@ void runMultiProcKernel(ipcEventInfo_t *shmEventInfo, int index) {
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_hipIpcEventHandle_Functional") {
ipcDevices_t *shmDevices;
ipcEventInfo_t *shmEventInfo;
shmDevices = reinterpret_cast<ipcDevices_t *> (mmap(NULL, sizeof(*shmDevices),
PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
ipcDevices_t* shmDevices;
ipcEventInfo_t* shmEventInfo;
shmDevices = reinterpret_cast<ipcDevices_t*>(
mmap(NULL, sizeof(*shmDevices), PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
REQUIRE(MAP_FAILED != shmDevices);
getDevices(shmDevices);
@@ -259,8 +255,8 @@ TEST_CASE("Unit_hipIpcEventHandle_Functional") {
g_processCnt = (shmDevices->count > MAX_DEVICES) ? MAX_DEVICES : shmDevices->count;
// Barrier is used to synchronize processes created.
g_Barrier = reinterpret_cast<ipcBarrier_t *> (mmap(NULL, sizeof(*g_Barrier),
PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
g_Barrier = reinterpret_cast<ipcBarrier_t*>(
mmap(NULL, sizeof(*g_Barrier), PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
REQUIRE(MAP_FAILED != g_Barrier);
memset(g_Barrier, 0, sizeof(*g_Barrier));
@@ -268,9 +264,9 @@ TEST_CASE("Unit_hipIpcEventHandle_Functional") {
g_procSense = 0;
// shared memory for Event and memHandle Info
shmEventInfo = reinterpret_cast<ipcEventInfo_t *>(mmap(NULL,
g_processCnt * sizeof(*shmEventInfo),
PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
shmEventInfo = reinterpret_cast<ipcEventInfo_t*>(mmap(NULL, g_processCnt * sizeof(*shmEventInfo),
PROT_READ | PROT_WRITE,
MAP_SHARED | MAP_ANONYMOUS, 0, 0));
REQUIRE(MAP_FAILED != shmEventInfo);
// initialize shared memory
@@ -279,14 +275,14 @@ TEST_CASE("Unit_hipIpcEventHandle_Functional") {
int index = 0;
for (int i = 1; i < g_processCnt; i++) {
int pid = fork();
int pid = fork();
if (!pid) {
index = i;
break;
} else {
shmEventInfo[i].pid = pid;
}
if (!pid) {
index = i;
break;
} else {
shmEventInfo[i].pid = pid;
}
}
shmEventInfo[index].device = shmDevices->ordinals[index];
@@ -297,9 +293,9 @@ TEST_CASE("Unit_hipIpcEventHandle_Functional") {
// Cleanup
if (index == 0) {
for (int i = 1; i < g_processCnt; i++) {
int status;
waitpid(shmEventInfo[i].pid, &status, 0);
HIP_ASSERT(WIFEXITED(status));
int status;
waitpid(shmEventInfo[i].pid, &status, 0);
HIP_ASSERT(WIFEXITED(status));
}
}
}
@@ -341,8 +337,7 @@ TEST_CASE("Unit_hipIpcEventHandle_ParameterValidation") {
hipEvent_t event;
hipIpcEventHandle_t eventHandle;
hipError_t ret;
HIP_CHECK(hipEventCreateWithFlags(&event,
hipEventDisableTiming | hipEventInterprocess));
HIP_CHECK(hipEventCreateWithFlags(&event, hipEventDisableTiming | hipEventInterprocess));
#if HT_AMD
// Test disabled for nvidia due to segfault with cuda api
SECTION("Get event handle with eventHandle(nullptr)") {
@@ -371,8 +366,7 @@ TEST_CASE("Unit_hipIpcEventHandle_ParameterValidation") {
HIP_CHECK(hipEventCreateWithFlags(&eventNoIpc, hipEventDisableTiming));
ret = hipIpcGetEventHandle(&eventHandle, eventNoIpc);
if ((ret != hipErrorInvalidResourceHandle) &&
(ret != hipErrorInvalidConfiguration)) {
if ((ret != hipErrorInvalidResourceHandle) && (ret != hipErrorInvalidConfiguration)) {
INFO("Error returned : " << ret);
REQUIRE(false);
}
@@ -33,7 +33,7 @@ THE SOFTWARE.
* @{
* @ingroup DeviceTest
* `hipIpcOpenMemHandle(void** devPtr, hipIpcMemHandle_t handle, unsigned int flags)` -
* Opens an interprocess memory handle exported from another process
* Opens an interprocess memory handle exported from another process
* and returns a device pointer usable in the local process.
*/
@@ -48,7 +48,6 @@ typedef struct mem_handle {
} hip_ipc_t;
// This testcase verifies the hipIpcMemAccess APIs as follows
// The following program spawns a child process and does the following
// Parent iterate through each device, create memory -- create hipIpcMemhandle
@@ -78,7 +77,7 @@ typedef struct mem_handle {
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
hip_ipc_t *shrd_mem = NULL;
hip_ipc_t* shrd_mem = NULL;
pid_t pid;
size_t N = 1024;
size_t Nbytes = N * sizeof(int);
@@ -90,19 +89,16 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
std::string cmd_line = "rm -rf /dev/shm/sem.my-sem-object*";
int res = system(cmd_line.c_str());
REQUIRE(res != -1);
sem_ob1 = sem_open("/my-sem-object1", O_CREAT|O_EXCL, 0660, 0);
sem_ob2 = sem_open("/my-sem-object2", O_CREAT|O_EXCL, 0660, 0);
sem_ob1 = sem_open("/my-sem-object1", O_CREAT | O_EXCL, 0660, 0);
sem_ob2 = sem_open("/my-sem-object2", O_CREAT | O_EXCL, 0660, 0);
REQUIRE(sem_ob1 != SEM_FAILED);
REQUIRE(sem_ob2 != SEM_FAILED);
shrd_mem = reinterpret_cast<hip_ipc_t *>(mmap(NULL, sizeof(hip_ipc_t),
PROT_READ | PROT_WRITE,
MAP_SHARED | MAP_ANONYMOUS,
0, 0));
shrd_mem = reinterpret_cast<hip_ipc_t*>(
mmap(NULL, sizeof(hip_ipc_t), PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
REQUIRE(shrd_mem != NULL);
shrd_mem->IfTestPassed = true;
HipTest::initArrays<int>(nullptr, nullptr, nullptr,
&A_h, nullptr, &C_h, N, false);
HipTest::initArrays<int>(nullptr, nullptr, nullptr, &A_h, nullptr, &C_h, N, false);
pid = fork();
if (pid != 0) {
// Parent process
@@ -111,9 +107,8 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
if (shrd_mem->IfTestPassed == true) {
HIP_CHECK(hipSetDevice(i));
HIP_CHECK(hipMalloc(&A_d, Nbytes));
HIP_CHECK(hipIpcGetMemHandle(reinterpret_cast<hipIpcMemHandle_t *>
(&shrd_mem->memHandle),
A_d));
HIP_CHECK(
hipIpcGetMemHandle(reinterpret_cast<hipIpcMemHandle_t*>(&shrd_mem->memHandle), A_d));
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
shrd_mem->device = i;
if ((sem_post(sem_ob1)) == -1) {
@@ -132,7 +127,7 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
// Child process
HIP_CHECK(hipGetDeviceCount(&Num_devices));
for (int j = 0; j < Num_devices; ++j) {
HIP_CHECK(hipSetDevice(j));
HIP_CHECK(hipSetDevice(j));
if ((sem_wait(sem_ob1)) == -1) {
shrd_mem->IfTestPassed = false;
WARN("sem_wait() call failed in child process.");
@@ -147,8 +142,7 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
HIP_CHECK(hipDeviceCanAccessPeer(&CanAccessPeer, i, shrd_mem->device));
if (CanAccessPeer == 1) {
HIP_CHECK(hipMalloc(&C_d, Nbytes));
HIP_CHECK(hipIpcOpenMemHandle(reinterpret_cast<void **>(&B_d),
shrd_mem->memHandle,
HIP_CHECK(hipIpcOpenMemHandle(reinterpret_cast<void**>(&B_d), shrd_mem->memHandle,
hipIpcMemLazyEnablePeerAccess));
HIP_CHECK(hipMemcpy(C_d, B_d, Nbytes, hipMemcpyDeviceToDevice));
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
@@ -157,8 +151,8 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
// Checking if the data obtained from Ipc shared memory is consistent
HIP_CHECK(hipMemcpy(C_h, B_d, Nbytes, hipMemcpyDeviceToHost));
HipTest::checkTest<int>(A_h, C_h, N);
HIP_CHECK(hipIpcCloseMemHandle(reinterpret_cast<void*>(B_d)));
HIP_CHECK(hipFree(C_d));
HIP_CHECK(hipIpcCloseMemHandle(reinterpret_cast<void*>(B_d)));
HIP_CHECK(hipFree(C_d));
}
}
if ((sem_post(sem_ob2)) == -1) {
@@ -178,8 +172,7 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
int rFlag = 0;
waitpid(pid, &rFlag, 0);
REQUIRE(shrd_mem->IfTestPassed == true);
HipTest::freeArrays<int>(nullptr, nullptr, nullptr,
A_h, nullptr, C_h, false);
HipTest::freeArrays<int>(nullptr, nullptr, nullptr, A_h, nullptr, C_h, false);
}
/**
@@ -246,14 +239,12 @@ TEST_CASE("Unit_hipIpcMemAccess_ParameterValidation") {
}
SECTION("Open mem handle with devptr as nullptr") {
ret = hipIpcOpenMemHandle(nullptr, MemHandle,
hipIpcMemLazyEnablePeerAccess);
ret = hipIpcOpenMemHandle(nullptr, MemHandle, hipIpcMemLazyEnablePeerAccess);
REQUIRE(ret == hipErrorInvalidValue);
}
SECTION("Open mem handle with handle as un-initialized") {
ret = hipIpcOpenMemHandle(&Ad2, MemHandleUninit,
hipIpcMemLazyEnablePeerAccess);
ret = hipIpcOpenMemHandle(&Ad2, MemHandleUninit, hipIpcMemLazyEnablePeerAccess);
REQUIRE((ret == hipErrorInvalidValue || ret == hipErrorInvalidDevicePointer));
}
#if HT_AMD
@@ -106,9 +106,11 @@ static bool validateMemoryOnGPU(int gpu, bool concurOnOneGPU = false) {
HIP_CHECK(hipMemGetInfo(&curAvl, &curTot));
if (!concurOnOneGPU && (prevAvl < curAvl || prevTot != curTot)) {
//In concurrent calls on one GPU, we cannot verify leaking in this way
printf("%s : Memory allocation mismatch observed."
"Possible memory leak.\n", __func__);
// In concurrent calls on one GPU, we cannot verify leaking in this way
printf(
"%s : Memory allocation mismatch observed."
"Possible memory leak.\n",
__func__);
TestPassed &= false;
}
@@ -117,9 +119,8 @@ static bool validateMemoryOnGPU(int gpu, bool concurOnOneGPU = false) {
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(B_d, B_h, Nbytes, hipMemcpyHostToDevice));
hipLaunchKernelGGL(HipTest::vectorADD, dim3(blocks), dim3(threadsPerBlock),
0, 0, static_cast<const int*>(A_d),
static_cast<const int*>(B_d), C_d, N);
hipLaunchKernelGGL(HipTest::vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, 0,
static_cast<const int*>(A_d), static_cast<const int*>(B_d), C_d, N);
HIP_CHECK(hipGetLastError());
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
@@ -189,8 +190,7 @@ TEST_CASE("Unit_hipMalloc_ChildConcurrencyDefaultGpu") {
// Wait and get result from child
pid = wait(&exitStatus);
if ((WEXITSTATUS(exitStatus) == resFailure) || (pid < 0))
TestPassed = false;
if ((WEXITSTATUS(exitStatus) == resFailure) || (pid < 0)) TestPassed = false;
}
REQUIRE(TestPassed == true);
@@ -41,7 +41,7 @@
#include <chrono>
#include "../unit/memory/hipSVMCommon.h"
__global__ void CoherentTst(int *ptr, volatile unsigned int *expired) {
__global__ void CoherentTst(int* ptr, volatile unsigned int* expired) {
// Incrementing the value by 1
atomicAdd_system(ptr, 1);
// The following while loop checks the value until expiration.
@@ -50,43 +50,42 @@ __global__ void CoherentTst(int *ptr, volatile unsigned int *expired) {
}
}
__global__ void SquareKrnl(int *ptr) {
__global__ void SquareKrnl(int* ptr) {
// ptr value squared here
*ptr = (*ptr) * (*ptr);
}
// The function tests the coherency of allocated memory
// Return false on failure, true on success.
bool static TstCoherency(int *Ptr, bool HmmMem) {
bool static TstCoherency(int* Ptr, bool HmmMem) {
using namespace std::chrono_literals;
int *Dptr = nullptr;
int* Dptr = nullptr;
hipStream_t strm;
HIP_CHECK(hipStreamCreate(&strm));
// storing value 1 in the memory created above
*Ptr = 1;
unsigned int *expired = nullptr;
HIP_CHECK(hipHostMalloc(&expired, sizeof(unsigned int))); // hipHostMallocCoherent by defaut
unsigned int* expired = nullptr;
HIP_CHECK(hipHostMalloc(&expired, sizeof(unsigned int))); // hipHostMallocCoherent by defaut
*expired = 0;
if (!HmmMem) {
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void **>(&Dptr), Ptr, 0));
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&Dptr), Ptr, 0));
CoherentTst<<<1, 1, 0, strm>>>(Dptr, expired);
} else {
CoherentTst<<<1, 1, 0, strm>>>(Ptr, expired);
}
// looping until the value is 2 for 3 seconds
std::chrono::steady_clock::time_point start =
std::chrono::steady_clock::now();
while (std::chrono::duration_cast<std::chrono::seconds>(
std::chrono::steady_clock::now() - start).count() < 3) {
std::chrono::steady_clock::time_point start = std::chrono::steady_clock::now();
while (std::chrono::duration_cast<std::chrono::seconds>(std::chrono::steady_clock::now() - start)
.count() < 3) {
if (*Ptr == 2) {
*Ptr += 1;
std::this_thread::sleep_for(200ms); // Make sure kernel gets updated Dptr
std::this_thread::sleep_for(200ms); // Make sure kernel gets updated Dptr
break;
}
}
*expired = 1; // Notify kernel loop to exit
*expired = 1; // Notify kernel loop to exit
HIP_CHECK(hipStreamSynchronize(strm));
HIP_CHECK(hipStreamDestroy(strm));
HIP_CHECK(hipHostFree(expired));
@@ -106,13 +105,12 @@ TEST_CASE("Unit_malloc_CoherentTst") {
CHECK_PCIE_ATOMICS_SUPPORT
hipDeviceProp_t prop;
HIPCHECK(hipGetDeviceProperties(&prop, 0));
char *p = NULL;
char* p = NULL;
p = strstr(prop.gcnArchName, "xnack+");
if (p) {
// Test Case execution begins from here
int managed = 0;
HIPCHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory,
0));
HIPCHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory, 0));
if (managed == 1) {
int *Ptr = nullptr, SIZE = sizeof(int);
bool HmmMem = true;
@@ -122,7 +120,7 @@ TEST_CASE("Unit_malloc_CoherentTst") {
auto ret = TstCoherency(Ptr, HmmMem);
free(Ptr);
REQUIRE(ret);
}
}
} else {
HipTest::HIP_SKIP_TEST("GPU is not xnack enabled hence skipping the test...\n");
}
@@ -137,12 +135,11 @@ TEST_CASE("Unit_malloc_CoherentTst") {
TEST_CASE("Unit_malloc_CoherentTstWthAdvise") {
hipDeviceProp_t prop;
HIPCHECK(hipGetDeviceProperties(&prop, 0));
char *p = NULL;
char* p = NULL;
p = strstr(prop.gcnArchName, "xnack+");
if (p) {
int managed = 0;
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory,
0));
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory, 0));
if (managed == 1) {
int *Ptr = nullptr, SIZE = sizeof(int);
@@ -154,7 +151,7 @@ TEST_CASE("Unit_malloc_CoherentTstWthAdvise") {
SquareKrnl<<<1, 1, 0, strm>>>(Ptr);
HIP_CHECK(hipStreamSynchronize(strm));
HIP_CHECK(hipStreamDestroy(strm));
REQUIRE (*Ptr == 16);
REQUIRE(*Ptr == 16);
}
} else {
HipTest::HIP_SKIP_TEST("GPU is not xnack enabled hence skipping the test...\n");
@@ -170,17 +167,15 @@ TEST_CASE("Unit_mmap_CoherentTst") {
CHECK_PCIE_ATOMICS_SUPPORT
hipDeviceProp_t prop;
HIPCHECK(hipGetDeviceProperties(&prop, 0));
char *p = NULL;
char* p = NULL;
p = strstr(prop.gcnArchName, "xnack+");
if (p) {
int managed = 0;
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory,
0));
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory, 0));
if (managed == 1) {
bool HmmMem = true;
int *Ptr = reinterpret_cast<int*>(mmap(NULL, sizeof(int),
PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS, 0, 0));
int* Ptr = reinterpret_cast<int*>(
mmap(NULL, sizeof(int), PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, 0, 0));
if (Ptr == MAP_FAILED) {
WARN("Mapping Failed\n");
REQUIRE(false);
@@ -191,7 +186,7 @@ TEST_CASE("Unit_mmap_CoherentTst") {
WARN("munmap failed\n");
}
REQUIRE(ret);
}
}
} else {
HipTest::HIP_SKIP_TEST("GPU is not xnack enabled hence skipping the test...\n");
}
@@ -205,17 +200,15 @@ TEST_CASE("Unit_mmap_CoherentTst") {
TEST_CASE("Unit_mmap_CoherentTstWthAdvise") {
hipDeviceProp_t prop;
HIPCHECK(hipGetDeviceProperties(&prop, 0));
char *p = NULL;
char* p = NULL;
p = strstr(prop.gcnArchName, "xnack+");
if (p) {
int managed = 0;
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory,
0));
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory, 0));
if (managed == 1) {
int SIZE = sizeof(int);
int *Ptr = reinterpret_cast<int*>(mmap(NULL, SIZE,
PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS, 0, 0));
int* Ptr = reinterpret_cast<int*>(
mmap(NULL, SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, 0, 0));
if (Ptr == MAP_FAILED) {
WARN("Mapping Failed\n");
REQUIRE(false);
@@ -230,13 +223,13 @@ TEST_CASE("Unit_mmap_CoherentTstWthAdvise") {
bool IfTstPassed = false;
if (*Ptr == 81) {
IfTstPassed = true;
}
}
int err = munmap(Ptr, SIZE);
if (err != 0) {
WARN("munmap failed\n");
}
REQUIRE(IfTstPassed);
}
}
} else {
HipTest::HIP_SKIP_TEST("GPU is not xnack enabled hence skipping the test...\n");
}
@@ -249,8 +242,8 @@ TEST_CASE("Unit_mmap_CoherentTstWthAdvise") {
#if HT_AMD
TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg1") {
if ((setenv("HIP_HOST_COHERENT", "0", 1)) != 0) {
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
}
int stat = 0;
if (fork() == 0) {
@@ -289,8 +282,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg1") {
#if HT_AMD
TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg2") {
if ((setenv("HIP_HOST_COHERENT", "0", 1)) != 0) {
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
}
int stat = 0;
if (fork() == 0) {
@@ -329,8 +322,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg2") {
#if HT_AMD
TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg3") {
if ((setenv("HIP_HOST_COHERENT", "0", 1)) != 0) {
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
}
int stat = 0;
if (fork() == 0) {
@@ -369,8 +362,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg3") {
#if HT_AMD
TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg4") {
if ((setenv("HIP_HOST_COHERENT", "0", 1)) != 0) {
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
}
int stat = 0;
if (fork() == 0) {
@@ -410,8 +403,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg4") {
#if HT_AMD
TEST_CASE("Unit_hipHostMalloc_WthEnv1") {
if ((setenv("HIP_HOST_COHERENT", "1", 1)) != 0) {
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
}
int stat = 0;
if (fork() == 0) { // child process
@@ -439,8 +432,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv1") {
#if HT_AMD
TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg1") {
if ((setenv("HIP_HOST_COHERENT", "1", 1)) != 0) {
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
}
int stat = 0;
if (fork() == 0) { // child process
@@ -467,8 +460,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg1") {
#if HT_AMD
TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg2") {
if ((setenv("HIP_HOST_COHERENT", "1", 1)) != 0) {
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
}
int stat = 0;
if (fork() == 0) { // child process
@@ -495,8 +488,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg2") {
#if HT_AMD
TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg3") {
if ((setenv("HIP_HOST_COHERENT", "1", 1)) != 0) {
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
REQUIRE(false);
}
int stat = 0;
if (fork() == 0) { // child process
@@ -24,16 +24,16 @@ THE SOFTWARE.
#include <sys/wait.h>
#include <sys/types.h>
#define ReadEnd 0
#define ReadEnd 0
#define WriteEnd 1
#define MAX_SIZE 32
#define FREE_MEM_TO_HIDE 4294967296
#define SIZE_TO_ALLOCATE 2147483648
/*
* In main process allocate 2 GB of device memory.
* Fork() a child process and verify that 2 GB has been
* allocated in parent process.
*/
* In main process allocate 2 GB of device memory.
* Fork() a child process and verify that 2 GB has been
* allocated in parent process.
*/
TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario1") {
constexpr size_t size = 2147483648; // 2GB
int fd[2], fd1[2], status;
@@ -42,10 +42,10 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario1") {
status = pipe(fd1);
REQUIRE(status == 0);
pid_t child_pid;
child_pid = fork(); // Create a new child process
child_pid = fork(); // Create a new child process
if (child_pid < 0) {
WARN("Fork failed!!!!");
} else if (child_pid == 0) { // child
} else if (child_pid == 0) { // child
close(fd1[WriteEnd]);
close(fd[ReadEnd]);
int result;
@@ -67,7 +67,7 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario1") {
REQUIRE(status != -1);
close(fd[WriteEnd]);
exit(0);
} else { // Parent
} else { // Parent
close(fd1[ReadEnd]);
close(fd[WriteEnd]);
// Allocate memory
@@ -90,10 +90,10 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario1") {
}
}
/**
* From main process Fork() a child process. In the child process allocate
* 2 GB of device memory. Signal the parent process. Verify from the parent
* process that 2 GB is allocated in the child process.
*/
* From main process Fork() a child process. In the child process allocate
* 2 GB of device memory. Signal the parent process. Verify from the parent
* process that 2 GB is allocated in the child process.
*/
TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
constexpr size_t size = 2147483648; // 2GB
int fd[2], fd2[2], status;
@@ -102,10 +102,10 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
status = pipe(fd2);
REQUIRE(status == 0);
pid_t child_pid;
child_pid = fork(); // Create a new child process
child_pid = fork(); // Create a new child process
if (child_pid < 0) {
WARN("Fork failed!!!!");
} else if (child_pid == 0) { // Child
} else if (child_pid == 0) { // Child
close(fd[ReadEnd]);
close(fd2[WriteEnd]);
// Allocate memory
@@ -124,7 +124,7 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
// Free allocated device memory
HIP_CHECK(hipFree(A_d));
exit(0);
} else { // Parent
} else { // Parent
size_t free = 0, total = 0;
close(fd[WriteEnd]);
close(fd2[ReadEnd]);
@@ -134,7 +134,7 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
REQUIRE(status != -1);
close(fd[ReadEnd]);
// Verify the memory
HIP_CHECK(hipMemGetInfo(&free , &total));
HIP_CHECK(hipMemGetInfo(&free, &total));
REQUIRE((total - free) >= size);
// Signal child that validation is over and child can free memory
int valid = 0;
@@ -146,21 +146,21 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
}
}
/*
* From main process Fork() a child process. In the child process
* allocate 2 GB of device memory. Free the memory and exit from
* child process. Verify from the parent process that 2 GB is
* freed in the child process.
*/
* From main process Fork() a child process. In the child process
* allocate 2 GB of device memory. Free the memory and exit from
* child process. Verify from the parent process that 2 GB is
* freed in the child process.
*/
TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario3") {
constexpr size_t size = 2147483648; // 2GB
int fd[2], status;
status = pipe(fd);
REQUIRE(status == 0);
pid_t child_pid;
child_pid = fork(); // Create a new child process
child_pid = fork(); // Create a new child process
if (child_pid < 0) {
WARN("Fork failed!!!!");
} else if (child_pid == 0) { // Child
} else if (child_pid == 0) { // Child
close(fd[ReadEnd]);
// Allocate the memory
void* A_d = nullptr;
@@ -173,7 +173,7 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario3") {
REQUIRE(status != -1);
close(fd[WriteEnd]);
exit(0);
} else { // Parent
} else { // Parent
close(fd[WriteEnd]);
// Wait for the signal from child about memory free
int check_parent;
@@ -182,43 +182,43 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario3") {
close(fd[ReadEnd]);
size_t free = 0, total = 0;
// Verify the memory
HIP_CHECK(hipMemGetInfo(&free , &total));
HIP_CHECK(hipMemGetInfo(&free, &total));
REQUIRE((total - free) >= 0);
// wait for child exit
wait(NULL);
}
}
/*
* From main process Fork() a child process. In the child process allocate
* 2 GB of device memory. Exit from child process. Verify from the parent
* process that 2 GB is freed in the child process.
*/
* From main process Fork() a child process. In the child process allocate
* 2 GB of device memory. Exit from child process. Verify from the parent
* process that 2 GB is freed in the child process.
*/
TEST_CASE("Unit_hipMemGetInfo_Functional_scenario4") {
constexpr size_t size = 2147483648; // 2GB
pid_t child_pid;
child_pid = fork(); // Create a new child process
child_pid = fork(); // Create a new child process
if (child_pid < 0) {
WARN("Fork failed!!!!");
} else if (child_pid == 0) { // Child
} else if (child_pid == 0) { // Child
// Allocate the memory
void* A_d = nullptr;
HIP_CHECK(hipMalloc(&A_d, size));
exit(0);
} else { // Parent
} else { // Parent
// wait for child exit
wait(NULL);
size_t free = 0, total = 0;
// Verify the memory
HIP_CHECK(hipMemGetInfo(&free , &total));
REQUIRE((total-free) >= 0);
HIP_CHECK(hipMemGetInfo(&free, &total));
REQUIRE((total - free) >= 0);
}
}
/*
* Multidevice Scenario: In main process allocate 2 GB of device memory
* in every device. Verify that 2 GB is allocated using hipMemGetInfo.
* Fork() a child process and verify that 2 GB has been allocated from
* parent process in every device.
*/
* Multidevice Scenario: In main process allocate 2 GB of device memory
* in every device. Verify that 2 GB is allocated using hipMemGetInfo.
* Fork() a child process and verify that 2 GB has been allocated from
* parent process in every device.
*/
TEST_CASE("Unit_hipMemGetInfo_Functional_MultiDevice_Scenario5") {
constexpr size_t size = 2147483648; // 2GB
size_t free = 0, total = 0;
@@ -228,29 +228,29 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_MultiDevice_Scenario5") {
status = pipe(fd2);
REQUIRE(status == 0);
pid_t child_pid;
child_pid = fork(); // Create a new child process
child_pid = fork(); // Create a new child process
if (child_pid < 0) {
WARN("Fork failed!!!!");
} else if (child_pid == 0) { // Child
close(fd1[WriteEnd]);
close(fd2[ReadEnd]);
// Wait for the signal from parent after memory allocatoin
int check_child;
status = read(fd1[ReadEnd], &check_child, sizeof(check_child));
REQUIRE(status != -1);
close(fd1[ReadEnd]);
int num_devices, result, count = 0;
// Get the device count
HIP_CHECK(hipGetDeviceCount(&num_devices));
for (int i = 0; i < num_devices; i++) {
HIP_CHECK(hipSetDevice(i));
// Check the memory
HIP_CHECK(hipMemGetInfo(&free , &total));
if ((total - free) >= size) {
count+=1;
}
} else if (child_pid == 0) { // Child
close(fd1[WriteEnd]);
close(fd2[ReadEnd]);
// Wait for the signal from parent after memory allocatoin
int check_child;
status = read(fd1[ReadEnd], &check_child, sizeof(check_child));
REQUIRE(status != -1);
close(fd1[ReadEnd]);
int num_devices, result, count = 0;
// Get the device count
HIP_CHECK(hipGetDeviceCount(&num_devices));
for (int i = 0; i < num_devices; i++) {
HIP_CHECK(hipSetDevice(i));
// Check the memory
HIP_CHECK(hipMemGetInfo(&free, &total));
if ((total - free) >= size) {
count += 1;
}
}
if ( count == num_devices ) {
if (count == num_devices) {
result = 1;
} else {
result = 0;
@@ -260,21 +260,21 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_MultiDevice_Scenario5") {
REQUIRE(status != -1);
close(fd2[WriteEnd]);
exit(0);
} else { // Parent
} else { // Parent
close(fd1[ReadEnd]);
close(fd2[WriteEnd]);
int num_devices;
// Get the device count
HIP_CHECK(hipGetDeviceCount(&num_devices));
std::vector<void*>v(num_devices, nullptr);
std::vector<void*> v(num_devices, nullptr);
for (int i = 0; i < num_devices; i++) {
HIP_CHECK(hipSetDevice(i));
// verify the memory
HIP_CHECK(hipMemGetInfo(&free , &total));
HIP_CHECK(hipMemGetInfo(&free, &total));
// Allocate memory
HIP_CHECK(hipMalloc(&v[i], size));
// Verify the memory
HIP_CHECK(hipMemGetInfo(&free , &total));
HIP_CHECK(hipMemGetInfo(&free, &total));
}
// Signal the child about memory allocation
int check = 0;
@@ -310,7 +310,7 @@ static bool testHiddenFreeMemFromChild() {
size_t free = 0, total = 0, min_size = 0;
close(fd_c2p[ReadEnd]);
close(fd_p2c[WriteEnd]);
int64_t size_tohide = (FREE_MEM_TO_HIDE/(1024*1024)); // in MB
int64_t size_tohide = (FREE_MEM_TO_HIDE / (1024 * 1024)); // in MB
// set environment variable from shell
unsetenv("HIP_HIDDEN_FREE_MEM");
setenv("HIP_HIDDEN_FREE_MEM", std::to_string(size_tohide).c_str(), 1);
@@ -377,7 +377,7 @@ TEST_CASE("Unit_hipMemGetInfo_SetHiddenFreeMemFromChild") {
*/
TEST_CASE("Unit_hipMemGetInfo_VerifyHiddenFreeMemForAllGpu") {
int numDevices = 0;
int64_t size_tohide = (FREE_MEM_TO_HIDE/(1024*1024)); // in MB
int64_t size_tohide = (FREE_MEM_TO_HIDE / (1024 * 1024)); // in MB
// set environment variable from shell
unsetenv("HIP_HIDDEN_FREE_MEM");
setenv("HIP_HIDDEN_FREE_MEM", std::to_string(size_tohide).c_str(), 1);
File diff suppressed because it is too large Load Diff
@@ -36,7 +36,7 @@
/**
* Fetches Gpu device count
*/
static void getDeviceCount(int *pdevCnt) {
static void getDeviceCount(int* pdevCnt) {
int fd[2], val = 0;
pid_t childpid;
@@ -82,8 +82,7 @@ static void getDeviceCount(int *pdevCnt) {
// Pass either -1 in deviceNumber or invalid device number
static void testInvalidDevice(int numDevices, bool useRocrEnv,
int deviceNumber) {
static void testInvalidDevice(int numDevices, bool useRocrEnv, int deviceNumber) {
bool testResult = true;
int device;
int tempCount = 0;
@@ -117,24 +116,24 @@ static void testInvalidDevice(int numDevices, bool useRocrEnv,
for (int i = 0; i < numDevices; i++) {
err = hipSetDevice(i);
if (err != hipSuccess) {
setDeviceErrorCheck+= 1;
setDeviceErrorCheck += 1;
}
err = hipGetDevice(&device);
if (err != hipSuccess) {
getDeviceErrorCheck+= 1;
getDeviceErrorCheck += 1;
}
}
if ((getDeviceCountErrorCheck == 1) && (setDeviceErrorCheck == numDevices)
&& (getDeviceErrorCheck == numDevices)) {
if ((getDeviceCountErrorCheck == 1) && (setDeviceErrorCheck == numDevices) &&
(getDeviceErrorCheck == numDevices)) {
testResult = true;
} else {
printf("Test failed for invalid device, getDeviceCountErrorCheck %d,"
"setDeviceErrorCheck %d, getDeviceErrorCheck %d\n",
getDeviceCountErrorCheck, setDeviceErrorCheck,
getDeviceErrorCheck);
printf(
"Test failed for invalid device, getDeviceCountErrorCheck %d,"
"setDeviceErrorCheck %d, getDeviceErrorCheck %d\n",
getDeviceCountErrorCheck, setDeviceErrorCheck, getDeviceErrorCheck);
testResult = false;
}
@@ -159,19 +158,18 @@ static void testInvalidDevice(int numDevices, bool useRocrEnv,
}
static void testValidDevices(int numDevices, bool useRocrEnv, int *deviceList,
int deviceListLength) {
static void testValidDevices(int numDevices, bool useRocrEnv, int* deviceList,
int deviceListLength) {
bool testResult = true;
int tempCount = 0;
int device;
int setDeviceErrorCheck = 0;
int getDeviceErrorCheck = 0;
int getDeviceCountErrorCheck = 0;
int *deviceListPtr = deviceList;
int* deviceListPtr = deviceList;
std::string visibleDeviceString;
if ((NULL == deviceList) || ((deviceListLength < 1) ||
deviceListLength > numDevices)) {
if ((NULL == deviceList) || ((deviceListLength < 1) || deviceListLength > numDevices)) {
INFO("Invalid argument for number of devices. Skipping current test");
REQUIRE(false);
}
@@ -213,17 +211,17 @@ static void testValidDevices(int numDevices, bool useRocrEnv, int *deviceList,
for (int i = 0; i < numDevices; i++) {
err = hipSetDevice(i);
if (err != hipSuccess) {
setDeviceErrorCheck+= 1;
setDeviceErrorCheck += 1;
}
err = hipGetDevice(&device);
if (err != hipSuccess) {
getDeviceErrorCheck+= 1;
getDeviceErrorCheck += 1;
}
}
if ((getDeviceCountErrorCheck == 1) && (setDeviceErrorCheck ==
(numDevices-deviceListLength)) && (getDeviceErrorCheck == 0)) {
if ((getDeviceCountErrorCheck == 1) &&
(setDeviceErrorCheck == (numDevices - deviceListLength)) && (getDeviceErrorCheck == 0)) {
testResult = true;
} else {
@@ -251,19 +249,19 @@ static void testValidDevices(int numDevices, bool useRocrEnv, int *deviceList,
}
static void Initialize(int *deviceList, int numDevices, int count,
std::string& min_visibleDeviceString, std::string& max_visibleDeviceString) {
int *deviceListPtr = deviceList;
for (int i =0; i < count; i++) {
if (i == count-1) {
static void Initialize(int* deviceList, int numDevices, int count,
std::string& min_visibleDeviceString, std::string& max_visibleDeviceString) {
int* deviceListPtr = deviceList;
for (int i = 0; i < count; i++) {
if (i == count - 1) {
min_visibleDeviceString.append(std::to_string(*deviceListPtr++));
} else {
min_visibleDeviceString.append(std::to_string(*deviceListPtr++) + ",");
}
}
for (int i =0; i < numDevices; i++) {
if (i == numDevices-1) {
for (int i = 0; i < numDevices; i++) {
if (i == numDevices - 1) {
max_visibleDeviceString.append(std::to_string(i));
} else {
max_visibleDeviceString.append(std::to_string(i) + ",");
@@ -271,7 +269,7 @@ static void Initialize(int *deviceList, int numDevices, int count,
}
}
static void testMaxRvdMinHvd(int numDevices, int *deviceList, int count) {
static void testMaxRvdMinHvd(int numDevices, int* deviceList, int count) {
bool testResult = true;
int device;
int validateCount = 0;
@@ -282,8 +280,7 @@ static void testMaxRvdMinHvd(int numDevices, int *deviceList, int count) {
pid_t cPid;
cPid = fork();
if (cPid == 0) { // child
Initialize(deviceList, numDevices,
count, min_visibleDeviceString, max_visibleDeviceString);
Initialize(deviceList, numDevices, count, min_visibleDeviceString, max_visibleDeviceString);
unsetenv("ROCR_VISIBLE_DEVICES");
unsetenv("HIP_VISIBLE_DEVICES");
setenv("ROCR_VISIBLE_DEVICES", max_visibleDeviceString.c_str(), 1);
@@ -293,7 +290,7 @@ static void testMaxRvdMinHvd(int numDevices, int *deviceList, int count) {
HIP_CHECK(hipSetDevice(i));
HIP_CHECK(hipGetDevice(&device));
if (device == i) {
validateCount+= 1;
validateCount += 1;
}
}
if (count != validateCount) {
@@ -312,19 +309,19 @@ static void testMaxRvdMinHvd(int numDevices, int *deviceList, int count) {
REQUIRE(testResult == true);
}
static void testRvdCvd(int numDevices, int *deviceList, int count) {
static void testRvdCvd(int numDevices, int* deviceList, int count) {
bool testResult = true;
int device;
int validateCount = 0;
std::string min_visibleDeviceString;
std::string max_visibleDeviceString;;
std::string max_visibleDeviceString;
;
int fd[2];
pipe(fd);
pid_t cPid;
cPid = fork();
if (cPid == 0) { // child
Initialize(deviceList, numDevices, count,
min_visibleDeviceString, max_visibleDeviceString);
Initialize(deviceList, numDevices, count, min_visibleDeviceString, max_visibleDeviceString);
unsetenv("ROCR_VISIBLE_DEVICES");
unsetenv("HIP_VISIBLE_DEVICES");
setenv("ROCR_VISIBLE_DEVICES", max_visibleDeviceString.c_str(), 1);
@@ -334,7 +331,7 @@ static void testRvdCvd(int numDevices, int *deviceList, int count) {
HIP_CHECK(hipSetDevice(i));
HIP_CHECK(hipGetDevice(&device));
if (device == i) {
validateCount+= 1;
validateCount += 1;
}
}
if (count != validateCount) {
@@ -353,7 +350,7 @@ static void testRvdCvd(int numDevices, int *deviceList, int count) {
REQUIRE(testResult == true);
}
static void testMinRvdMaxHvd(int numDevices, int *deviceList, int count) {
static void testMinRvdMaxHvd(int numDevices, int* deviceList, int count) {
bool testResult = true;
int device;
int validateCount = 0;
@@ -364,8 +361,7 @@ static void testMinRvdMaxHvd(int numDevices, int *deviceList, int count) {
pid_t cPid;
cPid = fork();
if (cPid == 0) { // child
Initialize(deviceList, numDevices, count,
min_visibleDeviceString, max_visibleDeviceString);
Initialize(deviceList, numDevices, count, min_visibleDeviceString, max_visibleDeviceString);
unsetenv("ROCR_VISIBLE_DEVICES");
unsetenv("HIP_VISIBLE_DEVICES");
setenv("ROCR_VISIBLE_DEVICES", min_visibleDeviceString.c_str(), 1);
@@ -375,7 +371,7 @@ static void testMinRvdMaxHvd(int numDevices, int *deviceList, int count) {
HIP_CHECK(hipSetDevice(i));
HIP_CHECK(hipGetDevice(&device));
if (device == i) {
validateCount+= 1;
validateCount += 1;
}
}
if (count != validateCount) {
@@ -407,17 +403,13 @@ TEST_CASE("Unit_hipSetDevice_InvalidVisibleDeviceList") {
getDeviceCount(&numDevices);
REQUIRE(numDevices != 0);
SECTION("Test setting -1 to HIP_VISIBLE_DEVICES") {
testInvalidDevice(numDevices, false, -1);
}
SECTION("Test setting -1 to HIP_VISIBLE_DEVICES") { testInvalidDevice(numDevices, false, -1); }
SECTION("Test setting invalid device to HIP_VISIBLE_DEVICES") {
testInvalidDevice(numDevices, false, numDevices);
}
#ifndef __HIP_PLATFORM_NVIDIA__
SECTION("Test setting -1 to ROCR_VISIBLE_DEVICES") {
testInvalidDevice(numDevices, true, -1);
}
SECTION("Test setting -1 to ROCR_VISIBLE_DEVICES") { testInvalidDevice(numDevices, true, -1); }
SECTION("Test setting invalid device to ROCR_VISIBLE_DEVICES") {
testInvalidDevice(numDevices, true, numDevices);
@@ -462,16 +454,14 @@ TEST_CASE("Unit_hipSetDevice_SubsetOfAvailableDevices") {
REQUIRE(numDevices != 0);
// Test for subset of available gpus
for (int i=0; i < deviceListLength; i++) {
deviceList[i] = deviceListLength-1-i;
for (int i = 0; i < deviceListLength; i++) {
deviceList[i] = deviceListLength - 1 - i;
}
#ifndef __HIP_PLATFORM_NVIDIA__
testValidDevices(numDevices, true, deviceList,
deviceListLength);
testValidDevices(numDevices, true, deviceList, deviceListLength);
#endif
testValidDevices(numDevices, false, deviceList,
deviceListLength);
testValidDevices(numDevices, false, deviceList, deviceListLength);
}
#ifndef __HIP_PLATFORM_NVIDIA__
@@ -494,8 +484,8 @@ TEST_CASE("Unit_hipSetDevice_MinRvdMaxHvdDevicesList") {
deviceList.push_back(0);
count = 1;
} else {
for (int i=0; i < numDevices; i++) {
if (i%2 == 0) {
for (int i = 0; i < numDevices; i++) {
if (i % 2 == 0) {
deviceList.push_back(i);
count++;
}
@@ -520,8 +510,8 @@ TEST_CASE("Unit_hipSetDevice_MaxRvdMinHvdDevicesList") {
if (numDevices == 1) {
deviceList.push_back(0);
} else {
for (int i=0; i < numDevices; i++) {
if (i%2 == 0) {
for (int i = 0; i < numDevices; i++) {
if (i % 2 == 0) {
deviceList.push_back(i);
}
}
@@ -546,8 +536,8 @@ TEST_CASE("Unit_hipSetDevice_RvdCvdDevicesList") {
deviceList[0] = 0;
count = 1;
} else {
for (int i=0; i < numDevices; i++) {
if (i%2 == 0) {
for (int i = 0; i < numDevices; i++) {
if (i % 2 == 0) {
deviceList[count] = i;
count++;
}
@@ -56,6 +56,6 @@ TEST_CASE("Performance_hipEventCreate") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -83,6 +83,6 @@ TEST_CASE("Performance_hipEventCreateWithFlags") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -21,15 +21,15 @@ THE SOFTWARE.
#include <hip_test_checkers.hh>
#include <hip/hip_ext.h>
#include <hip_array_common.hh>
#define THREADS_PER_BLOCK 64 // 64 threads per wave on Mi300, 32 threads per wave on Navi31
#define MAXITERS 100000 // maximum iteration number in a thread
#define FUNC1(K, I) (K + K * I / 37 - I) / 1091; // Shared by kernel and host
#define THREADS_PER_BLOCK 64 // 64 threads per wave on Mi300, 32 threads per wave on Navi31
#define MAXITERS 100000 // maximum iteration number in a thread
#define FUNC1(K, I) (K + K * I / 37 - I) / 1091; // Shared by kernel and host
using namespace std;
constexpr int w1 = 34;
//#define SHOW_DETAILS // Show statics details
//#define VERIFY // Verify data
// #define SHOW_DETAILS // Show statics details
// #define VERIFY // Verify data
/**
* @addtogroup kernelLaunch clock
@@ -37,8 +37,8 @@ constexpr int w1 = 34;
* @ingroup PerformanceTest
* Contains unit tests for clock, clock64, wall_clock64 and hipExtLaunchKernelGGL APIs
*/
__global__ void
kernel1(uint64_t* out, size_t maxIter, uint64_t* clockCount, uint64_t* wallClockCount) {
__global__ void kernel1(uint64_t* out, size_t maxIter, uint64_t* clockCount,
uint64_t* wallClockCount) {
uint64_t wallClock = wall_clock64();
uint64_t clock = clock64();
size_t tid = hipBlockDim_x * hipBlockIdx_x + hipThreadIdx_x;
@@ -47,8 +47,8 @@ kernel1(uint64_t* out, size_t maxIter, uint64_t* clockCount, uint64_t* wallClock
k = FUNC1(k, i);
}
out[tid] = k;
clockCount[tid] = clock64() - clock; // GPU cycle count
wallClockCount[tid] = wall_clock64() - wallClock; // Wall clock cycle count
clockCount[tid] = clock64() - clock; // GPU cycle count
wallClockCount[tid] = wall_clock64() - wallClock; // Wall clock cycle count
}
#ifdef VERIFY
@@ -66,17 +66,17 @@ static void host1(uint64_t* out, size_t maxIter, size_t totalThreadsSize) {
/*
* Roughly query the variable gpu frequency
*/
static bool query_gpu_frequency(
void (*kernel)(uint64_t* out, size_t maxIter, uint64_t* clockCount, uint64_t* wallClockCount),
const int wall_clock_rate, const uint32_t blocksMax, const uint32_t blockSizeMax,
const int CUs) {
static bool query_gpu_frequency(void (*kernel)(uint64_t* out, size_t maxIter, uint64_t* clockCount,
uint64_t* wallClockCount),
const int wall_clock_rate, const uint32_t blocksMax,
const uint32_t blockSizeMax, const int CUs) {
hipStream_t stream;
hipEvent_t start_event, end_event;
const size_t totalThreadsSize = static_cast<size_t>(blockSizeMax) * blocksMax;
const size_t totalBytesSize = totalThreadsSize * sizeof(uint64_t);
const size_t maxIter = MAXITERS;
uint64_t* out; // Data to verify kernel rightness
uint64_t* out; // Data to verify kernel rightness
uint64_t* clockCount;
uint64_t* wallClockCount;
@@ -98,58 +98,52 @@ static bool query_gpu_frequency(
#endif
for (uint32_t blocks = 1; blocks <= blocksMax; blocks++) {
for (uint32_t blockSize = THREADS_PER_BLOCK; blockSize <= blockSizeMax; blockSize *= 2) {
hipExtLaunchKernelGGL(kernel, dim3(blocks), dim3(blockSize), 0, stream, start_event,
end_event, 0, out, maxIter, clockCount, wallClockCount);
HIP_CHECK(hipStreamSynchronize(stream));
float totalGpuTime = 0; // Total GPU time
HIP_CHECK(hipEventElapsedTime(&totalGpuTime, start_event, end_event));
hipExtLaunchKernelGGL(kernel, dim3(blocks), dim3(blockSize), 0, stream, start_event,
end_event, 0, out, maxIter, clockCount, wallClockCount);
HIP_CHECK(hipStreamSynchronize(stream));
float totalGpuTime = 0; // Total GPU time
HIP_CHECK(hipEventElapsedTime(&totalGpuTime, start_event, end_event));
const size_t curThreadsSize = static_cast<size_t>(blockSize) * blocks;
const size_t curBytesSize = curThreadsSize * sizeof(uint64_t);
HIP_CHECK(hipMemcpy(hostClockCount.data(), clockCount, curBytesSize,
hipMemcpyDeviceToHost));
HIP_CHECK(hipMemcpy(hostWallClockCount.data(), wallClockCount, curBytesSize,
hipMemcpyDeviceToHost));
const size_t curThreadsSize = static_cast<size_t>(blockSize) * blocks;
const size_t curBytesSize = curThreadsSize * sizeof(uint64_t);
HIP_CHECK(hipMemcpy(hostClockCount.data(), clockCount, curBytesSize, hipMemcpyDeviceToHost));
HIP_CHECK(hipMemcpy(hostWallClockCount.data(), wallClockCount, curBytesSize,
hipMemcpyDeviceToHost));
double clockMean = 0;
double wallClockMean = 0;
#ifdef SHOW_DETAILS
double clockDeviation = 0;
double wallClockDeviation = 0;
getStatics(hostClockCount.data(), curThreadsSize, clockMean, &clockDeviation);
getStatics(hostWallClockCount.data(), curThreadsSize, wallClockMean, &wallClockDeviation);
#else
getStatics(hostClockCount.data(), curThreadsSize, clockMean);
getStatics(hostWallClockCount.data(), curThreadsSize, wallClockMean);
#endif
double clockMean = 0;
double wallClockMean = 0;
#ifdef SHOW_DETAILS
double clockDeviation = 0;
double wallClockDeviation = 0;
getStatics(hostClockCount.data(), curThreadsSize, clockMean, &clockDeviation);
getStatics(hostWallClockCount.data(), curThreadsSize, wallClockMean, &wallClockDeviation);
#else
getStatics(hostClockCount.data(), curThreadsSize, clockMean);
getStatics(hostWallClockCount.data(), curThreadsSize, wallClockMean);
#endif
double aveGpuTime = wallClockMean / wall_clock_rate; // in ms, should be < totalGpuTime
double avgGpuFrequency = clockMean / aveGpuTime; // in KHz
double aveGpuTime = wallClockMean / wall_clock_rate; // in ms, should be < totalGpuTime
double avgGpuFrequency = clockMean / aveGpuTime; // in KHz
cout <<
setw(8) << blocks <<
setw(11) << blockSize <<
setw(22) << fixed << setprecision(3) << avgGpuFrequency / 1000. <<
setw(20) << fixed << setprecision(3) << aveGpuTime <<
setw(20) << fixed << setprecision(3) << totalGpuTime <<
setw(26) << fixed << setprecision(6) << curBytesSize / totalGpuTime / 1000. <<
setw(31) << fixed << setprecision(6) << curBytesSize / totalGpuTime / 1000. / CUs;
cout << setw(8) << blocks << setw(11) << blockSize << setw(22) << fixed << setprecision(3)
<< avgGpuFrequency / 1000. << setw(20) << fixed << setprecision(3) << aveGpuTime
<< setw(20) << fixed << setprecision(3) << totalGpuTime << setw(26) << fixed
<< setprecision(6) << curBytesSize / totalGpuTime / 1000. << setw(31) << fixed
<< setprecision(6) << curBytesSize / totalGpuTime / 1000. / CUs;
#ifdef SHOW_DETAILS
cout <<
setw(15) << fixed << setprecision(3) << wallClockMean <<
setw(15) << fixed << setprecision(3) << wallClockDeviation <<
setw(15) << fixed << setprecision(3) << clockMean <<
setw(15) << fixed << setprecision(3) << clockDeviation;
#endif
cout << endl;
#ifdef SHOW_DETAILS
cout << setw(15) << fixed << setprecision(3) << wallClockMean << setw(15) << fixed
<< setprecision(3) << wallClockDeviation << setw(15) << fixed << setprecision(3)
<< clockMean << setw(15) << fixed << setprecision(3) << clockDeviation;
#endif
cout << endl;
#ifdef VERIFY
HIP_CHECK(hipMemcpy(hostOut.data(), out, curBytesSize, hipMemcpyDeviceToHost));
host1(hostOutExpected.data(), maxIter, curThreadsSize);
verified = verify(hostOutExpected.data(), hostOut.data(), curThreadsSize);
HIP_CHECK(hipMemcpy(hostOut.data(), out, curBytesSize, hipMemcpyDeviceToHost));
host1(hostOutExpected.data(), maxIter, curThreadsSize);
verified = verify(hostOutExpected.data(), hostOut.data(), curThreadsSize);
#endif
if(!verified) {
if (!verified) {
cout << "Failed" << endl;
break;
}
@@ -178,7 +172,7 @@ static bool query_gpu_frequency(
*/
TEST_CASE("Performance_hipExtLaunchKernelGGL_QueryGPUFrequency") {
HIP_CHECK(hipSetDevice(0));
int clock_rate = 0; // in kHz
int clock_rate = 0; // in kHz
int wall_clock_rate = 0; // in kHz
int occupancyBlocks = 0;
int occupancyBlockSize = 0;
@@ -190,42 +184,31 @@ TEST_CASE("Performance_hipExtLaunchKernelGGL_QueryGPUFrequency") {
cout << left;
cout << setw(w1)
<< "--------------------------------------------------------------------------------"
<< endl;
<< "--------------------------------------------------------------------------------"
<< endl;
cout << setw(w1) << "device#" << 0 << endl;
cout << setw(w1) << "Name: " << props.name << endl;
cout << setw(w1) << "gcnArchName: " << props.gcnArchName << endl;
cout << setw(w1) << "multiProcessorCount: " << props.multiProcessorCount << endl;
cout << setw(w1) << "maxThreadsPerMultiProcessor: " << props.maxThreadsPerMultiProcessor
<< endl;
cout << setw(w1) << "maxThreadsPerMultiProcessor: " << props.maxThreadsPerMultiProcessor << endl;
cout << setw(w1) << "maxThreadsPerBlock: " << props.maxThreadsPerBlock << endl;
cout << setw(w1) << "occupancyBlocks: " << occupancyBlocks << endl;
cout << setw(w1) << "occupancyBlockSize: " << occupancyBlockSize << endl;
cout << setw(w1) << "waveSize: " << props.warpSize << endl;
cout << setw(w1) << "clockRate: " << clock_rate / 1000.0 << " Mhz" << endl;
cout << setw(w1) << "wallClockRate: " << wall_clock_rate / 1000.0 << " Mhz" << endl;
cout << setw(w1) << "memoryClockRate: " << props.memoryClockRate / 1000.0 << " Mhz"
<< endl;
cout << setw(w1) << "memoryClockRate: " << props.memoryClockRate / 1000.0 << " Mhz" << endl;
cout << setw(w1) << "totalGlobalMem: " << fixed << setprecision(2)
<< props.totalGlobalMem / 1000000000. << " GB" << endl;
cout << setw(w1) << "sharedMemPerBlock: " << props.sharedMemPerBlock / 1024.0 << " KiB"
<< endl;
<< props.totalGlobalMem / 1000000000. << " GB" << endl;
cout << setw(w1) << "sharedMemPerBlock: " << props.sharedMemPerBlock / 1024.0 << " KiB" << endl;
cout << setw(w1) << "l2CacheSize: " << props.l2CacheSize << endl;
cout <<
setw(8) << "Blocks " <<
setw(11) << "BlockSize" <<
setw(22) << "avgGpuFrequency(MHz)" <<
setw(20) << "aveGpuTime(ms)" <<
setw(20) << "totalGpuTime(ms)" <<
setw(26) << "processCapacity(Mbytes/s)" <<
setw(31) << "processCapacityPerCU(Mbytes/s)";
cout << setw(8) << "Blocks " << setw(11) << "BlockSize" << setw(22) << "avgGpuFrequency(MHz)"
<< setw(20) << "aveGpuTime(ms)" << setw(20) << "totalGpuTime(ms)" << setw(26)
<< "processCapacity(Mbytes/s)" << setw(31) << "processCapacityPerCU(Mbytes/s)";
#ifdef SHOW_DETAILS
cout <<
setw(15) << "mean wallClock" <<
setw(15) << "deviation" <<
setw(15) << "mean clock" <<
setw(15) << "deviation";
cout << setw(15) << "mean wallClock" << setw(15) << "deviation" << setw(15) << "mean clock"
<< setw(15) << "deviation";
#endif
cout << endl;
@@ -29,14 +29,12 @@ THE SOFTWARE.
class MemcpyBenchmark : public Benchmark<MemcpyBenchmark> {
public:
void operator()(void* dst, const void* src, size_t size, hipMemcpyKind kind) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpy(dst, src, size, kind));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpy(dst, src, size, kind)); }
}
};
static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allocation_type,
size_t size, hipMemcpyKind kind, bool enable_peer_access=false) {
size_t size, hipMemcpyKind kind, bool enable_peer_access = false) {
MemcpyBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(GetAllocationSectionName(src_allocation_type));
@@ -49,7 +47,9 @@ static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allo
} else {
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard<int> src_allocation(src_allocation_type, size);
HIP_CHECK(hipSetDevice(dst_device));
@@ -162,7 +162,8 @@ TEST_CASE("Performance_hipMemcpy_DeviceToDevice_EnablePeerAccess") {
const auto allocation_size = GENERATE(4_KB, 4_MB, 16_MB);
const auto src_allocation_type = LinearAllocs::hipMalloc;
const auto dst_allocation_type = LinearAllocs::hipMalloc;
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice, true);
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice,
true);
}
/**
@@ -35,7 +35,8 @@ class Memcpy2DBenchmark : public Benchmark<Memcpy2DBenchmark> {
}
};
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool enable_peer_access=false) {
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
bool enable_peer_access = false) {
Memcpy2DBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
@@ -43,17 +44,15 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool e
LinearAllocGuard2D<int> device_allocation(width, height);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
device_allocation.width() * height);
benchmark.Run(host_allocation.ptr(), device_allocation.width(),
device_allocation.ptr(), device_allocation.pitch(),
device_allocation.width(), device_allocation.height(),
benchmark.Run(host_allocation.ptr(), device_allocation.width(), device_allocation.ptr(),
device_allocation.pitch(), device_allocation.width(), device_allocation.height(),
hipMemcpyDeviceToHost);
} else if (kind == hipMemcpyHostToDevice) {
LinearAllocGuard2D<int> device_allocation(width, height);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
device_allocation.width() * height);
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
host_allocation.ptr(), device_allocation.width(),
device_allocation.width(), device_allocation.height(),
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), host_allocation.ptr(),
device_allocation.width(), device_allocation.width(), device_allocation.height(),
hipMemcpyHostToDevice);
} else if (kind == hipMemcpyHostToHost) {
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
@@ -64,15 +63,16 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool e
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard2D<int> src_allocation(width, height);
HIP_CHECK(hipSetDevice(dst_device));
LinearAllocGuard2D<int> dst_allocation(width, height);
HIP_CHECK(hipSetDevice(src_device));
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(),
src_allocation.ptr(), src_allocation.pitch(),
dst_allocation.width(), dst_allocation.height(),
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(), src_allocation.ptr(),
src_allocation.pitch(), dst_allocation.width(), dst_allocation.height(),
hipMemcpyDeviceToDevice);
}
}
@@ -36,7 +36,8 @@ class Memcpy2DAsyncBenchmark : public Benchmark<Memcpy2DAsyncBenchmark> {
}
};
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool enable_peer_access=false) {
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
bool enable_peer_access = false) {
Memcpy2DAsyncBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
@@ -47,17 +48,15 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool e
LinearAllocGuard2D<int> device_allocation(width, height);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
device_allocation.width() * height);
benchmark.Run(host_allocation.ptr(), device_allocation.width(),
device_allocation.ptr(), device_allocation.pitch(),
device_allocation.width(), device_allocation.height(),
benchmark.Run(host_allocation.ptr(), device_allocation.width(), device_allocation.ptr(),
device_allocation.pitch(), device_allocation.width(), device_allocation.height(),
hipMemcpyDeviceToHost, stream);
} else if (kind == hipMemcpyHostToDevice) {
LinearAllocGuard2D<int> device_allocation(width, height);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
device_allocation.width() * height);
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
host_allocation.ptr(), device_allocation.width(),
device_allocation.width(), device_allocation.height(),
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), host_allocation.ptr(),
device_allocation.width(), device_allocation.width(), device_allocation.height(),
hipMemcpyHostToDevice, stream);
} else if (kind == hipMemcpyHostToHost) {
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
@@ -68,16 +67,17 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool e
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard2D<int> src_allocation(width, height);
HIP_CHECK(hipSetDevice(dst_device));
LinearAllocGuard2D<int> dst_allocation(width, height);
HIP_CHECK(hipSetDevice(src_device));
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(),
src_allocation.ptr(), src_allocation.pitch(),
dst_allocation.width(), dst_allocation.height(),
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(), src_allocation.ptr(),
src_allocation.pitch(), dst_allocation.width(), dst_allocation.height(),
hipMemcpyDeviceToDevice, stream);
}
}
@@ -27,7 +27,8 @@ THE SOFTWARE.
class Memcpy2DFromArrayBenchmark : public Benchmark<Memcpy2DFromArrayBenchmark> {
public:
void operator()(void* dst, size_t dst_pitch, hipArray_const_t src, size_t width, size_t height, hipMemcpyKind kind) {
void operator()(void* dst, size_t dst_pitch, hipArray_const_t src, size_t width, size_t height,
hipMemcpyKind kind) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpy2DFromArray(dst, dst_pitch, src, 0, 0, width, height, kind));
}
@@ -35,7 +36,7 @@ class Memcpy2DFromArrayBenchmark : public Benchmark<Memcpy2DFromArrayBenchmark>
};
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
bool enable_peer_access=false) {
bool enable_peer_access = false) {
Memcpy2DFromArrayBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
@@ -49,15 +50,16 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard2D<int> device_allocation(width, height);
HIP_CHECK(hipSetDevice(dst_device));
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
HIP_CHECK(hipSetDevice(src_device));
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
array_allocation.ptr(), device_allocation.width(),
device_allocation.height(), hipMemcpyDeviceToDevice);
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), array_allocation.ptr(),
device_allocation.width(), device_allocation.height(), hipMemcpyDeviceToDevice);
}
}
@@ -37,7 +37,7 @@ class Memcpy2DFromArrayAsyncBenchmark : public Benchmark<Memcpy2DFromArrayAsyncB
};
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
bool enable_peer_access=false) {
bool enable_peer_access = false) {
Memcpy2DFromArrayAsyncBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
@@ -48,22 +48,23 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
size_t allocation_size = width * height * sizeof(int);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, allocation_size);
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
benchmark.Run(host_allocation.ptr(), width * sizeof(int),
array_allocation.ptr(), width * sizeof(int),
height, hipMemcpyDeviceToHost, stream);
benchmark.Run(host_allocation.ptr(), width * sizeof(int), array_allocation.ptr(),
width * sizeof(int), height, hipMemcpyDeviceToHost, stream);
} else {
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard2D<int> device_allocation(width, height);
HIP_CHECK(hipSetDevice(dst_device));
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
HIP_CHECK(hipSetDevice(src_device));
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
array_allocation.ptr(), device_allocation.width(),
device_allocation.height(), hipMemcpyDeviceToDevice, stream);
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), array_allocation.ptr(),
device_allocation.width(), device_allocation.height(), hipMemcpyDeviceToDevice,
stream);
}
}
@@ -27,8 +27,8 @@ THE SOFTWARE.
class Memcpy2DToArrayBenchmark : public Benchmark<Memcpy2DToArrayBenchmark> {
public:
void operator()(hipArray_t dst, const void* src, size_t src_pitch, size_t width,
size_t height, hipMemcpyKind kind) {
void operator()(hipArray_t dst, const void* src, size_t src_pitch, size_t width, size_t height,
hipMemcpyKind kind) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpy2DToArray(dst, 0, 0, src, src_pitch, width, height, kind));
}
@@ -36,7 +36,7 @@ class Memcpy2DToArrayBenchmark : public Benchmark<Memcpy2DToArrayBenchmark> {
};
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
bool enable_peer_access=false) {
bool enable_peer_access = false) {
Memcpy2DToArrayBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
@@ -50,7 +50,9 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard2D<int> device_allocation(width, height);
HIP_CHECK(hipSetDevice(dst_device));
@@ -27,8 +27,8 @@ THE SOFTWARE.
class Memcpy2DToArrayAsyncBenchmark : public Benchmark<Memcpy2DToArrayAsyncBenchmark> {
public:
void operator()(hipArray_t dst, const void* src, size_t src_pitch, size_t width,
size_t height, hipMemcpyKind kind, const hipStream_t& stream) {
void operator()(hipArray_t dst, const void* src, size_t src_pitch, size_t width, size_t height,
hipMemcpyKind kind, const hipStream_t& stream) {
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
HIP_CHECK(hipMemcpy2DToArrayAsync(dst, 0, 0, src, src_pitch, width, height, kind, stream));
}
@@ -37,7 +37,7 @@ class Memcpy2DToArrayAsyncBenchmark : public Benchmark<Memcpy2DToArrayAsyncBench
};
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
bool enable_peer_access=false) {
bool enable_peer_access = false) {
Memcpy2DToArrayAsyncBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
@@ -48,22 +48,23 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
size_t allocation_size = width * height * sizeof(int);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, allocation_size);
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
benchmark.Run(array_allocation.ptr(), host_allocation.ptr(),
width * sizeof(int), width * sizeof(int), height,
hipMemcpyHostToDevice, stream);
benchmark.Run(array_allocation.ptr(), host_allocation.ptr(), width * sizeof(int),
width * sizeof(int), height, hipMemcpyHostToDevice, stream);
} else {
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard2D<int> device_allocation(width, height);
HIP_CHECK(hipSetDevice(dst_device));
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
HIP_CHECK(hipSetDevice(src_device));
benchmark.Run(array_allocation.ptr(), device_allocation.ptr(), device_allocation.pitch(),
device_allocation.width(), device_allocation.height(),
hipMemcpyDeviceToDevice, stream);
device_allocation.width(), device_allocation.height(), hipMemcpyDeviceToDevice,
stream);
}
}
@@ -29,49 +29,53 @@ class Memcpy3DBenchmark : public Benchmark<Memcpy3DBenchmark> {
public:
void operator()(const hipPitchedPtr& dst_ptr, const hipPitchedPtr& src_ptr,
const hipExtent extent, hipMemcpyKind kind) {
hipMemcpy3DParms params = CreateMemcpy3DParam(dst_ptr, make_hipPos(0, 0, 0),
src_ptr, make_hipPos(0, 0, 0),
extent, kind);
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpy3D(&params));
}
hipMemcpy3DParms params = CreateMemcpy3DParam(dst_ptr, make_hipPos(0, 0, 0), src_ptr,
make_hipPos(0, 0, 0), extent, kind);
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpy3D(&params)); }
}
};
static void RunBenchmark(const hipExtent extent, hipMemcpyKind kind, bool enable_peer_access=false) {
static void RunBenchmark(const hipExtent extent, hipMemcpyKind kind,
bool enable_peer_access = false) {
Memcpy3DBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(extent.width) + ", " + std::to_string(extent.height)
+ ", " + std::to_string(extent.depth) + ")");
benchmark.AddSectionName("(" + std::to_string(extent.width) + ", " +
std::to_string(extent.height) + ", " + std::to_string(extent.depth) +
")");
if (kind == hipMemcpyDeviceToHost) {
LinearAllocGuard3D<int> device_allocation(extent);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() *
device_allocation.height() * device_allocation.depth());
benchmark.Run(make_hipPitchedPtr(host_allocation.ptr(), device_allocation.width(),
LinearAllocGuard<int> host_allocation(
LinearAllocs::hipHostMalloc,
device_allocation.width() * device_allocation.height() * device_allocation.depth());
benchmark.Run(make_hipPitchedPtr(host_allocation.ptr(), device_allocation.width(),
device_allocation.width(), device_allocation.height()),
device_allocation.pitched_ptr(), device_allocation.extent(), kind);
} else if (kind == hipMemcpyHostToDevice) {
LinearAllocGuard3D<int> device_allocation(extent);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.pitch() *
device_allocation.height() * device_allocation.depth());
LinearAllocGuard<int> host_allocation(
LinearAllocs::hipHostMalloc,
device_allocation.pitch() * device_allocation.height() * device_allocation.depth());
benchmark.Run(device_allocation.pitched_ptr(),
make_hipPitchedPtr(host_allocation.ptr(), device_allocation.pitch(),
device_allocation.width(), device_allocation.height()),
device_allocation.extent(), kind);
} else if (kind == hipMemcpyHostToHost) {
LinearAllocGuard3D<int> device_allocation(extent);
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, extent.width *
extent.height * extent.depth);
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc, extent.width *
extent.height * extent.depth);
benchmark.Run(make_hipPitchedPtr(dst_allocation.ptr(), extent.width, extent.width, extent.height),
make_hipPitchedPtr(src_allocation.ptr(), extent.width, extent.width, extent.height),
extent, kind);
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc,
extent.width * extent.height * extent.depth);
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc,
extent.width * extent.height * extent.depth);
benchmark.Run(
make_hipPitchedPtr(dst_allocation.ptr(), extent.width, extent.width, extent.height),
make_hipPitchedPtr(src_allocation.ptr(), extent.width, extent.width, extent.height), extent,
kind);
} else {
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard3D<int> src_allocation(extent);
HIP_CHECK(hipSetDevice(dst_device));
@@ -29,55 +29,57 @@ class Memcpy3DAsyncBenchmark : public Benchmark<Memcpy3DAsyncBenchmark> {
public:
void operator()(const hipPitchedPtr& dst_ptr, const hipPitchedPtr& src_ptr,
const hipExtent extent, hipMemcpyKind kind, const hipStream_t& stream) {
hipMemcpy3DParms params = CreateMemcpy3DParam(dst_ptr, make_hipPos(0, 0, 0),
src_ptr, make_hipPos(0, 0, 0),
extent, kind);
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
HIP_CHECK(hipMemcpy3DAsync(&params, stream));
}
hipMemcpy3DParms params = CreateMemcpy3DParam(dst_ptr, make_hipPos(0, 0, 0), src_ptr,
make_hipPos(0, 0, 0), extent, kind);
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) { HIP_CHECK(hipMemcpy3DAsync(&params, stream)); }
HIP_CHECK(hipStreamSynchronize(stream));
}
};
static void RunBenchmark(const hipExtent extent, hipMemcpyKind kind, bool enable_peer_access=false) {
static void RunBenchmark(const hipExtent extent, hipMemcpyKind kind,
bool enable_peer_access = false) {
Memcpy3DAsyncBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(extent.width) + ", " + std::to_string(extent.height)
+ ", " + std::to_string(extent.depth) + ")");
benchmark.AddSectionName("(" + std::to_string(extent.width) + ", " +
std::to_string(extent.height) + ", " + std::to_string(extent.depth) +
")");
const StreamGuard stream_guard(Streams::created);
const hipStream_t stream = stream_guard.stream();
if (kind == hipMemcpyDeviceToHost) {
LinearAllocGuard3D<int> device_allocation(extent);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() *
device_allocation.height() * device_allocation.depth());
LinearAllocGuard<int> host_allocation(
LinearAllocs::hipHostMalloc,
device_allocation.width() * device_allocation.height() * device_allocation.depth());
benchmark.Run(make_hipPitchedPtr(host_allocation.ptr(), device_allocation.width(),
device_allocation.width(), device_allocation.height()),
device_allocation.pitched_ptr(), device_allocation.extent(), kind, stream);
} else if (kind == hipMemcpyHostToDevice) {
LinearAllocGuard3D<int> device_allocation(extent);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.pitch() *
device_allocation.height() * device_allocation.depth());
LinearAllocGuard<int> host_allocation(
LinearAllocs::hipHostMalloc,
device_allocation.pitch() * device_allocation.height() * device_allocation.depth());
benchmark.Run(device_allocation.pitched_ptr(),
make_hipPitchedPtr(host_allocation.ptr(),
device_allocation.pitch(),
device_allocation.width(),
device_allocation.height()),
make_hipPitchedPtr(host_allocation.ptr(), device_allocation.pitch(),
device_allocation.width(), device_allocation.height()),
device_allocation.extent(), kind, stream);
} else if (kind == hipMemcpyHostToHost) {
LinearAllocGuard3D<int> device_allocation(extent);
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, extent.width *
extent.height * extent.depth);
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc, extent.width *
extent.height * extent.depth);
benchmark.Run(make_hipPitchedPtr(dst_allocation.ptr(), extent.width, extent.width, extent.height),
make_hipPitchedPtr(src_allocation.ptr(), extent.width, extent.width, extent.height),
extent, kind, stream);
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc,
extent.width * extent.height * extent.depth);
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc,
extent.width * extent.height * extent.depth);
benchmark.Run(
make_hipPitchedPtr(dst_allocation.ptr(), extent.width, extent.width, extent.height),
make_hipPitchedPtr(src_allocation.ptr(), extent.width, extent.width, extent.height), extent,
kind, stream);
} else {
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard3D<int> src_allocation(extent);
HIP_CHECK(hipSetDevice(dst_device));
@@ -27,7 +27,8 @@ THE SOFTWARE.
class MemcpyAsyncBenchmark : public Benchmark<MemcpyAsyncBenchmark> {
public:
void operator()(void* dst, const void* src, size_t size, hipMemcpyKind kind, const hipStream_t& stream) {
void operator()(void* dst, const void* src, size_t size, hipMemcpyKind kind,
const hipStream_t& stream) {
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
HIP_CHECK(hipMemcpyAsync(dst, src, size, kind, stream));
}
@@ -36,7 +37,7 @@ class MemcpyAsyncBenchmark : public Benchmark<MemcpyAsyncBenchmark> {
};
static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allocation_type,
size_t size, hipMemcpyKind kind, bool enable_peer_access=false) {
size_t size, hipMemcpyKind kind, bool enable_peer_access = false) {
MemcpyAsyncBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(GetAllocationSectionName(src_allocation_type));
@@ -51,7 +52,9 @@ static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allo
} else {
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard<int> src_allocation(src_allocation_type, size);
HIP_CHECK(hipSetDevice(dst_device));
@@ -189,7 +192,8 @@ TEST_CASE("Performance_hipMemcpyAsync_DeviceToDevice_EnablePeerAccess") {
const auto allocation_size = GENERATE(4_KB, 4_MB, 16_MB);
const auto src_allocation_type = LinearAllocs::hipMalloc;
const auto dst_allocation_type = LinearAllocs::hipMalloc;
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice, true);
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice,
true);
}
/**
@@ -28,9 +28,7 @@ THE SOFTWARE.
class MemcpyAtoHBenchmark : public Benchmark<MemcpyAtoHBenchmark> {
public:
void operator()(void* dst, hipArray_t src_array, size_t allocation_size) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpyAtoH(dst, src_array, 0, allocation_size));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyAtoH(dst, src_array, 0, allocation_size)); }
}
};
@@ -28,19 +28,19 @@ THE SOFTWARE.
class MemcpyDtoDBenchmark : public Benchmark<MemcpyDtoDBenchmark> {
public:
void operator()(hipDeviceptr_t& dst, const hipDeviceptr_t& src, size_t size) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpyDtoD(dst, src, size));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyDtoD(dst, src, size)); }
}
};
static void RunBenchmark(size_t size, bool enable_peer_access=false) {
static void RunBenchmark(size_t size, bool enable_peer_access = false) {
MemcpyDtoDBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard<int> src_allocation(LinearAllocs::hipMalloc, size);
HIP_CHECK(hipSetDevice(dst_device));
@@ -27,7 +27,8 @@ THE SOFTWARE.
class MemcpyDtoDAsyncBenchmark : public Benchmark<MemcpyDtoDAsyncBenchmark> {
public:
void operator()(hipDeviceptr_t& dst, const hipDeviceptr_t& src, size_t size, const hipStream_t& stream) {
void operator()(hipDeviceptr_t& dst, const hipDeviceptr_t& src, size_t size,
const hipStream_t& stream) {
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
HIP_CHECK(hipMemcpyDtoDAsync(dst, src, size, stream));
}
@@ -35,7 +36,7 @@ class MemcpyDtoDAsyncBenchmark : public Benchmark<MemcpyDtoDAsyncBenchmark> {
}
};
static void RunBenchmark(size_t size, bool enable_peer_access=false) {
static void RunBenchmark(size_t size, bool enable_peer_access = false) {
MemcpyDtoDAsyncBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
@@ -43,15 +44,16 @@ static void RunBenchmark(size_t size, bool enable_peer_access=false) {
const hipStream_t stream = stream_guard.stream();
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard<int> src_allocation(LinearAllocs::hipMalloc, size);
HIP_CHECK(hipSetDevice(dst_device));
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipMalloc, size);
HIP_CHECK(hipSetDevice(src_device));
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(dst_allocation.ptr()),
reinterpret_cast<hipDeviceptr_t>(src_allocation.ptr()),
size, stream);
reinterpret_cast<hipDeviceptr_t>(src_allocation.ptr()), size, stream);
}
/**
@@ -28,21 +28,19 @@ THE SOFTWARE.
class MemcpyDtoHBenchmark : public Benchmark<MemcpyDtoHBenchmark> {
public:
void operator()(void* dst, const hipDeviceptr_t& src, size_t size) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpyDtoH(dst, src, size));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyDtoH(dst, src, size)); }
}
};
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type, size_t size) {
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
size_t size) {
MemcpyDtoHBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(GetAllocationSectionName(host_allocation_type));
LinearAllocGuard<int> device_allocation(device_allocation_type, size);
LinearAllocGuard<int> host_allocation(host_allocation_type, size);
benchmark.Run(host_allocation.ptr(),
reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
benchmark.Run(host_allocation.ptr(), reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
size);
}
@@ -35,7 +35,8 @@ class MemcpyDtoHAsyncBenchmark : public Benchmark<MemcpyDtoHAsyncBenchmark> {
}
};
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type, size_t size) {
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
size_t size) {
MemcpyDtoHAsyncBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(GetAllocationSectionName(host_allocation_type));
@@ -44,8 +45,7 @@ static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_
const hipStream_t stream = stream_guard.stream();
LinearAllocGuard<int> device_allocation(device_allocation_type, size);
LinearAllocGuard<int> host_allocation(host_allocation_type, size);
benchmark.Run(host_allocation.ptr(),
reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
benchmark.Run(host_allocation.ptr(), reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
size, stream);
}
@@ -38,7 +38,7 @@ class MemcpyFromSymbolBenchmark : public Benchmark<MemcpyFromSymbolBenchmark> {
}
};
static void RunBenchmark(const void* source, void* result, size_t size=1, size_t offset=0) {
static void RunBenchmark(const void* source, void* result, size_t size = 1, size_t offset = 0) {
MemcpyFromSymbolBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(std::to_string(offset));
@@ -112,7 +112,8 @@ TEST_CASE("Performance_hipMemcpyFromSymbol_WithOffset") {
std::fill_n(result.data(), size, 0);
size_t offset = GENERATE_REF(0, size / 2);
RunBenchmark(array.data() + offset, result.data() + offset, sizeof(int) * (size - offset), offset * sizeof(int));
RunBenchmark(array.data() + offset, result.data() + offset, sizeof(int) * (size - offset),
offset * sizeof(int));
}
/**
@@ -30,7 +30,8 @@ __device__ int devSymbol[1_MB];
class MemcpyFromSymbolAsyncBenchmark : public Benchmark<MemcpyFromSymbolAsyncBenchmark> {
public:
void operator()(const void* source, void* result, size_t size, size_t offset, const hipStream_t& stream) {
void operator()(const void* source, void* result, size_t size, size_t offset,
const hipStream_t& stream) {
HIP_CHECK(hipMemcpyToSymbolAsync(HIP_SYMBOL(devSymbol), source, size, offset,
hipMemcpyHostToDevice, stream));
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
@@ -41,7 +42,7 @@ class MemcpyFromSymbolAsyncBenchmark : public Benchmark<MemcpyFromSymbolAsyncBen
}
};
static void RunBenchmark(const void* source, void* result, size_t size=1, size_t offset=0) {
static void RunBenchmark(const void* source, void* result, size_t size = 1, size_t offset = 0) {
MemcpyFromSymbolAsyncBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(std::to_string(offset));
@@ -118,7 +119,8 @@ TEST_CASE("Performance_hipMemcpyFromSymbolAsync_WithOffset") {
std::fill_n(result.data(), size, 0);
size_t offset = GENERATE_REF(0, size / 2);
RunBenchmark(array.data() + offset, result.data() + offset, sizeof(int) * (size - offset), offset * sizeof(int));
RunBenchmark(array.data() + offset, result.data() + offset, sizeof(int) * (size - offset),
offset * sizeof(int));
}
/**
@@ -28,9 +28,7 @@ THE SOFTWARE.
class MemcpyHtoABenchmark : public Benchmark<MemcpyHtoABenchmark> {
public:
void operator()(hipArray_t dst_array, const void* src, size_t allocation_size) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpyHtoA(dst_array, 0, src, allocation_size));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyHtoA(dst_array, 0, src, allocation_size)); }
}
};
@@ -28,20 +28,20 @@ THE SOFTWARE.
class MemcpyHtoDBenchmark : public Benchmark<MemcpyHtoDBenchmark> {
public:
void operator()(hipDeviceptr_t& dst, void* src, size_t size) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpyHtoD(dst, src, size));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyHtoD(dst, src, size)); }
}
};
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type, size_t size) {
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
size_t size) {
MemcpyHtoDBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(GetAllocationSectionName(host_allocation_type));
LinearAllocGuard<int> device_allocation(device_allocation_type, size);
LinearAllocGuard<int> host_allocation(host_allocation_type, size);
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()), host_allocation.ptr(), size);
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()), host_allocation.ptr(),
size);
}
/**
@@ -35,7 +35,8 @@ class MemcpyHtoDAsyncBenchmark : public Benchmark<MemcpyHtoDAsyncBenchmark> {
}
};
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type, size_t size) {
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
size_t size) {
MemcpyHtoDAsyncBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(GetAllocationSectionName(host_allocation_type));
@@ -44,8 +45,8 @@ static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_
const hipStream_t stream = stream_guard.stream();
LinearAllocGuard<int> device_allocation(device_allocation_type, size);
LinearAllocGuard<int> host_allocation(host_allocation_type, size);
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
host_allocation.ptr(), size, stream);
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()), host_allocation.ptr(),
size, stream);
}
/**
@@ -27,54 +27,52 @@ THE SOFTWARE.
class MemcpyParam2DBenchmark : public Benchmark<MemcpyParam2DBenchmark> {
public:
void operator()(void* dst, size_t dst_pitch, void* src, size_t src_pitch,
size_t width, size_t height, hipMemcpyKind kind) {
hip_Memcpy2D params = CreateMemcpy2DParam(dst, dst_pitch, src, src_pitch,
width, height, kind);
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpyParam2D(&params));
}
void operator()(void* dst, size_t dst_pitch, void* src, size_t src_pitch, size_t width,
size_t height, hipMemcpyKind kind) {
hip_Memcpy2D params = CreateMemcpy2DParam(dst, dst_pitch, src, src_pitch, width, height, kind);
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyParam2D(&params)); }
}
};
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
bool enable_peer_access=false) {
bool enable_peer_access = false) {
MemcpyParam2DBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
if (kind == hipMemcpyDeviceToHost) {
LinearAllocGuard2D<int> device_allocation(width, height);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() * height);
benchmark.Run(host_allocation.ptr(), device_allocation.width(),
device_allocation.ptr(), device_allocation.pitch(),
device_allocation.width(), device_allocation.height(), kind);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
device_allocation.width() * height);
benchmark.Run(host_allocation.ptr(), device_allocation.width(), device_allocation.ptr(),
device_allocation.pitch(), device_allocation.width(), device_allocation.height(),
kind);
} else if (kind == hipMemcpyHostToDevice) {
LinearAllocGuard2D<int> device_allocation(width, height);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() * height);
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
host_allocation.ptr(), device_allocation.width(),
device_allocation.width(), device_allocation.height(), kind);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
device_allocation.width() * height);
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), host_allocation.ptr(),
device_allocation.width(), device_allocation.width(), device_allocation.height(),
kind);
} else if (kind == hipMemcpyHostToHost) {
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
benchmark.Run(dst_allocation.ptr(), width * sizeof(int),
src_allocation.ptr(), width * sizeof(int),
width * sizeof(int), height, kind);
benchmark.Run(dst_allocation.ptr(), width * sizeof(int), src_allocation.ptr(),
width * sizeof(int), width * sizeof(int), height, kind);
} else {
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard2D<int> src_allocation(width, height);
HIP_CHECK(hipSetDevice(dst_device));
LinearAllocGuard2D<int> dst_allocation(width, height);
HIP_CHECK(hipSetDevice(src_device));
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(),
src_allocation.ptr(), src_allocation.pitch(),
dst_allocation.width(), dst_allocation.height(),
kind);
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(), src_allocation.ptr(),
src_allocation.pitch(), dst_allocation.width(), dst_allocation.height(), kind);
}
}
@@ -27,19 +27,16 @@ THE SOFTWARE.
class MemcpyParam2DBenchmark : public Benchmark<MemcpyParam2DBenchmark> {
public:
void operator()(void* dst, size_t dst_pitch, void* src, size_t src_pitch,
size_t width, size_t height, hipMemcpyKind kind, const hipStream_t& stream) {
hip_Memcpy2D params = CreateMemcpy2DParam(dst, dst_pitch, src, src_pitch,
width, height, kind);
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpyParam2DAsync(&params, stream));
}
void operator()(void* dst, size_t dst_pitch, void* src, size_t src_pitch, size_t width,
size_t height, hipMemcpyKind kind, const hipStream_t& stream) {
hip_Memcpy2D params = CreateMemcpy2DParam(dst, dst_pitch, src, src_pitch, width, height, kind);
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyParam2DAsync(&params, stream)); }
HIP_CHECK(hipStreamSynchronize(stream));
}
};
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
bool enable_peer_access=false) {
bool enable_peer_access = false) {
MemcpyParam2DBenchmark benchmark;
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
@@ -48,38 +45,38 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
if (kind == hipMemcpyDeviceToHost) {
LinearAllocGuard2D<int> device_allocation(width, height);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() * height);
benchmark.Run(host_allocation.ptr(), device_allocation.width(),
device_allocation.ptr(), device_allocation.pitch(),
device_allocation.width(), device_allocation.height(),
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
device_allocation.width() * height);
benchmark.Run(host_allocation.ptr(), device_allocation.width(), device_allocation.ptr(),
device_allocation.pitch(), device_allocation.width(), device_allocation.height(),
kind, stream);
} else if (kind == hipMemcpyHostToDevice) {
LinearAllocGuard2D<int> device_allocation(width, height);
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() * height);
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
host_allocation.ptr(), device_allocation.width(),
device_allocation.width(), device_allocation.height(),
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
device_allocation.width() * height);
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), host_allocation.ptr(),
device_allocation.width(), device_allocation.width(), device_allocation.height(),
kind, stream);
} else if (kind == hipMemcpyHostToHost) {
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
benchmark.Run(dst_allocation.ptr(), width * sizeof(int),
src_allocation.ptr(), width * sizeof(int),
width * sizeof(int), height, kind, stream);
benchmark.Run(dst_allocation.ptr(), width * sizeof(int), src_allocation.ptr(),
width * sizeof(int), width * sizeof(int), height, kind, stream);
} else {
// hipMemcpyDeviceToDevice
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard2D<int> src_allocation(width, height);
HIP_CHECK(hipSetDevice(dst_device));
LinearAllocGuard2D<int> dst_allocation(width, height);
HIP_CHECK(hipSetDevice(src_device));
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(),
src_allocation.ptr(), src_allocation.pitch(),
dst_allocation.width(), dst_allocation.height(),
kind, stream);
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(), src_allocation.ptr(),
src_allocation.pitch(), dst_allocation.width(), dst_allocation.height(), kind,
stream);
}
}
@@ -36,7 +36,7 @@ class MemcpyToSymbolBenchmark : public Benchmark<MemcpyToSymbolBenchmark> {
}
};
static void RunBenchmark(const void* source, size_t size=1, size_t offset=0) {
static void RunBenchmark(const void* source, size_t size = 1, size_t offset = 0) {
MemcpyToSymbolBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(std::to_string(offset));
@@ -40,7 +40,7 @@ class MemcpyToSymbolAsyncBenchmark : public Benchmark<MemcpyToSymbolAsyncBenchma
}
};
static void RunBenchmark(const void* source, size_t size=1, size_t offset=0) {
static void RunBenchmark(const void* source, size_t size = 1, size_t offset = 0) {
MemcpyToSymbolAsyncBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(std::to_string(offset));
@@ -28,7 +28,8 @@ __global__ void Sum(void* ptr, size_t size) {
atomicAdd(&((unsigned long long*)ptr)[0], ((unsigned long long*)ptr)[index]);
}
}
class MemcpyHtoDKernelDtoHv1AsyncBenchmark : public Benchmark<MemcpyHtoDKernelDtoHv1AsyncBenchmark> {
class MemcpyHtoDKernelDtoHv1AsyncBenchmark
: public Benchmark<MemcpyHtoDKernelDtoHv1AsyncBenchmark> {
public:
void operator()(void* host_mem, void* device_mem, size_t size, const hipStream_t& stream) {
size_t count = size / sizeof(size_t);
@@ -36,19 +37,20 @@ class MemcpyHtoDKernelDtoHv1AsyncBenchmark : public Benchmark<MemcpyHtoDKernelDt
((size_t*)host_mem)[i] = i;
}
TIMED_SECTION_STREAM(kTimerTypeCpu, stream) {
HIP_CHECK(hipMemcpyHtoDAsync(reinterpret_cast<hipDeviceptr_t>(device_mem), host_mem,
size, stream));
HIP_CHECK(
hipMemcpyHtoDAsync(reinterpret_cast<hipDeviceptr_t>(device_mem), host_mem, size, stream));
int threads_num = 32;
Sum<<<count / threads_num + 1, threads_num, 0, stream>>>(device_mem, count);
HIP_CHECK(hipMemcpyDtoHAsync(host_mem, reinterpret_cast<hipDeviceptr_t>(device_mem),
size, stream));
HIP_CHECK(
hipMemcpyDtoHAsync(host_mem, reinterpret_cast<hipDeviceptr_t>(device_mem), size, stream));
HIP_CHECK(hipStreamSynchronize(stream));
}
size_t sum = ((size_t*)host_mem)[0];
REQUIRE(sum == count * (count - 1) / 2);
}
};
class MemcpyHtoDKernelDtoHv2AsyncBenchmark : public Benchmark<MemcpyHtoDKernelDtoHv2AsyncBenchmark> {
class MemcpyHtoDKernelDtoHv2AsyncBenchmark
: public Benchmark<MemcpyHtoDKernelDtoHv2AsyncBenchmark> {
public:
void operator()(void* host_mem, void* device_mem, size_t size, const hipStream_t& stream) {
size_t count = size / sizeof(size_t);
@@ -56,19 +58,17 @@ class MemcpyHtoDKernelDtoHv2AsyncBenchmark : public Benchmark<MemcpyHtoDKernelDt
((size_t*)host_mem)[i] = i;
}
TIMED_SECTION_STREAM(kTimerTypeCpu, stream) {
HIP_CHECK(hipMemcpyAsync(device_mem, host_mem, size, hipMemcpyHostToDevice,
stream));
HIP_CHECK(hipMemcpyAsync(device_mem, host_mem, size, hipMemcpyHostToDevice, stream));
int threads_num = 32;
Sum<<<count / threads_num + 1, threads_num, 0, stream>>>(device_mem, count);
HIP_CHECK(hipMemcpyWithStream(host_mem, device_mem, size,
hipMemcpyDeviceToHost, stream));
HIP_CHECK(hipMemcpyWithStream(host_mem, device_mem, size, hipMemcpyDeviceToHost, stream));
HIP_CHECK(hipStreamSynchronize(stream));
}
size_t sum = ((size_t*)host_mem)[0];
REQUIRE(sum == count * (count - 1) / 2);
}
};
template<typename BenchmarkType>
template <typename BenchmarkType>
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
size_t size) {
BenchmarkType benchmark;
@@ -28,14 +28,12 @@ THE SOFTWARE.
class MemcpyWithStreamBenchmark : public Benchmark<MemcpyWithStreamBenchmark> {
public:
void operator()(void* dst, const void* src, size_t size, hipMemcpyKind kind, hipStream_t stream) {
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemcpyWithStream(dst, src, size, kind, stream));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyWithStream(dst, src, size, kind, stream)); }
}
};
static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allocation_type,
size_t size, hipMemcpyKind kind, bool enable_peer_access=false) {
size_t size, hipMemcpyKind kind, bool enable_peer_access = false) {
MemcpyWithStreamBenchmark benchmark;
benchmark.AddSectionName(std::to_string(size));
benchmark.AddSectionName(GetAllocationSectionName(src_allocation_type));
@@ -51,7 +49,9 @@ static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allo
} else {
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
if (src_device == -1 && dst_device == -1) { return; }
if (src_device == -1 && dst_device == -1) {
return;
}
LinearAllocGuard<int> src_allocation(LinearAllocs::hipMalloc, size);
HIP_CHECK(hipSetDevice(dst_device));
@@ -189,7 +189,8 @@ TEST_CASE("Performance_hipMemcpyWithStream_DeviceToDevice_EnablePeerAccess") {
const auto allocation_size = GENERATE(4_KB, 4_MB, 16_MB);
const auto src_allocation_type = LinearAllocs::hipMalloc;
const auto dst_allocation_type = LinearAllocs::hipMalloc;
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice, true);
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice,
true);
}
/**
@@ -79,6 +79,6 @@ TEST_CASE("Performance_hipMemset") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -72,6 +72,6 @@ TEST_CASE("Performance_hipMemset2D") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -75,6 +75,6 @@ TEST_CASE("Performance_hipMemset2DAsync") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -73,6 +73,6 @@ TEST_CASE("Performance_hipMemset3D") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -75,6 +75,6 @@ TEST_CASE("Performance_hipMemset3DAsync") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -81,6 +81,6 @@ TEST_CASE("Performance_hipMemsetAsync") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -80,6 +80,6 @@ TEST_CASE("Performance_hipMemsetD16") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -82,6 +82,6 @@ TEST_CASE("Performance_hipMemsetD16Async") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -80,6 +80,6 @@ TEST_CASE("Performance_hipMemsetD32") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -82,6 +82,6 @@ TEST_CASE("Performance_hipMemsetD32Async") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -80,6 +80,6 @@ TEST_CASE("Performance_hipMemsetD8") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -82,6 +82,6 @@ TEST_CASE("Performance_hipMemsetD8Async") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -60,11 +60,9 @@ static void RunBenchmark() {
* - Platform specific (AMD)
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Performance_hipExtStreamCreateWithCUMask") {
RunBenchmark();
}
TEST_CASE("Performance_hipExtStreamCreateWithCUMask") { RunBenchmark(); }
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -62,11 +62,9 @@ static void RunBenchmark() {
* - Platform specific (AMD)
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Performance_hipExtStreamGetCUMask") {
RunBenchmark();
}
TEST_CASE("Performance_hipExtStreamGetCUMask") { RunBenchmark(); }
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -32,11 +32,10 @@ class FreeAsyncBenchmark : public Benchmark<FreeAsyncBenchmark> {
const StreamGuard stream_guard{Streams::created};
const hipStream_t stream = stream_guard.stream();
float* dev_ptr{nullptr};
HIP_CHECK(hipMallocAsync(reinterpret_cast<void**>(&dev_ptr), array_size * sizeof(float), stream));
HIP_CHECK(
hipMallocAsync(reinterpret_cast<void**>(&dev_ptr), array_size * sizeof(float), stream));
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
HIP_CHECK(hipFreeAsync(dev_ptr, stream));
}
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) { HIP_CHECK(hipFreeAsync(dev_ptr, stream)); }
HIP_CHECK(hipStreamSynchronize(stream));
}
@@ -69,6 +68,6 @@ TEST_CASE("Performance_hipFreeAsync") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -34,7 +34,8 @@ class MallocAsyncBenchmark : public Benchmark<MallocAsyncBenchmark> {
float* dev_ptr{nullptr};
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
HIP_CHECK(hipMallocAsync(reinterpret_cast<void**>(&dev_ptr), array_size * sizeof(float), stream));
HIP_CHECK(
hipMallocAsync(reinterpret_cast<void**>(&dev_ptr), array_size * sizeof(float), stream));
}
HIP_CHECK(hipStreamSynchronize(stream));
HIP_CHECK(hipFree(dev_ptr));
@@ -68,6 +69,6 @@ TEST_CASE("Performance_hipMallocAsync") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -73,8 +73,9 @@ static void RunBenchmark(const size_t array_size) {
*/
TEST_CASE("Performance_hipMallocFromPoolAsync") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
size_t array_size = GENERATE(4_KB, 4_MB, 16_MB);
@@ -82,6 +83,6 @@ TEST_CASE("Performance_hipMallocFromPoolAsync") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -31,9 +31,7 @@ class MemPoolCreateBenchmark : public Benchmark<MemPoolCreateBenchmark> {
hipMemPool_t mem_pool{nullptr};
hipMemPoolProps pool_props = CreateMemPoolProps(0, hipMemHandleTypeNone);
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props)); }
REQUIRE(mem_pool != nullptr);
HIP_CHECK(hipMemPoolDestroy(mem_pool));
@@ -63,14 +61,15 @@ static void RunBenchmark() {
*/
TEST_CASE("Performance_hipMemPoolCreate") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
RunBenchmark();
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -32,9 +32,7 @@ class MemPoolDestroyBenchmark : public Benchmark<MemPoolDestroyBenchmark> {
hipMemPoolProps pool_props = CreateMemPoolProps(0, hipMemHandleTypeNone);
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolDestroy(mem_pool));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolDestroy(mem_pool)); }
}
};
@@ -62,14 +60,15 @@ static void RunBenchmark() {
*/
TEST_CASE("Performance_hipMemPoolDestroy") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
RunBenchmark();
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -37,9 +37,7 @@ class MemPoolExportPointerBenchmark : public Benchmark<MemPoolExportPointerBench
HIP_CHECK(hipMallocFromPoolAsync(&device_ptr, array_size * sizeof(float), mem_pool, nullptr));
HIP_CHECK(hipStreamSynchronize(nullptr));
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolExportPointer(&exp_data, device_ptr));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolExportPointer(&exp_data, device_ptr)); }
HIP_CHECK(hipFreeAsync(device_ptr, nullptr));
HIP_CHECK(hipMemPoolDestroy(mem_pool));
@@ -75,8 +73,9 @@ static void RunBenchmark(const size_t array_size) {
*/
TEST_CASE("Performance_hipMemPoolExportPointer") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
size_t array_size = GENERATE(4_KB, 4_MB, 16_MB);
@@ -84,6 +83,6 @@ TEST_CASE("Performance_hipMemPoolExportPointer") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -25,7 +25,8 @@ THE SOFTWARE.
* @ingroup PerformanceTest
*/
class MemPoolExportToShareableHandleBenchmark : public Benchmark<MemPoolExportToShareableHandleBenchmark> {
class MemPoolExportToShareableHandleBenchmark
: public Benchmark<MemPoolExportToShareableHandleBenchmark> {
public:
void operator()() {
hipMemPool_t mem_pool{nullptr};
@@ -66,14 +67,15 @@ static void RunBenchmark() {
*/
TEST_CASE("Performance_hipMemPoolExportToShareableHandle") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
RunBenchmark();
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -33,13 +33,8 @@ class MemPoolGetAccessBenchmark : public Benchmark<MemPoolGetAccessBenchmark> {
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
hipMemAccessFlags flags = hipMemAccessFlagsProtNone;
hipMemLocation location = {
hipMemLocationTypeDevice,
0
};
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolGetAccess(&flags, mem_pool, location));
}
hipMemLocation location = {hipMemLocationTypeDevice, 0};
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolGetAccess(&flags, mem_pool, location)); }
HIP_CHECK(hipMemPoolDestroy(mem_pool));
}
@@ -68,14 +63,15 @@ static void RunBenchmark() {
*/
TEST_CASE("Performance_hipMemPoolGetAccess") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
RunBenchmark();
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -34,9 +34,7 @@ class MemPoolGetAttributeBenchmark : public Benchmark<MemPoolGetAttributeBenchma
uint64_t value{0};
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolGetAttribute(mem_pool, attribute, &value));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolGetAttribute(mem_pool, attribute, &value)); }
HIP_CHECK(hipMemPoolDestroy(mem_pool));
}
@@ -71,18 +69,18 @@ static void RunBenchmark(const hipMemPoolAttr attribute) {
*/
TEST_CASE("Performance_hipMemPoolGetAttribute") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
hipMemPoolAttr attribute = GENERATE(hipMemPoolAttrReleaseThreshold,
hipMemPoolReuseFollowEventDependencies,
hipMemPoolReuseAllowOpportunistic,
hipMemPoolReuseAllowInternalDependencies);
hipMemPoolAttr attribute =
GENERATE(hipMemPoolAttrReleaseThreshold, hipMemPoolReuseFollowEventDependencies,
hipMemPoolReuseAllowOpportunistic, hipMemPoolReuseAllowInternalDependencies);
RunBenchmark(attribute);
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -25,7 +25,8 @@ THE SOFTWARE.
* @ingroup PerformanceTest
*/
class MemPoolImportFromShareableHandleBenchmark : public Benchmark<MemPoolImportFromShareableHandleBenchmark> {
class MemPoolImportFromShareableHandleBenchmark
: public Benchmark<MemPoolImportFromShareableHandleBenchmark> {
public:
void operator()() {
hipMemPool_t mem_pool{nullptr};
@@ -41,8 +42,8 @@ class MemPoolImportFromShareableHandleBenchmark : public Benchmark<MemPoolImport
HIP_CHECK(hipMemPoolExportToShareableHandle(&share_handle, mem_pool, kHandleType, 0));
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolImportFromShareableHandle(
&mem_pool_shareable, (void*)share_handle, kHandleType, 0));
HIP_CHECK(hipMemPoolImportFromShareableHandle(&mem_pool_shareable, (void*)share_handle,
kHandleType, 0));
}
HIP_CHECK(hipMemPoolDestroy(mem_pool_shareable));
@@ -74,14 +75,15 @@ static void RunBenchmark() {
*/
TEST_CASE("Performance_hipMemPoolImportFromShareableHandle") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
RunBenchmark();
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -40,7 +40,8 @@ class MemPoolImportPointerBenchmark : public Benchmark<MemPoolImportPointerBench
HIP_CHECK(hipMemPoolExportPointer(&exp_data, device_ptr));
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolImportPointer(reinterpret_cast<void**>(&device_ptr_import), mem_pool, &exp_data));
HIP_CHECK(hipMemPoolImportPointer(reinterpret_cast<void**>(&device_ptr_import), mem_pool,
&exp_data));
}
HIP_CHECK(hipFree(device_ptr));
@@ -78,8 +79,9 @@ static void RunBenchmark(const size_t array_size) {
*/
TEST_CASE("Performance_hipMemPoolImportPointer") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
size_t array_size = GENERATE(4_KB, 4_MB, 16_MB);
@@ -87,6 +89,6 @@ TEST_CASE("Performance_hipMemPoolImportPointer") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -32,17 +32,9 @@ class MemPoolSetAccessBenchmark : public Benchmark<MemPoolSetAccessBenchmark> {
hipMemPoolProps pool_props = CreateMemPoolProps(0, hipMemHandleTypeNone);
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
hipMemAccessDesc desc_list = {
{
hipMemLocationTypeDevice,
0
},
hipMemAccessFlagsProtReadWrite
};
hipMemAccessDesc desc_list = {{hipMemLocationTypeDevice, 0}, hipMemAccessFlagsProtReadWrite};
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolSetAccess(mem_pool, &desc_list, 1));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolSetAccess(mem_pool, &desc_list, 1)); }
HIP_CHECK(hipMemPoolDestroy(mem_pool));
}
@@ -71,14 +63,15 @@ static void RunBenchmark() {
*/
TEST_CASE("Performance_hipMemPoolSetAccess") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
RunBenchmark();
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -34,9 +34,7 @@ class MemPoolSetAttributeBenchmark : public Benchmark<MemPoolSetAttributeBenchma
int value{0};
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolSetAttribute(mem_pool, attribute, &value));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolSetAttribute(mem_pool, attribute, &value)); }
HIP_CHECK(hipMemPoolDestroy(mem_pool));
}
@@ -71,18 +69,18 @@ static void RunBenchmark(const hipMemPoolAttr attribute) {
*/
TEST_CASE("Performance_hipMemPoolSetAttribute") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
hipMemPoolAttr attribute = GENERATE(hipMemPoolAttrReleaseThreshold,
hipMemPoolReuseFollowEventDependencies,
hipMemPoolReuseAllowOpportunistic,
hipMemPoolReuseAllowInternalDependencies);
hipMemPoolAttr attribute =
GENERATE(hipMemPoolAttrReleaseThreshold, hipMemPoolReuseFollowEventDependencies,
hipMemPoolReuseAllowOpportunistic, hipMemPoolReuseAllowInternalDependencies);
RunBenchmark(attribute);
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -32,9 +32,7 @@ class MemPoolTrimToBenchmark : public Benchmark<MemPoolTrimToBenchmark> {
hipMemPoolProps pool_props = CreateMemPoolProps(0, hipMemHandleTypeNone);
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipMemPoolTrimTo(mem_pool, min_bytes_to_hold));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolTrimTo(mem_pool, min_bytes_to_hold)); }
HIP_CHECK(hipMemPoolDestroy(mem_pool));
}
@@ -68,8 +66,9 @@ static void RunBenchmark(const size_t min_bytes_to_hold) {
*/
TEST_CASE("Performance_hipMemPoolTrimTo") {
if (!AreMemPoolsSupported(0)) {
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
HipTest::HIP_SKIP_TEST(
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
"attribute. Hence skipping the testing with Pass result.\n");
return;
}
size_t min_bytes_to_hold = GENERATE(4_KB, 4_MB, 16_MB);
@@ -77,6 +76,6 @@ TEST_CASE("Performance_hipMemPoolTrimTo") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -34,9 +34,7 @@ class StreamAddCallbackBenchmark : public Benchmark<StreamAddCallbackBenchmark>
const StreamGuard stream_guard{Streams::created};
const hipStream_t stream = stream_guard.stream();
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipStreamAddCallback(stream, Callback, nullptr, 0));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipStreamAddCallback(stream, Callback, nullptr, 0)); }
}
};
@@ -56,11 +54,9 @@ static void RunBenchmark() {
* ------------------------
* - HIP_VERSION >= 5.2
*/
TEST_CASE("Performance_hipStreamAddCallback") {
RunBenchmark();
}
TEST_CASE("Performance_hipStreamAddCallback") { RunBenchmark(); }
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -27,12 +27,15 @@ THE SOFTWARE.
* @ingroup PerformanceTest
* Contains performance tests for all hipStream related APIs
*/
class HipDeviceGetStreamPriorityRangeBenchmark : public Benchmark<HipDeviceGetStreamPriorityRangeBenchmark> {
class HipDeviceGetStreamPriorityRangeBenchmark
: public Benchmark<HipDeviceGetStreamPriorityRangeBenchmark> {
public:
void operator()() {
int priority_min, priority_max;
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipDeviceGetStreamPriorityRange(&priority_min, &priority_max)); }
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipDeviceGetStreamPriorityRange(&priority_min, &priority_max));
}
}
};
@@ -42,19 +45,19 @@ class HipStreamQueryBenchmark : public Benchmark<HipStreamQueryBenchmark> {
hipError_t error;
hipStream_t stream;
HIP_CHECK(hipStreamCreate(&stream));
void *dptr;
if(perform_work) {
void* dptr;
if (perform_work) {
HIP_CHECK(hipMallocAsync(&dptr, 2048 * 4, stream));
}
TIMED_SECTION(kTimerTypeCpu) { error = hipStreamQuery(stream); }
if(perform_work) {
if (perform_work) {
HIP_CHECK(hipFreeAsync(dptr, stream));
HIP_CHECK(hipStreamSynchronize(stream));
}
HIP_CHECK(hipStreamDestroy(stream));
}
};
@@ -65,9 +68,9 @@ class HipStreamSynchronizeBenchmark : public Benchmark<HipStreamSynchronizeBench
hipError_t error;
hipStream_t stream;
HIP_CHECK(hipStreamCreate(&stream));
TIMED_SECTION(kTimerTypeCpu) { error = hipStreamSynchronize(stream); }
HIP_CHECK(hipStreamDestroy(stream));
}
};
@@ -93,23 +96,25 @@ class HipStreamCreateBenchmark : public Benchmark<HipStreamCreateBenchmark> {
}
};
class HipStreamCreateWithPriorityBenchmark : public Benchmark<HipStreamCreateWithPriorityBenchmark> {
class HipStreamCreateWithPriorityBenchmark
: public Benchmark<HipStreamCreateWithPriorityBenchmark> {
public:
void operator()(unsigned int flag) {
hipStream_t stream;
int priority_min, priority_max, priority_mid;
HIP_CHECK(hipDeviceGetStreamPriorityRange(&priority_min, &priority_max));
priority_mid = (priority_max + priority_min) / 2;
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipStreamCreateWithPriority(&stream, flag, priority_mid)); }
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipStreamCreateWithPriority(&stream, flag, priority_mid));
}
HIP_CHECK(hipStreamDestroy(stream));
}
};
static std::string GetStreamCreateFlagName(unsigned flag) {
switch (flag) {
case hipStreamDefault:
@@ -244,7 +249,7 @@ TEST_CASE("Performance_hipDeviceGetStreamPriorityRange") {
TEST_CASE("Performance_hipStreamQuery") {
const auto perform_work = GENERATE(true, false);
HipStreamQueryBenchmark benchmark;
if(perform_work) {
if (perform_work) {
benchmark.AddSectionName("stream with work");
} else {
benchmark.AddSectionName("stream without work");
@@ -269,6 +274,6 @@ TEST_CASE("Performance_hipStreamSynchronize") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -33,10 +33,8 @@ class StreamGetFlagsBenchmark : public Benchmark<StreamGetFlagsBenchmark> {
hipStream_t stream;
HIP_CHECK(hipStreamCreateWithFlags(&stream, expected_flag));
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipStreamGetFlags(stream, &returned_flags))
}
HIP_CHECK(hipStreamDestroy(stream));
TIMED_SECTION(kTimerTypeCpu){HIP_CHECK(hipStreamGetFlags(stream, &returned_flags))} HIP_CHECK(
hipStreamDestroy(stream));
}
};
@@ -75,6 +73,6 @@ TEST_CASE("Performance_hipStreamGetFlags") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -33,9 +33,7 @@ class StreamGetPriorityBenchmark : public Benchmark<StreamGetPriorityBenchmark>
const hipStream_t stream = stream_guard.stream();
int priority{};
TIMED_SECTION(kTimerTypeCpu) {
HIP_CHECK(hipStreamGetPriority(stream, &priority));
}
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipStreamGetPriority(stream, &priority)); }
}
};
@@ -74,6 +72,6 @@ TEST_CASE("Performance_hipStreamGetPriority") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -80,6 +80,6 @@ TEST_CASE("Performance_hipStreamWaitEvent") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -172,6 +172,6 @@ TEST_CASE("Performance_hipStreamWaitValue64") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -123,6 +123,6 @@ TEST_CASE("Performance_hipStreamWriteValue64") {
}
/**
* End doxygen group PerformanceTest.
* @}
*/
* End doxygen group PerformanceTest.
* @}
*/
@@ -33,73 +33,49 @@ THE SOFTWARE.
#include <map>
/**
* @addtogroup __reduce_op_sync __reduce_op_sync
* @{
* @ingroup WarpSyncPerformance
* __reduce_op_sync(MaskT mask, T val)
* Reduces the val as per the lanes described in mask and calculates the
* aggregated result
*/
* @addtogroup __reduce_op_sync __reduce_op_sync
* @{
* @ingroup WarpSyncPerformance
* __reduce_op_sync(MaskT mask, T val)
* Reduces the val as per the lanes described in mask and calculates the
* aggregated result
*/
static constexpr int kBlockDim = 1024;
template <class T>
struct AtomicAddOp {
__device__ T operator()(T* lhs, const T& rhs)
{
return atomicAdd(lhs, rhs);
}
template <class T> struct AtomicAddOp {
__device__ T operator()(T* lhs, const T& rhs) { return atomicAdd(lhs, rhs); }
};
template <class T>
struct AtomicMinOp {
__device__ T operator()(T* lhs, const T& rhs)
{
return atomicMin(lhs, rhs);
}
template <class T> struct AtomicMinOp {
__device__ T operator()(T* lhs, const T& rhs) { return atomicMin(lhs, rhs); }
};
template <class T>
struct AtomicMaxOp {
__device__ T operator()(T* lhs, const T& rhs)
{
return atomicMax(lhs, rhs);
}
template <class T> struct AtomicMaxOp {
__device__ T operator()(T* lhs, const T& rhs) { return atomicMax(lhs, rhs); }
};
template <class T>
struct AtomicAndOp {
__device__ T operator()(T* lhs, const T& rhs)
{
return atomicAnd(lhs, rhs);
}
template <class T> struct AtomicAndOp {
__device__ T operator()(T* lhs, const T& rhs) { return atomicAnd(lhs, rhs); }
};
template <class T>
struct AtomicOrOp {
__device__ T operator()(T* lhs, const T& rhs)
{
return atomicOr(lhs, rhs);
}
template <class T> struct AtomicOrOp {
__device__ T operator()(T* lhs, const T& rhs) { return atomicOr(lhs, rhs); }
};
template <class T>
struct AtomicXorOp {
__device__ T operator()(T* lhs, const T& rhs)
{
return atomicXor(lhs, rhs);
}
template <class T> struct AtomicXorOp {
__device__ T operator()(T* lhs, const T& rhs) { return atomicXor(lhs, rhs); }
};
// uses atomics to reduce the whole warp; depending on the mask our reduce should be faster
// @output to store the result, one per warp
// @numItems must be a multiple of warpSize
template <class T, template <typename> class Op>
__global__ void reduceAllAtomics(T* __restrict__ output, const T* __restrict__ input, unsigned long long mask)
{
__global__ void reduceAllAtomics(T* __restrict__ output, const T* __restrict__ input,
unsigned long long mask) {
int idx = threadIdx.x + blockIdx.x * kBlockDim;
extern __shared__ uint8_t shared_mem[];
T* result = reinterpret_cast<T*>(shared_mem); // one per warp
T* result = reinterpret_cast<T*>(shared_mem); // one per warp
Op<T> op;
int numWarp = threadIdx.x / warpSize;
@@ -115,18 +91,16 @@ __global__ void reduceAllAtomics(T* __restrict__ output, const T* __restrict__ i
__syncthreads();
if (mask & (1ul << __ockl_lane_u32()))
op(&result[numWarp], input[idx]);
if (mask & (1ul << __ockl_lane_u32())) op(&result[numWarp], input[idx]);
__syncthreads();
if (__ockl_lane_u32() == 0)
output[idx / warpSize] = result[numWarp];
if (__ockl_lane_u32() == 0) output[idx / warpSize] = result[numWarp];
}
template <class T, template<typename> class Op>
__global__ void reduceOpSync(T* __restrict__ output, const T* __restrict__ input, unsigned long long mask)
{
template <class T, template <typename> class Op>
__global__ void reduceOpSync(T* __restrict__ output, const T* __restrict__ input,
unsigned long long mask) {
int idx = threadIdx.x + blockIdx.x * kBlockDim;
T result;
@@ -146,18 +120,16 @@ __global__ void reduceOpSync(T* __restrict__ output, const T* __restrict__ input
else
static_assert(std::is_void<T>::value, "Unsupported operator");
if (__ockl_activelane_u32() == 0)
output[idx / warpSize] = result;
if (__ockl_activelane_u32() == 0) output[idx / warpSize] = result;
}
}
template <class T, template <typename> class Op>
class AtomicBenchmark : public Benchmark<AtomicBenchmark<T, Op>> {
public:
void operator()(T* output, const T* input, int numItems, unsigned long long mask)
{
dim3 blockDim = { kBlockDim };
dim3 gridDim = { static_cast<uint32_t>(std::ceil(numItems / static_cast<float>(blockDim.x))) };
public:
void operator()(T* output, const T* input, int numItems, unsigned long long mask) {
dim3 blockDim = {kBlockDim};
dim3 gridDim = {static_cast<uint32_t>(std::ceil(numItems / static_cast<float>(blockDim.x)))};
hipDeviceProp_t props;
HIP_CHECK(hipGetDeviceProperties(&props, 0));
@@ -187,11 +159,10 @@ public:
template <class T, template <typename> class Op>
class ReduceSyncBenchmark : public Benchmark<ReduceSyncBenchmark<T, Op>> {
public:
void operator()(T* output, T* input, int numItems, unsigned long long mask)
{
dim3 blockDim = { kBlockDim };
dim3 gridDim = { static_cast<uint32_t>(std::ceil(numItems / static_cast<float>(blockDim.x))) };
public:
void operator()(T* output, T* input, int numItems, unsigned long long mask) {
dim3 blockDim = {kBlockDim};
dim3 gridDim = {static_cast<uint32_t>(std::ceil(numItems / static_cast<float>(blockDim.x)))};
TIMED_SECTION(kTimerTypeEvent) {
@@ -202,8 +173,7 @@ public:
};
template <class T, template <typename> class Op>
void checkResults(T* d_atomicsResult, T* d_reduceResult, size_t numBytes, unsigned long long mask)
{
void checkResults(T* d_atomicsResult, T* d_reduceResult, size_t numBytes, unsigned long long mask) {
using namespace Catch::Matchers;
LinearAllocGuard<T> outputAtomic(LinearAllocs::malloc, numBytes);
LinearAllocGuard<T> outputReduce(LinearAllocs::malloc, numBytes);
@@ -229,47 +199,38 @@ void checkResults(T* d_atomicsResult, T* d_reduceResult, size_t numBytes, unsign
}
}
template <class T, template <typename> class Op>
struct IsLogicalOp {
template <class T, template <typename> class Op> struct IsLogicalOp {
static constexpr bool value = false;
};
template <class T>
struct IsLogicalOp<T, std::logical_and> {
template <class T> struct IsLogicalOp<T, std::logical_and> {
static constexpr bool value = true;
};
template <class T>
struct IsLogicalOp<T, std::logical_or> {
template <class T> struct IsLogicalOp<T, std::logical_or> {
static constexpr bool value = true;
};
template <class T>
struct IsLogicalOp<T, XorOp> {
template <class T> struct IsLogicalOp<T, XorOp> {
static constexpr bool value = true;
};
// Neither long long or fp16 have atomic operations. In those cases
// we only benchmark reduce sync operations, we cannot compare with native atomics
template <class T>
struct HasAtomicOps {
template <class T> struct HasAtomicOps {
static constexpr bool value = true;
};
template <>
struct HasAtomicOps<half> {
template <> struct HasAtomicOps<half> {
static constexpr bool value = false;
};
template <>
struct HasAtomicOps<long long> {
template <> struct HasAtomicOps<long long> {
static constexpr bool value = false;
};
template <class T, template <typename> class Op>
struct ReduceBenchmark {
void Run()
{
template <class T, template <typename> class Op> struct ReduceBenchmark {
void Run() {
static constexpr int numMasks = 6;
using distribution = typename DistributionType<T>::type;
ReduceSyncBenchmark<T, Op> benchmarkReduce;
@@ -287,27 +248,27 @@ struct ReduceBenchmark {
distribution dist;
int halfWaveSize = wavefrontSize / 2;
unsigned long long halfBitsOn = (1ul << (wavefrontSize / 2)) - 1;
unsigned long long fullMask = -1ul,
halfHighBitsOn = halfBitsOn << halfWaveSize,
unsigned long long fullMask = -1ul, halfHighBitsOn = halfBitsOn << halfWaveSize,
high16BitsOn = halfBitsOn << (wavefrontSize - 16),
high8BitsOn = halfBitsOn << (wavefrontSize - 8),
high4BitsOn = halfBitsOn << (wavefrontSize - 4),
allButOne = -1 & ~1;
high4BitsOn = halfBitsOn << (wavefrontSize - 4), allButOne = -1 & ~1;
const char* typeStr = typeToString<T>();
const char* opStr = opToString<T, Op>();
std::map<std::string, unsigned long long> masks;
std::pair<std::string, unsigned long long> masksPairs[] = { { "full mask", fullMask },
{ "high order 32 bits on", halfHighBitsOn },
{ "high order 16 bits on", high16BitsOn },
{ "high order 8 bits on", high8BitsOn },
{ "high order 4 bits on", high4BitsOn },
{ "all but one", allButOne } };
std::pair<std::string, unsigned long long> masksPairs[] = {
{"full mask", fullMask},
{"high order 32 bits on", halfHighBitsOn},
{"high order 16 bits on", high16BitsOn},
{"high order 8 bits on", high8BitsOn},
{"high order 4 bits on", high4BitsOn},
{"all but one", allButOne}};
int pos = 0, numMask = 0;
for (auto& mask : masksPairs) {
// don't use 'halfHighBitsOn' on warp size 32; it's the same as high16BitsOn
if (wavefrontSize != 32 || mask.second != halfHighBitsOn) {
masks.emplace(std::to_string(numMask) + " - " + mask.first, wavefrontSize == 64? mask.second : mask.second & 0xFFFFFFFF);
masks.emplace(std::to_string(numMask) + " - " + mask.first,
wavefrontSize == 64 ? mask.second : mask.second & 0xFFFFFFFF);
numMask++;
}
}
@@ -315,8 +276,7 @@ struct ReduceBenchmark {
// avoid generating values different than 1 or 0 for logical operators;
// otherwise the atomic version of the kernels would produce different results as
// atomicAnd/Or() are bitwise operations, not logical
if constexpr (IsLogicalOp<T, Op>::value)
dist = distribution(0, 1);
if constexpr (IsLogicalOp<T, Op>::value) dist = distribution(0, 1);
for (int i = 0; i < numItems; i++) {
input.ptr()[i] = dist(gen);
@@ -356,7 +316,8 @@ struct ReduceBenchmark {
printf("Checking results...\n");
for (const auto& mask : masks) {
checkResults<T, Op>(d_outputsAtomic[pos].ptr(), d_outputsReduce[pos].ptr(), outputNumBytes, mask.second);
checkResults<T, Op>(d_outputsAtomic[pos].ptr(), d_outputsReduce[pos].ptr(), outputNumBytes,
mask.second);
pos++;
}
}
@@ -370,31 +331,36 @@ TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Add", "", int, unsigned int, unsigne
benchmark.Run();
}
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Min", "", int, unsigned int, unsigned long long, long long, float, half, double) {
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Min", "", int, unsigned int, unsigned long long,
long long, float, half, double) {
ReduceBenchmark<TestType, MinOp> benchmark;
benchmark.Run();
}
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Max", "", int, unsigned int, unsigned long long, long long, float, half, double) {
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Max", "", int, unsigned int, unsigned long long,
long long, float, half, double) {
ReduceBenchmark<TestType, MaxOp> benchmark;
benchmark.Run();
}
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_And", "", int, unsigned int, unsigned long long, long long) {
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_And", "", int, unsigned int, unsigned long long,
long long) {
ReduceBenchmark<TestType, std::logical_and> benchmark;
benchmark.Run();
}
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Or", "", int, unsigned int, unsigned long long, long long) {
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Or", "", int, unsigned int, unsigned long long,
long long) {
ReduceBenchmark<TestType, std::logical_or> benchmark;
benchmark.Run();
}
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Xor", "", int, unsigned int, unsigned long long, long long) {
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Xor", "", int, unsigned int, unsigned long long,
long long) {
ReduceBenchmark<TestType, XorOp> benchmark;
benchmark.Run();
@@ -21,13 +21,13 @@ THE SOFTWARE.
#include <hip_test_defgroups.hh>
#include <unistd.h>
#include <vector>
#define HIP_CHECK_PERF(a) \
{ \
auto err = a; \
if ((err != hipSuccess) && (err != hipErrorNotReady)) { \
printf(#a "= Error! %s\n", hipGetErrorString(err)); \
exit(1); \
} \
#define HIP_CHECK_PERF(a) \
{ \
auto err = a; \
if ((err != hipSuccess) && (err != hipErrorNotReady)) { \
printf(#a "= Error! %s\n", hipGetErrorString(err)); \
exit(1); \
} \
}
/**
* @addtogroup hipEventRecord hipEventRecord
@@ -40,7 +40,7 @@ __global__ void null_kernel() {
__shared__ int temp[256];
temp[threadIdx.x] = sinf(float(threadIdx.x));
}
void rocm_null_gpu_job(void *stream) {
void rocm_null_gpu_job(void* stream) {
hipLaunchKernelGGL(null_kernel, 1, 256, 0, (hipStream_t)stream);
}
std::vector<std::vector<hipStream_t>> stream_pool;
@@ -48,12 +48,12 @@ std::atomic<int> counter(0);
bool do_kill = false;
std::chrono::system_clock::time_point thread_reports[16];
void thread_job(int dev, int virt) {
HIP_CHECK_PERF(hipSetDevice(dev)); // use dev
uint8_t *mem;
HIP_CHECK_PERF(hipSetDevice(dev)); // use dev
uint8_t* mem;
HIP_CHECK_PERF(hipMalloc(&mem, 512));
void *hmem2;
void* hmem2;
HIP_CHECK_PERF(hipHostAlloc(&hmem2, 512, 0));
uint8_t *hmem = (uint8_t *)hmem2;
uint8_t* hmem = (uint8_t*)hmem2;
hipStream_t exec_stream = stream_pool[dev][virt];
hipStream_t h2d_stream = stream_pool[dev][virt + 4];
hipStream_t d2h_stream = stream_pool[dev][virt + 8];
@@ -63,10 +63,8 @@ void thread_job(int dev, int virt) {
uint64_t n = 0;
while (!do_kill) {
rocm_null_gpu_job(exec_stream);
HIP_CHECK_PERF(
hipMemcpyAsync(hmem, mem, 4, hipMemcpyDeviceToHost, d2h_stream));
HIP_CHECK_PERF(hipMemcpyAsync(mem + 256, hmem + 256, 4,
hipMemcpyHostToDevice, h2d_stream));
HIP_CHECK_PERF(hipMemcpyAsync(hmem, mem, 4, hipMemcpyDeviceToHost, d2h_stream));
HIP_CHECK_PERF(hipMemcpyAsync(mem + 256, hmem + 256, 4, hipMemcpyHostToDevice, h2d_stream));
HIP_CHECK_PERF(hipEventRecord(eh2d, h2d_stream));
HIP_CHECK_PERF(hipEventRecord(ed2h, d2h_stream));
HIP_CHECK_PERF(hipEventQuery(eh2d));
@@ -109,14 +107,13 @@ TEST_CASE("Unit_hipEventOverFlow_PerfTest") {
HIP_CHECK_PERF(hipGetDeviceCount(&mgpu));
stream_pool.resize(mgpu);
HIP_CHECK_PERF(hipSetDeviceFlags(hipDeviceScheduleSpin));
std::vector<uint8_t *> memory_buffers[2];
std::vector<uint8_t*> memory_buffers[2];
for (int i = 0; i < mgpu; i++) {
HIP_CHECK_PERF(hipSetDevice(i));
stream_pool[i].resize(12);
memory_buffers[i].resize(128);
for (int j = 0; j < 12; j++)
HIP_CHECK_PERF(
hipStreamCreateWithFlags(&stream_pool[i][j], hipStreamNonBlocking));
HIP_CHECK_PERF(hipStreamCreateWithFlags(&stream_pool[i][j], hipStreamNonBlocking));
for (int j = 0; j < 128; j++)
HIP_CHECK_PERF(hipMalloc(&memory_buffers[i][j], 4096 * ((j & 1) + 1)));
}
@@ -125,8 +122,7 @@ TEST_CASE("Unit_hipEventOverFlow_PerfTest") {
printf("RUNNING ON %d DEVICES\n", nDev);
do_kill = false;
std::vector<std::thread> threads;
for (int i = 0; i < nDev * 4; i++)
threads.push_back(std::thread(thread_job, i / 4, i % 4));
for (int i = 0; i < nDev * 4; i++) threads.push_back(std::thread(thread_job, i / 4, i % 4));
usleep(1000000);
auto t1 = std::chrono::system_clock::now();
int count = int(counter);
@@ -135,15 +131,12 @@ TEST_CASE("Unit_hipEventOverFlow_PerfTest") {
for (int t = 0; t < 10; t++) {
usleep(1000000);
auto t2 = std::chrono::system_clock::now();
auto duration =
std::chrono::duration_cast<std::chrono::microseconds>(t2 - t1)
.count();
auto duration = std::chrono::duration_cast<std::chrono::microseconds>(t2 - t1).count();
int count2 = int(counter);
for (int i = 0; i < nDev * 4; i++) {
if (std::chrono::duration_cast<std::chrono::microseconds>(
t2 - thread_reports[i])
.count() >= 1000000) {
printf("Thread %d/%d is stuck\n", i/4, i%4);
if (std::chrono::duration_cast<std::chrono::microseconds>(t2 - thread_reports[i]).count() >=
1000000) {
printf("Thread %d/%d is stuck\n", i / 4, i % 4);
}
}
total_count += count2 - count;
@@ -153,8 +146,7 @@ TEST_CASE("Unit_hipEventOverFlow_PerfTest") {
}
printf("AVERAGE: %ld / %f = %f job/s\n", total_count, total_time, total_count / total_time);
do_kill = true;
for (auto &t : threads)
t.join();
for (auto& t : threads) t.join();
for (int i = 0; i < nDev; i++) {
HIP_CHECK_PERF(hipSetDevice(i));
HIP_CHECK_PERF(hipDeviceSynchronize());
@@ -21,13 +21,13 @@ THE SOFTWARE.
#include <hip_test_defgroups.hh>
#include <unistd.h>
#include <vector>
#define HIP_CHECK_PERF(a) \
{ \
auto err = a; \
if ((err != hipSuccess) && (err != hipErrorNotReady)) { \
printf(#a "= Error! %s\n", hipGetErrorString(err)); \
exit(1); \
} \
#define HIP_CHECK_PERF(a) \
{ \
auto err = a; \
if ((err != hipSuccess) && (err != hipErrorNotReady)) { \
printf(#a "= Error! %s\n", hipGetErrorString(err)); \
exit(1); \
} \
}
/**
* @addtogroup hipLaunchKernelGGL hipLaunchKernelGGL
@@ -38,7 +38,7 @@ __global__ void empty_kernel() {
__shared__ int temp[256];
temp[threadIdx.x] = sinf(float(threadIdx.x));
}
void rocm_empty_gpu_job(void *stream) {
void rocm_empty_gpu_job(void* stream) {
hipLaunchKernelGGL(empty_kernel, 1, 256, 0, (hipStream_t)stream);
}
std::vector<std::vector<hipStream_t>> stream_pools;
@@ -84,16 +84,14 @@ TEST_CASE("Unit_hipKernelLookUp_PerfTest") {
HIP_CHECK_PERF(hipSetDevice(i));
stream_pools[i].resize(12);
for (int j = 0; j < 12; j++)
HIP_CHECK_PERF(
hipStreamCreateWithFlags(&stream_pools[i][j], hipStreamNonBlocking));
HIP_CHECK_PERF(hipStreamCreateWithFlags(&stream_pools[i][j], hipStreamNonBlocking));
}
for (int nDev = 1; nDev <= mgpu; nDev++) {
count = 0;
INFO("RUNNING ON "<<nDev<<" DEVICES\n");
INFO("RUNNING ON " << nDev << " DEVICES\n");
kill = false;
std::vector<std::thread> threads;
for (int i = 0; i < nDev * 4; i++)
threads.push_back(std::thread(thread_jobs, i / 4, i % 4));
for (int i = 0; i < nDev * 4; i++) threads.push_back(std::thread(thread_jobs, i / 4, i % 4));
usleep(1000000);
auto t1 = std::chrono::system_clock::now();
int counter = int(count);
@@ -102,15 +100,12 @@ TEST_CASE("Unit_hipKernelLookUp_PerfTest") {
for (int t = 0; t < 10; t++) {
usleep(1000000);
auto t2 = std::chrono::system_clock::now();
auto duration =
std::chrono::duration_cast<std::chrono::microseconds>(t2 - t1)
.count();
auto duration = std::chrono::duration_cast<std::chrono::microseconds>(t2 - t1).count();
int counter2 = int(count);
for (int i = 0; i < nDev * 4; i++) {
if (std::chrono::duration_cast<std::chrono::microseconds>(
t2 - thread_report[i])
.count() >= 1000000) {
INFO("Thread "<<i/4<<"/"<<i%4<<" is stuck\n");
if (std::chrono::duration_cast<std::chrono::microseconds>(t2 - thread_report[i]).count() >=
1000000) {
INFO("Thread " << i / 4 << "/" << i % 4 << " is stuck\n");
}
}
total_count += counter2 - counter;
@@ -118,10 +113,9 @@ TEST_CASE("Unit_hipKernelLookUp_PerfTest") {
t1 = t2;
counter = counter2;
}
INFO("AVERAGE: "<<total_count<<"/"<<total_time<<" = "<<total_count / total_time);
INFO("AVERAGE: " << total_count << "/" << total_time << " = " << total_count / total_time);
kill = true;
for (auto &t : threads)
t.join();
for (auto& t : threads) t.join();
for (int i = 0; i < nDev; i++) {
HIP_CHECK_PERF(hipSetDevice(i));
HIP_CHECK_PERF(hipDeviceSynchronize());
@@ -39,7 +39,7 @@ static constexpr int launches = 5;
/**
* In fillKernel, all elements of the array filled with given value
*/
static __global__ void fillKernel(int *arr, int size, int value) {
static __global__ void fillKernel(int* arr, int size, int value) {
int offset = blockDim.x * blockIdx.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
for (int i = offset; i < size; i += stride) {
@@ -50,7 +50,7 @@ static __global__ void fillKernel(int *arr, int size, int value) {
/**
* In addOneKernel, all elements of the array are incremented by 1
*/
static __global__ void addOneKernel(int *arr, int size) {
static __global__ void addOneKernel(int* arr, int size) {
int offset = blockDim.x * blockIdx.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
for (int i = offset; i < size; i += stride) {
@@ -62,7 +62,7 @@ static __global__ void addOneKernel(int *arr, int size) {
* In addKernel, Array1 and Array2 will be added by element wise
* and stored in Array 1
*/
static __global__ void addKernel(int *arr1, int *arr2, int size) {
static __global__ void addKernel(int* arr1, int* arr2, int size) {
int offset = blockDim.x * blockIdx.x + threadIdx.x;
int stride = blockDim.x * gridDim.x;
for (int i = offset; i < size; i += stride) {
@@ -93,7 +93,7 @@ static __global__ void addKernel(int *arr1, int *arr2, int size) {
TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
constexpr int numberOfNodes = 1024;
int *devMem[numberOfNodes];
int* devMem[numberOfNodes];
for (int i = 0; i < numberOfNodes; i++) {
devMem[i] = nullptr;
}
@@ -116,17 +116,15 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
memAllocNodeParams.bytesize = sizeof(int);
if (i == 0) {
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0,
&memAllocNodeParams));
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0, &memAllocNodeParams));
} else {
::std::vector<hipGraphNode_t> memAllocNodeDependencies;
memAllocNodeDependencies.push_back(memAllocNode[i - 1]);
HIP_CHECK(hipGraphAddMemAllocNode(
&memAllocNode[i], graph, memAllocNodeDependencies.data(),
memAllocNodeDependencies.size(), &memAllocNodeParams));
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, memAllocNodeDependencies.data(),
memAllocNodeDependencies.size(), &memAllocNodeParams));
}
devMem[i] = reinterpret_cast<int *>(memAllocNodeParams.dptr);
devMem[i] = reinterpret_cast<int*>(memAllocNodeParams.dptr);
REQUIRE(devMem[i] != nullptr);
}
@@ -136,16 +134,16 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
memFreeNodeDependencies.push_back(memAllocNode[numberOfNodes - 1]);
HIP_CHECK(hipGraphAddMemFreeNode(
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(), reinterpret_cast<void *>(devMem[i])));
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(),
reinterpret_cast<void*>(devMem[i])));
} else {
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
memFreeNodeDependencies.push_back(memFreeNode[i - 1]);
HIP_CHECK(hipGraphAddMemFreeNode(
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(), reinterpret_cast<void *>(devMem[i])));
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(),
reinterpret_cast<void*>(devMem[i])));
}
}
@@ -165,27 +163,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
}
auto launch_stop = std::chrono::high_resolution_clock::now();
auto launch_result =
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto sync_start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipStreamSynchronize(stream));
auto sync_stop = std::chrono::high_resolution_clock::now();
auto sync_result =
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
std::cout << "Time taken to Execute : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
launch_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
<< " millisecs " << std::endl;
std::cout << "Time taken to Synchronize : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
sync_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
<< " millisecs " << std::endl;
HIP_CHECK(hipGraphExecDestroy(graphExec));
@@ -225,7 +217,7 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
constexpr int SIZE = 100;
char *dev[SIZE];
char* dev[SIZE];
for (int i = 0; i < SIZE; i++) {
dev[i] = nullptr;
}
@@ -241,8 +233,8 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
hipGraph_t graph;
HIP_CHECK(hipGraphCreate(&graph, 0));
hipGraphNode_t memAllocNode[SIZE], memsetNode[SIZE], kernelNode[SIZE],
memcpyNode[SIZE], memFreeNode[SIZE];
hipGraphNode_t memAllocNode[SIZE], memsetNode[SIZE], kernelNode[SIZE], memcpyNode[SIZE],
memFreeNode[SIZE];
// Prapare Mem alloc Nodes
for (int i = 0; i < SIZE; i++) {
@@ -254,24 +246,22 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
memAllocNodeParams.bytesize = sizeof(char);
if (i == 0) {
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0,
&memAllocNodeParams));
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0, &memAllocNodeParams));
} else {
::std::vector<hipGraphNode_t> memAllocNodeDependencies;
memAllocNodeDependencies.push_back(memAllocNode[i - 1]);
HIP_CHECK(hipGraphAddMemAllocNode(
&memAllocNode[i], graph, memAllocNodeDependencies.data(),
memAllocNodeDependencies.size(), &memAllocNodeParams));
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, memAllocNodeDependencies.data(),
memAllocNodeDependencies.size(), &memAllocNodeParams));
}
dev[i] = reinterpret_cast<char *>(memAllocNodeParams.dptr);
dev[i] = reinterpret_cast<char*>(memAllocNodeParams.dptr);
REQUIRE(dev[i] != nullptr);
}
// Prapare Memset Nodes
for (int i = 0; i < SIZE; i++) {
hipMemsetParams pMemsetParams{};
pMemsetParams.dst = reinterpret_cast<void *>(dev[i]);
pMemsetParams.dst = reinterpret_cast<void*>(dev[i]);
pMemsetParams.elementSize = 1;
pMemsetParams.height = 1;
pMemsetParams.pitch = 1;
@@ -284,21 +274,19 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
} else {
memsetNodeDependencies.push_back(memsetNode[i - 1]);
}
HIP_CHECK(hipGraphAddMemsetNode(
&memsetNode[i], graph, memsetNodeDependencies.data(),
memsetNodeDependencies.size(), &pMemsetParams));
HIP_CHECK(hipGraphAddMemsetNode(&memsetNode[i], graph, memsetNodeDependencies.data(),
memsetNodeDependencies.size(), &pMemsetParams));
}
// Prapare Kernel Nodes
for (int i = 0; i < SIZE; i++) {
hipKernelNodeParams kernelNodeParams{};
kernelNodeParams.func = reinterpret_cast<void *>(addOneKernel);
kernelNodeParams.func = reinterpret_cast<void*>(addOneKernel);
kernelNodeParams.gridDim = dim3(1, 1, 1);
kernelNodeParams.blockDim = dim3(1, 1, 1);
kernelNodeParams.sharedMemBytes = 0;
int size = 1;
void *kernelArgs[2] = {reinterpret_cast<void *>(&dev[i]),
reinterpret_cast<void *>(&size)};
void* kernelArgs[2] = {reinterpret_cast<void*>(&dev[i]), reinterpret_cast<void*>(&size)};
kernelNodeParams.kernelParams = kernelArgs;
kernelNodeParams.extra = nullptr;
@@ -309,9 +297,8 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
kernelNodeDependencies.push_back(kernelNode[i - 1]);
}
HIP_CHECK(hipGraphAddKernelNode(
&kernelNode[i], graph, kernelNodeDependencies.data(),
kernelNodeDependencies.size(), &kernelNodeParams));
HIP_CHECK(hipGraphAddKernelNode(&kernelNode[i], graph, kernelNodeDependencies.data(),
kernelNodeDependencies.size(), &kernelNodeParams));
}
// Prapare Memcpy Nodes
@@ -330,9 +317,8 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
} else {
memcpyNodeDependencies.push_back(memcpyNode[i - 1]);
}
HIP_CHECK(hipGraphAddMemcpyNode(
&memcpyNode[i], graph, memcpyNodeDependencies.data(),
memcpyNodeDependencies.size(), &pMemcpyParams));
HIP_CHECK(hipGraphAddMemcpyNode(&memcpyNode[i], graph, memcpyNodeDependencies.data(),
memcpyNodeDependencies.size(), &pMemcpyParams));
}
// Prapare Mem free Nodes
@@ -341,16 +327,16 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
memFreeNodeDependencies.push_back(memcpyNode[SIZE - 1]);
HIP_CHECK(hipGraphAddMemFreeNode(
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(), reinterpret_cast<void *>(dev[i])));
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(),
reinterpret_cast<void*>(dev[i])));
} else {
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
memFreeNodeDependencies.push_back(memFreeNode[i - 1]);
HIP_CHECK(hipGraphAddMemFreeNode(
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(), reinterpret_cast<void *>(dev[i])));
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(),
reinterpret_cast<void*>(dev[i])));
}
}
@@ -369,27 +355,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
HIP_CHECK(hipGraphLaunch(graphExec, stream));
}
auto launch_stop = std::chrono::high_resolution_clock::now();
auto launch_result =
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto sync_start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipStreamSynchronize(stream));
auto sync_stop = std::chrono::high_resolution_clock::now();
auto sync_result =
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
std::cout << "Time taken to Execute : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
launch_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
<< " millisecs " << std::endl;
std::cout << "Time taken to Synchronize : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
sync_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
<< " millisecs " << std::endl;
for (int i = 0; i < SIZE; i++) {
@@ -434,14 +414,14 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
constexpr int BRANCHES = 10;
int value = 100;
int *hostMemSrc = new int[SIZE];
int* hostMemSrc = new int[SIZE];
REQUIRE(hostMemSrc != nullptr);
int *devMemSrc1 = nullptr;
int* devMemSrc1 = nullptr;
HIP_CHECK(hipMalloc(&devMemSrc1, NBYTES));
REQUIRE(devMemSrc1 != nullptr);
int *devMemSrc2[BRANCHES];
int* devMemSrc2[BRANCHES];
for (int i = 0; i < BRANCHES; i++) {
devMemSrc2[i] = nullptr;
}
@@ -455,14 +435,12 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
hipGraph_t graph;
HIP_CHECK(hipGraphCreate(&graph, 0));
hipGraphNode_t memcpyNodeH2D, memAllocNode[BRANCHES],
fillKernelNode[BRANCHES], addKernelNode[BRANCHES],
memcpyNodeD2H[BRANCHES], memFreeNode[BRANCHES], memcpyNodeH2H;
hipGraphNode_t memcpyNodeH2D, memAllocNode[BRANCHES], fillKernelNode[BRANCHES],
addKernelNode[BRANCHES], memcpyNodeD2H[BRANCHES], memFreeNode[BRANCHES], memcpyNodeH2H;
// Add H2D Node
HIP_CHECK(hipGraphAddMemcpyNode1D(&memcpyNodeH2D, graph, nullptr, 0,
devMemSrc1, hostMemSrc, NBYTES,
hipMemcpyHostToDevice));
HIP_CHECK(hipGraphAddMemcpyNode1D(&memcpyNodeH2D, graph, nullptr, 0, devMemSrc1, hostMemSrc,
NBYTES, hipMemcpyHostToDevice));
for (int branch = 0; branch < BRANCHES; branch++) {
// Add Mem alloc Nodes
@@ -476,11 +454,10 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
memAllocNodeParams.poolProps.location.id = 0;
memAllocNodeParams.bytesize = NBYTES;
HIP_CHECK(hipGraphAddMemAllocNode(
&memAllocNode[branch], graph, memAllocNodeDependencies.data(),
memAllocNodeDependencies.size(), &memAllocNodeParams));
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[branch], graph, memAllocNodeDependencies.data(),
memAllocNodeDependencies.size(), &memAllocNodeParams));
devMemSrc2[branch] = reinterpret_cast<int *>(memAllocNodeParams.dptr);
devMemSrc2[branch] = reinterpret_cast<int*>(memAllocNodeParams.dptr);
REQUIRE(devMemSrc2[branch] != nullptr);
// Add Kernel Nodes (fillKernel)
@@ -488,57 +465,52 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
kernelNodeDependencies.push_back(memAllocNode[branch]);
hipKernelNodeParams kernelNodeParams{};
kernelNodeParams.func = reinterpret_cast<void *>(fillKernel);
kernelNodeParams.func = reinterpret_cast<void*>(fillKernel);
kernelNodeParams.gridDim = dim3(1, 1, 1);
kernelNodeParams.blockDim = dim3(1, 1, 1);
kernelNodeParams.sharedMemBytes = 0;
int size = SIZE;
void *kernelArgs[3] = {reinterpret_cast<void *>(&devMemSrc2[branch]),
reinterpret_cast<void *>(&size),
reinterpret_cast<void *>(&value)};
void* kernelArgs[3] = {reinterpret_cast<void*>(&devMemSrc2[branch]),
reinterpret_cast<void*>(&size), reinterpret_cast<void*>(&value)};
kernelNodeParams.kernelParams = kernelArgs;
kernelNodeParams.extra = nullptr;
HIP_CHECK(hipGraphAddKernelNode(
&fillKernelNode[branch], graph, kernelNodeDependencies.data(),
kernelNodeDependencies.size(), &kernelNodeParams));
HIP_CHECK(hipGraphAddKernelNode(&fillKernelNode[branch], graph, kernelNodeDependencies.data(),
kernelNodeDependencies.size(), &kernelNodeParams));
// Add Kernel Nodes (addKernel)
::std::vector<hipGraphNode_t> kernelNodeDependencies2;
kernelNodeDependencies2.push_back(fillKernelNode[branch]);
hipKernelNodeParams kernelNodeParams2{};
kernelNodeParams2.func = reinterpret_cast<void *>(addKernel);
kernelNodeParams2.func = reinterpret_cast<void*>(addKernel);
kernelNodeParams2.gridDim = dim3(1, 1, 1);
kernelNodeParams2.blockDim = dim3(1, 1, 1);
kernelNodeParams2.sharedMemBytes = 0;
int size2 = SIZE;
void *kernelArgs2[3] = {reinterpret_cast<void *>(&devMemSrc2[branch]),
reinterpret_cast<void *>(&devMemSrc1),
reinterpret_cast<void *>(&size2)};
void* kernelArgs2[3] = {reinterpret_cast<void*>(&devMemSrc2[branch]),
reinterpret_cast<void*>(&devMemSrc1), reinterpret_cast<void*>(&size2)};
kernelNodeParams2.kernelParams = kernelArgs2;
kernelNodeParams2.extra = nullptr;
HIP_CHECK(hipGraphAddKernelNode(
&addKernelNode[branch], graph, kernelNodeDependencies2.data(),
kernelNodeDependencies2.size(), &kernelNodeParams2));
HIP_CHECK(hipGraphAddKernelNode(&addKernelNode[branch], graph, kernelNodeDependencies2.data(),
kernelNodeDependencies2.size(), &kernelNodeParams2));
// Add D2H Nodes
::std::vector<hipGraphNode_t> memcpyNodeD2HDependencies;
memcpyNodeD2HDependencies.push_back(addKernelNode[branch]);
HIP_CHECK(hipGraphAddMemcpyNode1D(
&memcpyNodeD2H[branch], graph, memcpyNodeD2HDependencies.data(),
memcpyNodeD2HDependencies.size(), hostMemDst[branch],
devMemSrc2[branch], NBYTES, hipMemcpyDeviceToHost));
HIP_CHECK(hipGraphAddMemcpyNode1D(&memcpyNodeD2H[branch], graph,
memcpyNodeD2HDependencies.data(),
memcpyNodeD2HDependencies.size(), hostMemDst[branch],
devMemSrc2[branch], NBYTES, hipMemcpyDeviceToHost));
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
memFreeNodeDependencies.push_back(memcpyNodeD2H[branch]);
HIP_CHECK(hipGraphAddMemFreeNode(
&memFreeNode[branch], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(),
reinterpret_cast<void *>(devMemSrc2[branch])));
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[branch], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(),
reinterpret_cast<void*>(devMemSrc2[branch])));
}
// Add H2H Node
@@ -547,10 +519,9 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
memcpyNodeH2HDependencies.push_back(memFreeNode[i]);
}
HIP_CHECK(hipGraphAddMemcpyNode1D(
&memcpyNodeH2H, graph, memcpyNodeH2HDependencies.data(),
memcpyNodeH2HDependencies.size(), finalHostDst, hostMemDst,
BRANCHES * SIZE * sizeof(int), hipMemcpyHostToHost));
HIP_CHECK(hipGraphAddMemcpyNode1D(&memcpyNodeH2H, graph, memcpyNodeH2HDependencies.data(),
memcpyNodeH2HDependencies.size(), finalHostDst, hostMemDst,
BRANCHES * SIZE * sizeof(int), hipMemcpyHostToHost));
hipGraphExec_t graphExec;
HIP_CHECK(hipGraphInstantiateWithFlags(&graphExec, graph, 0));
@@ -570,27 +541,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
}
auto launch_stop = std::chrono::high_resolution_clock::now();
auto launch_result =
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto sync_start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipStreamSynchronize(stream));
auto sync_stop = std::chrono::high_resolution_clock::now();
auto sync_result =
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
std::cout << "Time taken to Execute : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
launch_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
<< " millisecs " << std::endl;
std::cout << "Time taken to Synchronize : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
sync_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
<< " millisecs " << std::endl;
for (int branch = 0; branch < BRANCHES; branch++) {
@@ -633,7 +598,7 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
constexpr int BRANCHES = 10;
char *dev[BRANCHES];
char* dev[BRANCHES];
for (int i = 0; i < BRANCHES; i++) {
dev[i] = nullptr;
}
@@ -649,8 +614,8 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
hipGraph_t graph;
HIP_CHECK(hipGraphCreate(&graph, 0));
hipGraphNode_t memAllocNode[BRANCHES], memsetNode[BRANCHES],
kernelNode[BRANCHES], memcpyNode[BRANCHES], memFreeNode[BRANCHES];
hipGraphNode_t memAllocNode[BRANCHES], memsetNode[BRANCHES], kernelNode[BRANCHES],
memcpyNode[BRANCHES], memFreeNode[BRANCHES];
// Prapare Mem alloc Nodes
for (int i = 0; i < BRANCHES; i++) {
@@ -661,14 +626,13 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
memAllocNodeParams.poolProps.location.id = 0;
memAllocNodeParams.bytesize = sizeof(char);
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0,
&memAllocNodeParams));
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0, &memAllocNodeParams));
dev[i] = reinterpret_cast<char *>(memAllocNodeParams.dptr);
dev[i] = reinterpret_cast<char*>(memAllocNodeParams.dptr);
REQUIRE(dev[i] != nullptr);
hipMemsetParams pMemsetParams{};
pMemsetParams.dst = reinterpret_cast<void *>(dev[i]);
pMemsetParams.dst = reinterpret_cast<void*>(dev[i]);
pMemsetParams.elementSize = 1;
pMemsetParams.height = 1;
pMemsetParams.pitch = 1;
@@ -678,27 +642,24 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
::std::vector<hipGraphNode_t> memsetNodeDependencies;
memsetNodeDependencies.push_back(memAllocNode[i]);
HIP_CHECK(hipGraphAddMemsetNode(
&memsetNode[i], graph, memsetNodeDependencies.data(),
memsetNodeDependencies.size(), &pMemsetParams));
HIP_CHECK(hipGraphAddMemsetNode(&memsetNode[i], graph, memsetNodeDependencies.data(),
memsetNodeDependencies.size(), &pMemsetParams));
hipKernelNodeParams kernelNodeParams{};
kernelNodeParams.func = reinterpret_cast<void *>(addOneKernel);
kernelNodeParams.func = reinterpret_cast<void*>(addOneKernel);
kernelNodeParams.gridDim = dim3(1, 1, 1);
kernelNodeParams.blockDim = dim3(1, 1, 1);
kernelNodeParams.sharedMemBytes = 0;
int size = 1;
void *kernelArgs[2] = {reinterpret_cast<void *>(&dev[i]),
reinterpret_cast<void *>(&size)};
void* kernelArgs[2] = {reinterpret_cast<void*>(&dev[i]), reinterpret_cast<void*>(&size)};
kernelNodeParams.kernelParams = kernelArgs;
kernelNodeParams.extra = nullptr;
::std::vector<hipGraphNode_t> kernelNodeDependencies;
kernelNodeDependencies.push_back(memsetNode[i]);
HIP_CHECK(hipGraphAddKernelNode(
&kernelNode[i], graph, kernelNodeDependencies.data(),
kernelNodeDependencies.size(), &kernelNodeParams));
HIP_CHECK(hipGraphAddKernelNode(&kernelNode[i], graph, kernelNodeDependencies.data(),
kernelNodeDependencies.size(), &kernelNodeParams));
hipMemcpy3DParms pMemcpyParams{};
pMemcpyParams.srcPos = make_hipPos(0, 0, 0);
@@ -711,16 +672,15 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
::std::vector<hipGraphNode_t> memcpyNodeDependencies;
memcpyNodeDependencies.push_back(kernelNode[i]);
HIP_CHECK(hipGraphAddMemcpyNode(
&memcpyNode[i], graph, memcpyNodeDependencies.data(),
memcpyNodeDependencies.size(), &pMemcpyParams));
HIP_CHECK(hipGraphAddMemcpyNode(&memcpyNode[i], graph, memcpyNodeDependencies.data(),
memcpyNodeDependencies.size(), &pMemcpyParams));
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
memFreeNodeDependencies.push_back(memcpyNode[i]);
HIP_CHECK(hipGraphAddMemFreeNode(
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(), reinterpret_cast<void *>(dev[i])));
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
memFreeNodeDependencies.size(),
reinterpret_cast<void*>(dev[i])));
}
hipGraphExec_t graphExec;
@@ -738,27 +698,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
HIP_CHECK(hipGraphLaunch(graphExec, stream));
}
auto launch_stop = std::chrono::high_resolution_clock::now();
auto launch_result =
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto sync_start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipStreamSynchronize(stream));
auto sync_stop = std::chrono::high_resolution_clock::now();
auto sync_result =
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
std::cout << "Time taken to Execute : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
launch_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
<< " millisecs " << std::endl;
std::cout << "Time taken to Synchronize : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
sync_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
<< " millisecs " << std::endl;
for (int i = 0; i < BRANCHES; i++) {
@@ -791,7 +745,7 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_OneBranchNoOps_AutoFreeOnLaunch") {
constexpr int SIZE = 1024;
int *devMem[SIZE];
int* devMem[SIZE];
for (int i = 0; i < SIZE; i++) {
devMem[i] = nullptr;
}
@@ -813,23 +767,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_OneBranchNoOps_AutoFreeOnLaunch") {
memAllocNodeParams.bytesize = sizeof(int);
if (i == 0) {
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0,
&memAllocNodeParams));
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0, &memAllocNodeParams));
} else {
::std::vector<hipGraphNode_t> memAllocNodeDependencies;
memAllocNodeDependencies.push_back(memAllocNode[i - 1]);
HIP_CHECK(hipGraphAddMemAllocNode(
&memAllocNode[i], graph, memAllocNodeDependencies.data(),
memAllocNodeDependencies.size(), &memAllocNodeParams));
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, memAllocNodeDependencies.data(),
memAllocNodeDependencies.size(), &memAllocNodeParams));
}
devMem[i] = reinterpret_cast<int *>(memAllocNodeParams.dptr);
devMem[i] = reinterpret_cast<int*>(memAllocNodeParams.dptr);
REQUIRE(devMem[i] != nullptr);
}
hipGraphExec_t graphExec;
HIP_CHECK(hipGraphInstantiateWithFlags(
&graphExec, graph, hipGraphInstantiateFlagAutoFreeOnLaunch));
HIP_CHECK(
hipGraphInstantiateWithFlags(&graphExec, graph, hipGraphInstantiateFlagAutoFreeOnLaunch));
// Warm up call
HIP_CHECK(hipGraphLaunch(graphExec, stream));
@@ -844,27 +796,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_OneBranchNoOps_AutoFreeOnLaunch") {
}
auto launch_stop = std::chrono::high_resolution_clock::now();
auto launch_result =
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
auto sync_start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipStreamSynchronize(stream));
auto sync_stop = std::chrono::high_resolution_clock::now();
auto sync_result =
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
std::cout << "Time taken to Execute : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
launch_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
<< " millisecs " << std::endl;
std::cout << "Time taken to Synchronize : "
<< std::chrono::duration_cast<std::chrono::milliseconds>(
sync_result)
.count()
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
<< " millisecs " << std::endl;
HIP_CHECK(hipGraphExecDestroy(graphExec));
@@ -26,7 +26,7 @@ THE SOFTWARE.
static constexpr int N = 1024;
static constexpr int Nbytes = N * sizeof(int);
static size_t NElem{N};
static constexpr int blocksPerCU = 6; // to hide latency
static constexpr int blocksPerCU = 6; // to hide latency
static constexpr int threadsPerBlock = 256;
// Num of parallel Branches
const unsigned int kNumNode = 5;
@@ -39,8 +39,7 @@ const unsigned int kNumNode = 5;
* - Launches an executable graph in the specified stream.
*/
unsigned blocks = HipTest::setNumBlocks(blocksPerCU, threadsPerBlock, N);
template <typename T>
__global__ void vectorADD(const T *A_d, const T *B_d, T *C_d, size_t NELEM) {
template <typename T> __global__ void vectorADD(const T* A_d, const T* B_d, T* C_d, size_t NELEM) {
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
size_t stride = blockDim.x * gridDim.x;
@@ -73,32 +72,31 @@ TEST_CASE("Unit_hipGraph_Performance_Improvement_ParallelGraph") {
HIP_CHECK(hipStreamCreate(&stream));
HIP_CHECK(hipGraphCreate(&graph, 0));
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy1, graph, nullptr, 0, A_d, A_h,
Nbytes, hipMemcpyHostToDevice));
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy2, graph, nullptr, 0, B_d, B_h,
Nbytes, hipMemcpyHostToDevice));
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy1, graph, nullptr, 0, A_d, A_h, Nbytes,
hipMemcpyHostToDevice));
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy2, graph, nullptr, 0, B_d, B_h, Nbytes,
hipMemcpyHostToDevice));
HIP_CHECK(hipGraphAddDependencies(graph, &memCpy1, &memCpy2, 1));
for (int i = 0; i < kNumNode; i++) {
hipKernelNodeParams kernelNodeParams{};
void *kernelArgs[] = {&A_d, &B_d, &C_d, reinterpret_cast<void *>(&NElem)};
kernelNodeParams.func = reinterpret_cast<void *>(vectorADD<int>);
void* kernelArgs[] = {&A_d, &B_d, &C_d, reinterpret_cast<void*>(&NElem)};
kernelNodeParams.func = reinterpret_cast<void*>(vectorADD<int>);
kernelNodeParams.gridDim = dim3(blocks);
kernelNodeParams.blockDim = dim3(threadsPerBlock);
kernelNodeParams.sharedMemBytes = 0;
kernelNodeParams.kernelParams = reinterpret_cast<void **>(kernelArgs);
kernelNodeParams.kernelParams = reinterpret_cast<void**>(kernelArgs);
kernelNodeParams.extra = nullptr;
HIP_CHECK(
hipGraphAddKernelNode(&kNode[i], graph, nullptr, 0, &kernelNodeParams));
HIP_CHECK(hipGraphAddKernelNode(&kNode[i], graph, nullptr, 0, &kernelNodeParams));
HIP_CHECK(hipGraphAddDependencies(graph, &memCpy2, &kNode[i], 1));
}
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy3, graph, nullptr, 0, C_h, C_d,
Nbytes, hipMemcpyDeviceToHost));
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy3, graph, nullptr, 0, C_h, C_d, Nbytes,
hipMemcpyDeviceToHost));
for (int i = 0; i < kNumNode; i++) {
HIP_CHECK(hipGraphAddDependencies(graph, &kNode[i], &memCpy3, 1));
}
hipGraphNode_t *nodes{nullptr};
hipGraphNode_t* nodes{nullptr};
size_t numNodes = 0;
HIP_CHECK(hipGraphGetNodes(graph, nodes, &numNodes));
INFO("Num of nodes in the graph: " << numNodes);
@@ -113,10 +111,9 @@ TEST_CASE("Unit_hipGraph_Performance_Improvement_ParallelGraph") {
// Stop time
auto stop = std::chrono::high_resolution_clock::now();
auto duration = stop - start;
INFO(
"Time taken for Graph: "
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
<< " milliSeconds");
INFO("Time taken for Graph: "
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
<< " milliSeconds");
// Verify graph execution result
HipTest::checkVectorADD(A_h, B_h, C_h, N);
@@ -150,18 +147,17 @@ TEST_CASE("Unit_hipGraph_Performance_With_Stream_Operations") {
HIP_CHECK(hipMemcpyAsync(A_d, A_h, Nbytes, hipMemcpyDefault, stream));
HIP_CHECK(hipMemcpyAsync(B_d, B_h, Nbytes, hipMemcpyDefault, stream));
for (int i = 0; i < kNumNode; i++) {
hipLaunchKernelGGL(vectorADD, dim3(blocks), dim3(threadsPerBlock), 0,
stream, A_d, B_d, C_d, NElem);
hipLaunchKernelGGL(vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, stream, A_d, B_d, C_d,
NElem);
}
HIP_CHECK(hipMemcpyAsync(C_h, C_d, Nbytes, hipMemcpyDefault, stream));
HIP_CHECK(hipStreamSynchronize(stream));
}
auto stop = std::chrono::high_resolution_clock::now();
auto duration = stop - start;
INFO(
"Time taken for Stream: "
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
<< " milliSeconds");
INFO("Time taken for Stream: "
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
<< " milliSeconds");
// Verify graph execution result
HipTest::checkVectorADD(A_h, B_h, C_h, N);
@@ -193,8 +189,8 @@ TEST_CASE("Unit_hipGraph_Performance_With_Stream_Capture") {
HIP_CHECK(hipMemcpyAsync(A_d, A_h, Nbytes, hipMemcpyDefault, stream));
HIP_CHECK(hipMemcpyAsync(B_d, B_h, Nbytes, hipMemcpyDefault, stream));
for (int i = 0; i < kNumNode; i++) {
hipLaunchKernelGGL(vectorADD, dim3(blocks), dim3(threadsPerBlock), 0,
stream, A_d, B_d, C_d, NElem);
hipLaunchKernelGGL(vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, stream, A_d, B_d, C_d,
NElem);
}
HIP_CHECK(hipMemcpyAsync(C_h, C_d, Nbytes, hipMemcpyDefault, stream));
HIP_CHECK(hipStreamEndCapture(stream, &graph));
@@ -208,10 +204,9 @@ TEST_CASE("Unit_hipGraph_Performance_With_Stream_Capture") {
HIP_CHECK(hipStreamSynchronize(streamForGraph));
auto stop = std::chrono::high_resolution_clock::now();
auto duration = stop - start;
INFO(
"Time taken for Graph via Stream Capture: "
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
<< " milliSeconds");
INFO("Time taken for Graph via Stream Capture: "
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
<< " milliSeconds");
// Verify graph execution result
HipTest::checkVectorADD(A_h, B_h, C_h, N);
@@ -223,4 +218,3 @@ TEST_CASE("Unit_hipGraph_Performance_With_Stream_Capture") {
* End doxygen group GraphTest.
* @}
*/
@@ -18,40 +18,39 @@ THE SOFTWARE.
*/
/**
* @addtogroup hipMemcpyAsync hipMemcpyAsync
* @{
* @ingroup perfMemoryTest
* `hipMemcpyAsync(void* dst, const void* src, size_t count,
* hipMemcpyKind kind, hipStream_t stream = 0)` -
* Copies data between host and device.
*/
* @addtogroup hipMemcpyAsync hipMemcpyAsync
* @{
* @ingroup perfMemoryTest
* `hipMemcpyAsync(void* dst, const void* src, size_t count,
* hipMemcpyKind kind, hipStream_t stream = 0)` -
* Copies data between host and device.
*/
#include <hip_test_common.hh>
#define NUM_SIZES 9
// 4KB, 8KB, 64KB, 256KB, 1 MB, 4MB, 16 MB, 16MB+10
static const unsigned int Sizes[NUM_SIZES] =
{4096, 8192, 65536, 262144, 524288, 1048576, 4194304, 16777216, 16777216+10};
static const unsigned int Sizes[NUM_SIZES] = {4096, 8192, 65536, 262144, 524288,
1048576, 4194304, 16777216, 16777216 + 10};
static const unsigned int Iterations[2] = {1, 1000};
#define BUF_TYPES 4
// 16 ways to combine 4 different buffer types
#define NUM_SUBTESTS (BUF_TYPES*BUF_TYPES)
#define NUM_SUBTESTS (BUF_TYPES * BUF_TYPES)
static void setData(void *ptr, unsigned int size, char value) {
char *ptr2 = reinterpret_cast<char *>(ptr);
for (unsigned int i = 0; i < size ; i++) {
static void setData(void* ptr, unsigned int size, char value) {
char* ptr2 = reinterpret_cast<char*>(ptr);
for (unsigned int i = 0; i < size; i++) {
ptr2[i] = value;
}
}
static void checkData(void *ptr, unsigned int size, char value) {
char *ptr2 = reinterpret_cast<char *>(ptr);
static void checkData(void* ptr, unsigned int size, char value) {
char* ptr2 = reinterpret_cast<char*>(ptr);
for (unsigned int i = 0; i < size; i++) {
if (ptr2[i] != value) {
INFO("Validation failed at " << i << " Got " << ptr2[i] <<
" Expected " << value);
INFO("Validation failed at " << i << " Got " << ptr2[i] << " Expected " << value);
REQUIRE(false);
}
}
@@ -63,17 +62,17 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
bool hostMalloc[2] = {false};
bool hostRegister[2] = {false};
bool unpinnedMalloc[2] = {false};
void *memptr[2] = {NULL};
void *alignedmemptr[2] = {NULL};
void *srcBuffer = NULL;
void *dstBuffer = NULL;
void* memptr[2] = {NULL};
void* alignedmemptr[2] = {NULL};
void* srcBuffer = NULL;
void* dstBuffer = NULL;
int numTests = (p_tests == -1) ? (NUM_SIZES*NUM_SUBTESTS*2 - 1) : p_tests;
int numTests = (p_tests == -1) ? (NUM_SIZES * NUM_SUBTESTS * 2 - 1) : p_tests;
int test = (p_tests == -1) ? 0 : p_tests;
for ( ; test <= numTests; test++ ) {
for (; test <= numTests; test++) {
unsigned int srcTest = (test / NUM_SIZES) % BUF_TYPES;
unsigned int dstTest = (test / (NUM_SIZES*BUF_TYPES)) % BUF_TYPES;
unsigned int dstTest = (test / (NUM_SIZES * BUF_TYPES)) % BUF_TYPES;
bufSize_ = Sizes[test % NUM_SIZES];
hostMalloc[0] = hostMalloc[1] = false;
hostRegister[0] = hostRegister[1] = false;
@@ -101,8 +100,7 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
numIter = Iterations[test / (NUM_SIZES * NUM_SUBTESTS)];
if (hostMalloc[0]) {
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&srcBuffer),
bufSize_, 0));
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&srcBuffer), bufSize_, 0));
setData(srcBuffer, bufSize_, 0xd0);
} else if (hostRegister[0]) {
memptr[0] = malloc(bufSize_ + 4096);
@@ -121,8 +119,7 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
}
if (hostMalloc[1]) {
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&dstBuffer),
bufSize_, 0));
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&dstBuffer), bufSize_, 0));
} else if (hostRegister[1]) {
memptr[1] = malloc(bufSize_ + 4096);
alignedmemptr[1] = reinterpret_cast<void*>(memptr[1]);
@@ -143,8 +140,7 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
auto all_start = std::chrono::steady_clock::now();
for (unsigned int i = 0; i < numIter; i++) {
HIP_CHECK(hipMemcpyAsync(dstBuffer, srcBuffer, bufSize_,
hipMemcpyDefault, NULL));
HIP_CHECK(hipMemcpyAsync(dstBuffer, srcBuffer, bufSize_, hipMemcpyDefault, NULL));
}
HIP_CHECK(hipDeviceSynchronize());
@@ -152,11 +148,11 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
std::chrono::duration<double> elapsed_secs = all_end - all_start;
// read speed in GB/s
double perf = (static_cast<double>(bufSize_ * numIter) *
static_cast<double>(1e-09)) / elapsed_secs.count();
double perf = (static_cast<double>(bufSize_ * numIter) * static_cast<double>(1e-09)) /
elapsed_secs.count();
const char *strSrc = NULL;
const char *strDst = NULL;
const char* strSrc = NULL;
const char* strDst = NULL;
if (hostMalloc[0])
strSrc = "hHM";
else if (hostRegister[0])
@@ -178,15 +174,15 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
// Double results when src and dst are both on device
if ((!hostMalloc[0] && !hostRegister[0] && !unpinnedMalloc[0]) &&
(!hostMalloc[1] && !hostRegister[1] && !unpinnedMalloc[1]))
perf *= 2.0;
perf *= 2.0;
// Double results when src and dst are both in sysmem
if ((hostMalloc[0] || hostRegister[0] || unpinnedMalloc[0]) &&
(hostMalloc[1] || hostRegister[1] || unpinnedMalloc[1]))
perf *= 2.0;
perf *= 2.0;
INFO("HIPPerfBufferCopySpeed[" << test << "]\t( " << bufSize_ <<
")\ts:" << strSrc << " d:" << strDst << "\ti:" << numIter <<
"\t(GB/s) perf\t" << (float)perf);
INFO("HIPPerfBufferCopySpeed[" << test << "]\t( " << bufSize_ << ")\ts:" << strSrc
<< " d:" << strDst << "\ti:" << numIter << "\t(GB/s) perf\t"
<< (float)perf);
// Verification
void* temp = malloc(bufSize_ + 4096);
@@ -224,40 +220,42 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
}
/**
* Test Description
* ------------------------
*  - Verify hipPerfBufferCopySpeed status.
* Test source
* ------------------------
*  - perftests/memory/hipPerfBufferCopySpeed.cc
* Test requirements
* ------------------------
*  - HIP_VERSION >= 5.6
*/
* Test Description
* ------------------------
*  - Verify hipPerfBufferCopySpeed status.
* Test source
* ------------------------
*  - perftests/memory/hipPerfBufferCopySpeed.cc
* Test requirements
* ------------------------
*  - HIP_VERSION >= 5.6
*/
TEST_CASE("Perf_hipPerfBufferCopySpeed_test") {
int numDevices = 0;
HIP_CHECK(hipGetDeviceCount(&numDevices));
if (numDevices <= 0) {
SUCCEED("Skipped testcase hipPerfBufferCopySpeed as"
"there is no device to test.");
SUCCEED(
"Skipped testcase hipPerfBufferCopySpeed as"
"there is no device to test.");
} else {
int deviceId = 0;
HIP_CHECK(hipSetDevice(deviceId));
hipDeviceProp_t props;
HIP_CHECK(hipGetDeviceProperties(&props, deviceId));
INFO("hipPerfBufferCopySpeed - info: Set device to " << deviceId
<< " : " << props.name << "Legend: unp - unpinned(malloc),"
" hM - hipMalloc(device)\n hHR - hipHostRegister(pinned),"
" hHM - hipHostMalloc(prePinned)\n");
INFO("hipPerfBufferCopySpeed - info: Set device to "
<< deviceId << " : " << props.name
<< "Legend: unp - unpinned(malloc),"
" hM - hipMalloc(device)\n hHR - hipHostRegister(pinned),"
" hHM - hipHostMalloc(prePinned)\n");
REQUIRE(true == hipPerfBufferCopySpeed_test(1));
}
}
/**
* End doxygen group perfMemoryTest.
* @}
*/
* End doxygen group perfMemoryTest.
* @}
*/
@@ -32,72 +32,70 @@ THE SOFTWARE.
#include <iostream>
#include <sstream>
#include <iomanip>
//#define VERIFY_DATA
// #define VERIFY_DATA
using namespace std;
enum DEV_MEM_TYPE { COARSE_GRAINED, FINE_GRAINED, EXTENDED_FINE_GRAINED, UNKNOWN_MEM};
enum DEV_MEM_TYPE { COARSE_GRAINED, FINE_GRAINED, EXTENDED_FINE_GRAINED, UNKNOWN_MEM };
typedef long long T; // You may change to any type
typedef long long T; // You may change to any type
static constexpr int nWarmup = 1; // warmup iteration number
static constexpr int nIters = 10; // interation number for test
static constexpr size_t dataBytes = 1024*1024*1024;
static constexpr int nWarmup = 1; // warmup iteration number
static constexpr int nIters = 10; // interation number for test
static constexpr size_t dataBytes = 1024 * 1024 * 1024;
template <typename T>
static __global__ void copy_kernel(T* dst, T* src, size_t N) {
template <typename T> static __global__ void copy_kernel(T* dst, T* src, size_t N) {
const size_t off = blockDim.x * gridDim.x;
for (size_t i = blockIdx.x * blockDim.x + threadIdx.x; i < N; i += off)
dst[i] = src[i];
for (size_t i = blockIdx.x * blockDim.x + threadIdx.x; i < N; i += off) dst[i] = src[i];
}
static string getMemType(DEV_MEM_TYPE memType) {
switch (memType) {
case COARSE_GRAINED:
return "coarse";
case FINE_GRAINED:
return "fine";
case EXTENDED_FINE_GRAINED:
// Extended - Scope Fine Grained Memory: read is cached, write is not
return "extended fine";
default:
return "unknown mem type";
case COARSE_GRAINED:
return "coarse";
case FINE_GRAINED:
return "fine";
case EXTENDED_FINE_GRAINED:
// Extended - Scope Fine Grained Memory: read is cached, write is not
return "extended fine";
default:
return "unknown mem type";
}
}
static void mallocDevBuf(void** pp, size_t size, DEV_MEM_TYPE memType) {
switch (memType) {
case COARSE_GRAINED:
HIP_CHECK(hipMalloc(pp, size));
break;
case FINE_GRAINED:
case COARSE_GRAINED:
HIP_CHECK(hipMalloc(pp, size));
break;
case FINE_GRAINED:
#if HT_AMD
HIP_CHECK(hipExtMallocWithFlags(pp, size, hipDeviceMallocFinegrained));
HIP_CHECK(hipExtMallocWithFlags(pp, size, hipDeviceMallocFinegrained));
#else
fprintf(stderr, "Unsupported memType for nvidia hardware: %d\n", memType);
REQUIRE(false);
fprintf(stderr, "Unsupported memType for nvidia hardware: %d\n", memType);
REQUIRE(false);
#endif
break;
case EXTENDED_FINE_GRAINED:
// Extended - Scope Fine Grained Memory: read is cached, write is not
// Perf gain compared with cacheable write
break;
case EXTENDED_FINE_GRAINED:
// Extended - Scope Fine Grained Memory: read is cached, write is not
// Perf gain compared with cacheable write
#if HT_AMD
HIP_CHECK(hipExtMallocWithFlags(pp, size, hipDeviceMallocUncached));
HIP_CHECK(hipExtMallocWithFlags(pp, size, hipDeviceMallocUncached));
#else
fprintf(stderr, "Unsupported memType for nvidia hardware: %d\n", memType);
REQUIRE(false);
fprintf(stderr, "Unsupported memType for nvidia hardware: %d\n", memType);
REQUIRE(false);
#endif
break;
default:
fprintf(stderr, "Unknown memType = %d\n", memType);
REQUIRE(false);
break;
break;
default:
fprintf(stderr, "Unknown memType = %d\n", memType);
REQUIRE(false);
break;
}
}
static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
DEV_MEM_TYPE srcType, DEV_MEM_TYPE dstType) {
static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu, DEV_MEM_TYPE srcType,
DEV_MEM_TYPE dstType) {
int nGpus = 0;
unsigned int threadsPerBlock = 1024;
unsigned int blocks = 16; // DEBUG_CLR_LIMIT_BLIT_WG
unsigned int blocks = 16; // DEBUG_CLR_LIMIT_BLIT_WG
HIP_CHECK(hipGetDeviceCount(&nGpus));
if (nGpus < 2) {
fprintf(stderr, "Need at least 2 GPUs, skipped!\n");
@@ -120,7 +118,7 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
#endif
char** srcBuf = reinterpret_cast<char**>(malloc(nGpus * nGpus * sizeof(char*)));
char** dstBuf = reinterpret_cast<char**>(malloc(nGpus * nGpus * sizeof(char*)));
hipStream_t *streams = (hipStream_t*)malloc(nGpus*nGpus*sizeof(hipStream_t));
hipStream_t* streams = (hipStream_t*)malloc(nGpus * nGpus * sizeof(hipStream_t));
for (int local = 0; local < nGpus; local++) {
HIP_CHECK(hipSetDevice(local));
for (int remote = 0; remote < nGpus; remote++) {
@@ -130,48 +128,49 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
HIP_CHECK(hipStreamCreateWithFlags(&streams[local * nGpus + remote], hipStreamNonBlocking));
HIP_CHECK(hipDeviceEnablePeerAccess(remote, 0));
#ifdef VERIFY_DATA
HIP_CHECK(hipMemcpy(srcBuf[local * nGpus + remote], hostMem0.data(), dataBytes, hipMemcpyHostToDevice));
HIP_CHECK(hipMemcpy(srcBuf[local * nGpus + remote], hostMem0.data(), dataBytes,
hipMemcpyHostToDevice));
#endif
}
}
unsigned N = dataBytes / sizeof(T); // Number of T in buffer of dataBytes bytes.
unsigned N = dataBytes / sizeof(T); // Number of T in buffer of dataBytes bytes.
REQUIRE(N * sizeof(T) == dataBytes);
auto test = [&](int iters) {
for (int it = 0; it < iters; it++) {
for (int local = 0; local < nGpus; local++) {
HIP_CHECK(hipSetDevice(local));
for (int i = 0; i < nGpus-1; i++) {
for (int i = 0; i < nGpus - 1; i++) {
int remote = (local + i + 1) % nGpus;
if (toRemote) {
// local to remotes
if (kernelCopy) {
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
streams[local * nGpus + remote],
reinterpret_cast<T*>(dstBuf[remote * nGpus + local]),
reinterpret_cast<T*>(srcBuf[local * nGpus + remote]),
static_cast<size_t>(N));
HIP_CHECK(hipGetLastError());
} else {
HIP_CHECK(hipMemcpyPeerAsync(dstBuf[remote * nGpus + local], remote,
srcBuf[local * nGpus + remote], local,
dataBytes, streams[local * nGpus + remote]));
}
// local to remotes
if (kernelCopy) {
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
streams[local * nGpus + remote],
reinterpret_cast<T*>(dstBuf[remote * nGpus + local]),
reinterpret_cast<T*>(srcBuf[local * nGpus + remote]),
static_cast<size_t>(N));
HIP_CHECK(hipGetLastError());
} else {
HIP_CHECK(hipMemcpyPeerAsync(dstBuf[remote * nGpus + local], remote,
srcBuf[local * nGpus + remote], local, dataBytes,
streams[local * nGpus + remote]));
}
} else {
// remotes to local
if (kernelCopy) {
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
streams[remote * nGpus + local],
reinterpret_cast<T*>(dstBuf[local * nGpus + remote]),
reinterpret_cast<T*>(srcBuf[remote * nGpus + local]),
static_cast<size_t>(N));
HIP_CHECK(hipGetLastError());
} else {
HIPCHECK(hipMemcpyPeerAsync(dstBuf[local * nGpus + remote], local,
srcBuf[remote* nGpus + local], remote,
dataBytes, streams[remote * nGpus + local]));
}
// remotes to local
if (kernelCopy) {
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
streams[remote * nGpus + local],
reinterpret_cast<T*>(dstBuf[local * nGpus + remote]),
reinterpret_cast<T*>(srcBuf[remote * nGpus + local]),
static_cast<size_t>(N));
HIP_CHECK(hipGetLastError());
} else {
HIPCHECK(hipMemcpyPeerAsync(dstBuf[local * nGpus + remote], local,
srcBuf[remote * nGpus + local], remote, dataBytes,
streams[remote * nGpus + local]));
}
}
}
if (onOneGpu) break;
@@ -197,10 +196,9 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
test(nWarmup);
auto cpuStart = std::chrono::steady_clock::now();
test(nIters);
std::chrono::duration<double, std::milli> cpuMS =
std::chrono::steady_clock::now() - cpuStart;
std::chrono::duration<double, std::milli> cpuMS = std::chrono::steady_clock::now() - cpuStart;
fprintf(stderr, "%s: Time: %f ms/iter, AvgCopyBW: %f GB/s per GPU\n", title.c_str(),
cpuMS.count()/nIters, (nGpus-1)*dataBytes/cpuMS.count()*nIters/1e6);
cpuMS.count() / nIters, (nGpus - 1) * dataBytes / cpuMS.count() * nIters / 1e6);
// exit
for (int local = 0; local < nGpus; local++) {
@@ -214,18 +212,17 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
memset(hostMem1.data(), 0, dataBytes);
if (toRemote) {
HIP_CHECK(hipMemcpy(hostMem1.data(), dstBuf[remote * nGpus + local], dataBytes,
hipMemcpyDeviceToHost));
}
else {
hipMemcpyDeviceToHost));
} else {
HIP_CHECK(hipMemcpy(hostMem1.data(), dstBuf[local * nGpus + remote], dataBytes,
hipMemcpyDeviceToHost));
hipMemcpyDeviceToHost));
}
REQUIRE(hostMem1 == hostMem0);
} else if (!onOneGpu) {
// All dstBuf will be enumed regardless of toRemote
memset(hostMem1.data(), 0, dataBytes);
HIP_CHECK(hipMemcpy(hostMem1.data(), dstBuf[local * nGpus + remote], dataBytes,
hipMemcpyDeviceToHost));
hipMemcpyDeviceToHost));
REQUIRE(hostMem1 == hostMem0);
}
#endif
@@ -245,8 +242,8 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu) {
#if HT_AMD
for (int srcType = COARSE_GRAINED; srcType < UNKNOWN_MEM; srcType++) {
for (int dstType = COARSE_GRAINED; dstType < UNKNOWN_MEM; dstType++) {
testCopyPerf(toRemote, kernelCopy, onOneGpu,
static_cast<DEV_MEM_TYPE>(srcType), static_cast<DEV_MEM_TYPE>(dstType));
testCopyPerf(toRemote, kernelCopy, onOneGpu, static_cast<DEV_MEM_TYPE>(srcType),
static_cast<DEV_MEM_TYPE>(dstType));
}
}
#else
@@ -271,7 +268,7 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu) {
* - HIP_VERSION >= 6.0
*/
TEST_CASE("Perf_PerfBufferCopySpeedAll2All_test - hipMemcpyPeerAsync - remotes to local") {
testCopyPerf(false, false, false);
testCopyPerf(false, false, false);
}
/**
@@ -414,6 +411,6 @@ TEST_CASE("Perf_PerfBufferCopySpeedOne2All_test - kernel copy - local to remotes
}
/**
* End doxygen group perfMemoryTest.
* @}
*/
* End doxygen group perfMemoryTest.
* @}
*/
@@ -18,13 +18,13 @@ THE SOFTWARE.
*/
/**
* @addtogroup hipMemcpyAsync
* @{
* @ingroup perfMemoryTest
* `hipMemcpyAsync(void* dst, const void* src, size_t count,
* hipMemcpyKind kind, hipStream_t stream = 0)` -
* Copies data between devices.
*/
* @addtogroup hipMemcpyAsync
* @{
* @ingroup perfMemoryTest
* `hipMemcpyAsync(void* dst, const void* src, size_t count,
* hipMemcpyKind kind, hipStream_t stream = 0)` -
* Copies data between devices.
*/
#include <hip_test_common.hh>
#include <hip_array_common.hh>
@@ -33,27 +33,27 @@ THE SOFTWARE.
#include <sstream>
#include <iomanip>
using namespace std;
typedef long long T; // You may change to any type
typedef long long T; // You may change to any type
//#define VERIFY_DATA
// #define VERIFY_DATA
enum TIMING_MODE { TIMING_MODE_CPU, TIMING_MODE_GPU };
// -sizes are in bytes, +sizes are in kb, last size must be largest
#if 1
static constexpr int sizes[] = {-64, -256, -512, 1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024,
2048, 4096, 8192, 16384, 32768, 65536, 131072, 262144};
static constexpr int sizes[] = {-64, -256, -512, 1, 2, 4, 8, 16,
32, 64, 128, 256, 512, 1024, 2048, 4096,
8192, 16384, 32768, 65536, 131072, 262144};
#else
static constexpr int sizes[] = { 262144 };
static constexpr int sizes[] = {262144};
#endif
static constexpr int nSizes = sizeof(sizes) / sizeof(sizes[0]);
static constexpr double megaSize = 1000000.;
static constexpr int defaultIterations = 200;
static constexpr unsigned threadsPerBlock = 1024;
template <typename T>
static __global__ void copy_kernel(T* dst, T* src, size_t N) {
template <typename T> static __global__ void copy_kernel(T* dst, T* src, size_t N) {
size_t idx = blockIdx.x * blockDim.x + threadIdx.x;
if(idx < N) dst[idx] = src[idx]; // We make sure idx < N
if (idx < N) dst[idx] = src[idx]; // We make sure idx < N
}
static size_t sizeToBytes(int size) { return (size < 0) ? -size : size * 1024; }
@@ -68,7 +68,7 @@ static string sizeToString(int size) {
return ss.str();
}
static void checkP2PSupport(){
static void checkP2PSupport() {
int deviceCnt = 0;
HIP_CHECK(hipGetDeviceCount(&deviceCnt));
cout << "Total no. of available gpu #" << deviceCnt << "\n" << endl;
@@ -87,8 +87,7 @@ static void checkP2PSupport(){
++PeerCnt;
}
}
if (PeerCnt == 0)
cout << "NONE" << " ";
if (PeerCnt == 0) cout << "NONE" << " ";
cout << std::endl;
cout << " peer2peer not supported : ";
@@ -101,21 +100,20 @@ static void checkP2PSupport(){
++nonPeerCnt;
}
}
if (nonPeerCnt == 0)
cout << "NONE" << " ";
if (nonPeerCnt == 0) cout << "NONE" << " ";
cout << "\n" << endl;
}
cout << "\nNote: For non-supported peer2peer devices, memcopy will use/follow the normal "
"behaviour (GPU1-->host then host-->GPU2)\n\n"
<< endl;
"behaviour (GPU1-->host then host-->GPU2)\n\n"
<< endl;
}
static void outputMatrix(const string &title, const int numGPUs, const vector<double> &data,
const TIMING_MODE mode, const size_t dataSize, const int iterations) {
fprintf(stderr, "%s, Timing %s, Data %zu KB, Iterations %d\n ",
title.c_str(), mode == TIMING_MODE_GPU ? "GPU" : "CPU", dataSize, iterations);
static void outputMatrix(const string& title, const int numGPUs, const vector<double>& data,
const TIMING_MODE mode, const size_t dataSize, const int iterations) {
fprintf(stderr, "%s, Timing %s, Data %zu KB, Iterations %d\n ", title.c_str(),
mode == TIMING_MODE_GPU ? "GPU" : "CPU", dataSize, iterations);
for (int j = 0; j < numGPUs; j++) {
fprintf(stderr, "%9d ", j);
}
@@ -130,7 +128,7 @@ static void outputMatrix(const string &title, const int numGPUs, const vector<do
}
static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingMode,
const bool useHipMemcpyAsync) {
const bool useHipMemcpyAsync) {
const char* method = useHipMemcpyAsync ? "hipMemcpyAsync()" : "copy kernel";
int gpuCount = 0;
HIP_CHECK(hipGetDeviceCount(&gpuCount));
@@ -150,8 +148,8 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
for (int peerGpu = 0; peerGpu < gpuCount; peerGpu++) {
HIP_CHECK(hipSetDevice(currentGpu));
fprintf(stderr, "Uni: Gpu%d -> Gpu%d by %s, Timing %s, Iterations %d\n",
currentGpu, peerGpu, method, timingMode == TIMING_MODE_GPU ? "GPU" : "CPU", iterations);
fprintf(stderr, "Uni: Gpu%d -> Gpu%d by %s, Timing %s, Iterations %d\n", currentGpu, peerGpu,
method, timingMode == TIMING_MODE_GPU ? "GPU" : "CPU", iterations);
if (currentGpu != peerGpu) {
int canAccessPeer = 0;
@@ -173,18 +171,16 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
#ifdef VERIFY_DATA
HIP_CHECK(hipMemcpy(currentGpuMem, hostMem0.data(), numMax, hipMemcpyHostToDevice));
#endif
unsigned N = numMax / sizeof(T); // Number of T in buffer of numMax bytes.
REQUIRE(N * sizeof(T) == numMax); // To prevent verification failure
unsigned N = numMax / sizeof(T); // Number of T in buffer of numMax bytes.
REQUIRE(N * sizeof(T) == numMax); // To prevent verification failure
unsigned blocks = (N + threadsPerBlock - 1) / threadsPerBlock;
// Warmup
if (useHipMemcpyAsync) {
HIP_CHECK(hipMemcpyAsync(peerGpuMem, currentGpuMem, numMax,
hipMemcpyDeviceToDevice, 0));
}
else {
HIP_CHECK(hipMemcpyAsync(peerGpuMem, currentGpuMem, numMax, hipMemcpyDeviceToDevice, 0));
} else {
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, 0,
reinterpret_cast<T*>(peerGpuMem), reinterpret_cast<T*>(currentGpuMem),
static_cast<size_t>(N));
reinterpret_cast<T*>(peerGpuMem), reinterpret_cast<T*>(currentGpuMem),
static_cast<size_t>(N));
HIP_CHECK(hipGetLastError());
}
HIP_CHECK(hipDeviceSynchronize());
@@ -193,7 +189,7 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
HIP_CHECK(hipMemcpy(hostMem1.data(), peerGpuMem, numMax, hipMemcpyDeviceToHost));
REQUIRE(hostMem1 == hostMem0);
#endif
float t = 0; // in ms
float t = 0; // in ms
auto cpuStart = std::chrono::steady_clock::now();
hipEvent_t eventStart, eventStop;
hipStream_t stream;
@@ -216,17 +212,17 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
HIP_CHECK(hipEventRecord(eventStart, stream));
}
for (size_t offsetEnd = numMax - nbytes, offset = 0, j = 0;
j < iterations; j++, offset += nbytes) {
for (size_t offsetEnd = numMax - nbytes, offset = 0, j = 0; j < iterations;
j++, offset += nbytes) {
if (offset > offsetEnd) offset = 0;
if (useHipMemcpyAsync) {
HIP_CHECK(hipMemcpyAsync(peerGpuMem + offset,
currentGpuMem + offset, nbytes, hipMemcpyDeviceToDevice, stream));
}
else {
HIP_CHECK(hipMemcpyAsync(peerGpuMem + offset, currentGpuMem + offset, nbytes,
hipMemcpyDeviceToDevice, stream));
} else {
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, stream,
reinterpret_cast<T*>(peerGpuMem + offset),
reinterpret_cast<T*>(currentGpuMem + offset), static_cast<size_t>(N));
reinterpret_cast<T*>(peerGpuMem + offset),
reinterpret_cast<T*>(currentGpuMem + offset),
static_cast<size_t>(N));
HIP_CHECK(hipGetLastError());
}
}
@@ -237,14 +233,14 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
} else if (timingMode == TIMING_MODE_CPU) {
HIP_CHECK(hipDeviceSynchronize());
std::chrono::duration<double, std::milli> cpuMs =
std::chrono::steady_clock::now() - cpuStart;
std::chrono::steady_clock::now() - cpuStart;
t = cpuMs.count();
}
t /= iterations;
double bandwidth = nbytes / megaSize / t; // GByte/s
double bandwidth = nbytes / megaSize / t; // GByte/s
fprintf(stderr, "%8s, %-9.06lf, %-4.08lf\n",
sizeToString(thisSize).c_str(), t, bandwidth);
fprintf(stderr, "%8s, %-9.06lf, %-4.08lf\n", sizeToString(thisSize).c_str(),
t, bandwidth);
if (i == (nSizes - 1)) {
timeMs[currentGpu * gpuCount + peerGpu] = t;
bandWidth[currentGpu * gpuCount + peerGpu] = bandwidth;
@@ -269,10 +265,10 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
HIP_CHECK(hipStreamDestroy(stream));
}
}
outputMatrix(string("Unidirectional ") + method + " Time Table(ms)", gpuCount, timeMs,
timingMode, sizes[nSizes - 1], iterations);
outputMatrix(string("Unidirectional ") + method + " Bandwith Table(GB/s)", gpuCount,
bandWidth, timingMode, sizes[nSizes - 1], iterations);
outputMatrix(string("Unidirectional ") + method + " Time Table(ms)", gpuCount, timeMs, timingMode,
sizes[nSizes - 1], iterations);
outputMatrix(string("Unidirectional ") + method + " Bandwith Table(GB/s)", gpuCount, bandWidth,
timingMode, sizes[nSizes - 1], iterations);
}
static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsync) {
@@ -289,8 +285,8 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
for (int currentGpu = 0; currentGpu < gpuCount; currentGpu++) {
for (int peerGpu = 0; peerGpu < gpuCount; peerGpu++) {
HIP_CHECK(hipSetDevice(currentGpu));
fprintf(stderr, "Bi: Gpu%d <-> Gpu%d by %s, Timing GPU, Iterations %d\n",
currentGpu, peerGpu, method, iterations);
fprintf(stderr, "Bi: Gpu%d <-> Gpu%d by %s, Timing GPU, Iterations %d\n", currentGpu, peerGpu,
method, iterations);
if (currentGpu != peerGpu) {
int canAccessPeer = 0;
@@ -310,10 +306,13 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
HIP_CHECK(hipSetDevice(currentGpu));
HIP_CHECK(hipDeviceEnablePeerAccess(peerGpu, 0));
}
fprintf(stderr, "Gpu%d -> Gpu%d *"
"* Gpu%d -> Gpu%d\n", currentGpu, peerGpu, peerGpu, currentGpu);
fprintf(stderr, "Size(KB) Time(ms) Bandwidth(GB/s) "
" Time(ms) Bandwidth(GB/s)\n");
fprintf(stderr,
"Gpu%d -> Gpu%d *"
"* Gpu%d -> Gpu%d\n",
currentGpu, peerGpu, peerGpu, currentGpu);
fprintf(stderr,
"Size(KB) Time(ms) Bandwidth(GB/s) "
" Time(ms) Bandwidth(GB/s)\n");
unsigned char *currentGpuMem[2], *peerGpuMem[2];
HIP_CHECK(hipMalloc((void**)&currentGpuMem[0], numMax));
@@ -326,30 +325,31 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
HIP_CHECK(hipSetDevice(currentGpu));
unsigned N = numMax / sizeof(T); // Number of T in buffer of numMax bytes.
unsigned N = numMax / sizeof(T); // Number of T in buffer of numMax bytes.
REQUIRE(N * sizeof(T) == numMax);
unsigned blocks = (N + threadsPerBlock - 1) / threadsPerBlock;
// Warmup. currentGpu is the current device
if (useHipMemcpyAsync) {
HIP_CHECK(hipMemcpyAsync(peerGpuMem[0], currentGpuMem[0], numMax, hipMemcpyDeviceToDevice, 0));
HIP_CHECK(
hipMemcpyAsync(peerGpuMem[0], currentGpuMem[0], numMax, hipMemcpyDeviceToDevice, 0));
HIP_CHECK(hipDeviceSynchronize());
HIP_CHECK(hipSetDevice(peerGpu));
HIP_CHECK(hipMemcpyAsync(currentGpuMem[1], peerGpuMem[1], numMax, hipMemcpyDeviceToDevice, 0));
HIP_CHECK(
hipMemcpyAsync(currentGpuMem[1], peerGpuMem[1], numMax, hipMemcpyDeviceToDevice, 0));
HIP_CHECK(hipDeviceSynchronize());
}
else {
} else {
// Warmup
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
0, reinterpret_cast<T*>(peerGpuMem[0]),
reinterpret_cast<T*>(currentGpuMem[0]), static_cast<size_t>(N));
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, 0,
reinterpret_cast<T*>(peerGpuMem[0]),
reinterpret_cast<T*>(currentGpuMem[0]), static_cast<size_t>(N));
HIP_CHECK(hipGetLastError());
HIP_CHECK(hipDeviceSynchronize());
HIP_CHECK(hipSetDevice(peerGpu));
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
0, reinterpret_cast<T*>(currentGpuMem[1]),
reinterpret_cast<T*>(peerGpuMem[1]), static_cast<size_t>(N));
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, 0,
reinterpret_cast<T*>(currentGpuMem[1]),
reinterpret_cast<T*>(peerGpuMem[1]), static_cast<size_t>(N));
HIP_CHECK(hipGetLastError());
HIP_CHECK(hipDeviceSynchronize());
}
@@ -377,21 +377,22 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
HIP_CHECK(hipEventRecord(eventStart[1], stream[1]));
for (size_t offsetEnd = numMax - nbytes, offset = 0, j = 0; j < iterations;
j++, offset += nbytes) {
j++, offset += nbytes) {
if (offset > offsetEnd) offset = 0;
if (useHipMemcpyAsync) {
HIP_CHECK(hipMemcpyAsync(peerGpuMem[0] + offset, currentGpuMem[0] + offset, nbytes,
hipMemcpyDeviceToDevice, stream[0]));
hipMemcpyDeviceToDevice, stream[0]));
HIP_CHECK(hipMemcpyAsync(currentGpuMem[1] + offset, peerGpuMem[1] + offset, nbytes,
hipMemcpyDeviceToDevice, stream[1]));
}
else {
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
stream[0], reinterpret_cast<T*>(peerGpuMem[0] + offset),
reinterpret_cast<T*>(currentGpuMem[0] + offset), static_cast<size_t>(N));
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
stream[1], reinterpret_cast<T*>(currentGpuMem[1] + offset),
reinterpret_cast<T*>(peerGpuMem[1] + offset), static_cast<size_t>(N));
hipMemcpyDeviceToDevice, stream[1]));
} else {
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, stream[0],
reinterpret_cast<T*>(peerGpuMem[0] + offset),
reinterpret_cast<T*>(currentGpuMem[0] + offset),
static_cast<size_t>(N));
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, stream[1],
reinterpret_cast<T*>(currentGpuMem[1] + offset),
reinterpret_cast<T*>(peerGpuMem[1] + offset),
static_cast<size_t>(N));
}
}
@@ -406,12 +407,13 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
for (int n = 0; n < 2; n++) {
HIP_CHECK(hipEventElapsedTime(&t[n], eventStart[n], eventStop[n]));
t[n] /= iterations;
bandwidth[n] = nbytes / megaSize / t[n]; // GByte/s
bandwidth[n] = nbytes / megaSize / t[n]; // GByte/s
}
fprintf(stderr, "%8s, %-9.06lf, %-4.08lf, "
"%-9.06lf, %-4.08lf\n", sizeToString(thisSize).c_str(), t[0],
bandwidth[0], t[1], bandwidth[1]);
fprintf(stderr,
"%8s, %-9.06lf, %-4.08lf, "
"%-9.06lf, %-4.08lf\n",
sizeToString(thisSize).c_str(), t[0], bandwidth[0], t[1], bandwidth[1]);
if (i == (nSizes - 1)) {
timeMs[currentGpu * gpuCount + peerGpu] = (t[0] + t[1]) / 2;
@@ -434,9 +436,9 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
}
}
outputMatrix(string("Bidirectional ") + method + " Time Table(ms)", gpuCount, timeMs,
TIMING_MODE_GPU, sizes[nSizes - 1], iterations);
outputMatrix(string("Bidirectional ") + method + " Bandwith Table(GB/s)", gpuCount,
bandWidth, TIMING_MODE_GPU, sizes[nSizes - 1], iterations);
TIMING_MODE_GPU, sizes[nSizes - 1], iterations);
outputMatrix(string("Bidirectional ") + method + " Bandwith Table(GB/s)", gpuCount, bandWidth,
TIMING_MODE_GPU, sizes[nSizes - 1], iterations);
}
/**
@@ -455,14 +457,14 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
*  - HIP_VERSION >= 6.0
*/
TEST_CASE("Perf_hipTestP2PUniDirMemcpyAsync_test - Timing CPU") {
const int iterations = cmd_options.iterations == 1000 ?
defaultIterations : cmd_options.iterations;
const int iterations =
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
testP2PUniDirMemPerf(iterations, TIMING_MODE_CPU, true);
}
TEST_CASE("Perf_hipTestP2PUniDirMemcpyAsync_test - Timing GPU") {
const int iterations = cmd_options.iterations == 1000 ?
defaultIterations : cmd_options.iterations;
const int iterations =
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
testP2PUniDirMemPerf(iterations, TIMING_MODE_GPU, true);
}
@@ -480,14 +482,14 @@ TEST_CASE("Perf_hipTestP2PUniDirMemcpyAsync_test - Timing GPU") {
*  - HIP_VERSION >= 6.0
*/
TEST_CASE("Perf_hipTestP2PUniDirKernelCopy_test - Timing CPU") {
const int iterations = cmd_options.iterations == 1000 ?
defaultIterations : cmd_options.iterations;
const int iterations =
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
testP2PUniDirMemPerf(iterations, TIMING_MODE_CPU, false);
}
TEST_CASE("Perf_hipTestP2PUniDirKernelCopy_test - Timing GPU") {
const int iterations = cmd_options.iterations == 1000 ?
defaultIterations : cmd_options.iterations;
const int iterations =
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
testP2PUniDirMemPerf(iterations, TIMING_MODE_GPU, false);
}
@@ -507,8 +509,8 @@ TEST_CASE("Perf_hipTestP2PUniDirKernelCopy_test - Timing GPU") {
*  - HIP_VERSION >= 6.0
*/
TEST_CASE("Perf_hipTestP2PBiDirMemcpyAsync_test") {
const int iterations = cmd_options.iterations == 1000 ?
defaultIterations : cmd_options.iterations;
const int iterations =
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
testP2PBiDirMemPerf(iterations, true);
}
@@ -526,9 +528,9 @@ TEST_CASE("Perf_hipTestP2PBiDirMemcpyAsync_test") {
*  - HIP_VERSION >= 6.0
*/
TEST_CASE("Perf_hipTestP2PBiDirKernelCopy_test") {
const int iterations = cmd_options.iterations == 1000 ?
defaultIterations : cmd_options.iterations;
testP2PBiDirMemPerf(iterations, false);
const int iterations =
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
testP2PBiDirMemPerf(iterations, false);
}
/**
@@ -542,11 +544,9 @@ TEST_CASE("Perf_hipTestP2PBiDirKernelCopy_test") {
* ------------------------
*  - HIP_VERSION >= 6.0
*/
TEST_CASE("Perf_hipCheckP2PSupport") {
checkP2PSupport();
}
TEST_CASE("Perf_hipCheckP2PSupport") { checkP2PSupport(); }
/**
* End doxygen group perfMemoryTest.
* @}
*/
* End doxygen group perfMemoryTest.
* @}
*/
@@ -25,12 +25,11 @@ __global__ void mallocTest() {
memset(ptr, 0, size);
free(ptr);
}
__global__ void mallocTest_1()
{
size_t size = 1024;
int* ptr = (int*)malloc(size);
memset(ptr, 0, size);
free(ptr);
__global__ void mallocTest_1() {
size_t size = 1024;
int* ptr = (int*)malloc(size);
memset(ptr, 0, size);
free(ptr);
}
/**
* The tests in this file are added to see the performance improvement with the
@@ -64,7 +63,7 @@ __global__ void mallocTest_1()
* - HIP_VERSION >= 6.5
*/
TEST_CASE("Unit_Perf_Device_Heap_Memory_Allocation") {
HIP_CHECK(hipDeviceSetLimit(hipLimitMallocHeapSize, 128*1024*1024));
HIP_CHECK(hipDeviceSetLimit(hipLimitMallocHeapSize, 128 * 1024 * 1024));
hipEvent_t event;
HIP_CHECK(hipEventCreate(&event));
REQUIRE(event != nullptr);
@@ -87,8 +86,8 @@ TEST_CASE("Unit_Perf_Device_Heap_Memory_Allocation") {
REQUIRE(time > time_1);
HIP_CHECK(hipEventDestroy(event));
HIP_CHECK(hipStreamDestroy(stream));
std::cout<<"First Kernel Latency: "<<time<<" micro seconds"<<std::endl;
std::cout<<"Second Kernel Latency: "<<time_1<<" micro seconds"<<std::endl;
std::cout << "First Kernel Latency: " << time << " micro seconds" << std::endl;
std::cout << "Second Kernel Latency: " << time_1 << " micro seconds" << std::endl;
}
/**
* End doxygen group PerformanceTest.
@@ -29,14 +29,12 @@ THE SOFTWARE.
* Helper function to get and print the Total device and free device memory,
* and reserved current, used current memory from pool.
*/
void getAndPrintMemoryDetails(const hipMemPool_t &pool) {
void getAndPrintMemoryDetails(const hipMemPool_t& pool) {
size_t freeVRAM = 0, totalVRAM = 0, reservedCurrent = 0, usedCurrent = 0;
HIP_CHECK(hipMemGetInfo(&freeVRAM, &totalVRAM));
HIP_CHECK(hipMemPoolGetAttribute(pool, hipMemPoolAttrReservedMemCurrent,
&reservedCurrent));
HIP_CHECK(hipMemPoolGetAttribute(pool, hipMemPoolAttrUsedMemCurrent,
&usedCurrent));
HIP_CHECK(hipMemPoolGetAttribute(pool, hipMemPoolAttrReservedMemCurrent, &reservedCurrent));
HIP_CHECK(hipMemPoolGetAttribute(pool, hipMemPoolAttrUsedMemCurrent, &usedCurrent));
std::cout << "\n Total device memory (GB) : " << totalVRAM / 1_GB;
std::cout << "\n Free device memory (GB) : " << freeVRAM / 1_GB;
@@ -80,8 +78,7 @@ TEST_CASE("Perf_MempoolManager_hipMallocAsync_hipFreeAsync") {
HIP_CHECK(hipDeviceGetDefaultMemPool(&pool, device));
uint64_t threshold = 30_GB;
HIP_CHECK(hipMemPoolSetAttribute(pool, hipMemPoolAttrReleaseThreshold,
&threshold));
HIP_CHECK(hipMemPoolSetAttribute(pool, hipMemPoolAttrReleaseThreshold, &threshold));
std::cout << "\n Memory details at start : ";
getAndPrintMemoryDetails(pool);
@@ -90,7 +87,7 @@ TEST_CASE("Perf_MempoolManager_hipMallocAsync_hipFreeAsync") {
HIP_CHECK(hipStreamCreate(&stream));
constexpr int ptrs = 20;
void *dPtr[ptrs];
void* dPtr[ptrs];
for (int i = 0; i < ptrs; i++) {
dPtr[i] = nullptr;
}
@@ -23,9 +23,7 @@
using namespace std;
__global__
static void _noop_kernel() {
}
__global__ static void _noop_kernel() {}
TEST_CASE("Perf_KernelLaunchLatency_IncreasingNumberOfStreams") {
@@ -67,12 +65,11 @@ TEST_CASE("Perf_KernelLaunchLatency_IncreasingNumberOfStreams") {
for (auto& numHipStreams : streamsNumber) {
vector<hipStream_t> streams(numHipStreams);
if(isBlocking) {
if (isBlocking) {
for (int i = 0; i < numHipStreams; ++i) {
HIP_CHECK(hipStreamCreate(&streams[i]));
}
}
else {
} else {
for (int i = 0; i < numHipStreams; ++i) {
HIP_CHECK(hipStreamCreateWithFlags(&streams[i], hipStreamNonBlocking));
}
@@ -48,7 +48,7 @@ class hipPerfStreamCreateCopyDestroy {
numStreams_(0),
totalStreams_{1, 2, 4, 8},
totalBuffers_{1, 100, 1000, 5000} {};
~hipPerfStreamCreateCopyDestroy(){};
~hipPerfStreamCreateCopyDestroy() {};
bool open(int deviceID);
bool run(unsigned int testNumber);
};
@@ -89,20 +89,19 @@ bool ValidateUsingCopy(int deviceId, void* dev_ptr, size_t data_size,
HIP_CHECK(hipMemcpy(dev_ptr, A_h.data(), data_size, hipMemcpyHostToDevice));
auto end = std::chrono::high_resolution_clock::now();
h2d_elapsed = std::chrono::duration_cast<std::chrono::microseconds>(end - start);
start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipMemcpy(B_h.data(), dev_ptr, data_size, hipMemcpyDeviceToHost));
end = std::chrono::high_resolution_clock::now();
d2h_elapsed = std::chrono::duration_cast<std::chrono::microseconds>(end - start);
if (debug_failure) {
REQUIRE(true == std::equal(B_h.begin(), B_h.end(), A_h.data()));
} else {
assert(A_h.size() == B_h.size());
for (size_t idx = 0; idx < A_h.size(); ++idx) {
if (A_h[idx] != B_h[idx]) {
std::cout << "Failed at first index: " << idx
<< " Expected: " << A_h[idx]
std::cout << "Failed at first index: " << idx << " Expected: " << A_h[idx]
<< " Value: " << B_h[idx] << std::endl;
break;
}
@@ -135,8 +134,8 @@ bool TestOnDevice(int deviceId) {
auto start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipMemAddressReserve(&dev_ptr, size_idx, granularity, nullptr, 0));
auto end = std::chrono::high_resolution_clock::now();
std::chrono::microseconds reserve_elapsed
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
std::chrono::microseconds reserve_elapsed =
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
std::vector<hipMemGenericAllocationHandle_t> physmem_handles;
std::chrono::microseconds alloc_elapsed;
std::chrono::microseconds map_elapsed;
@@ -233,34 +232,34 @@ bool TestOnDevice(int deviceId) {
++chunk_idx;
}
end = std::chrono::high_resolution_clock::now();
std::chrono::microseconds unmap_elapsed
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
std::chrono::microseconds unmap_elapsed =
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
start = std::chrono::high_resolution_clock::now();
for (auto& physmem_handle : physmem_handles) {
HIP_CHECK(hipMemRelease(physmem_handle));
}
end = std::chrono::high_resolution_clock::now();
std::chrono::microseconds release_elapsed
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
std::chrono::microseconds release_elapsed =
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipMemAddressFree(dev_ptr, size_idx));
end = std::chrono::high_resolution_clock::now();
std::chrono::microseconds free_elapsed
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
std::chrono::microseconds free_elapsed =
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
// Print the results
std::cout << "-------------Size: " << (size_idx / kGB) << " GB----------------" << std::endl;
std::cout << "Time taken to reserve : " << reserve_elapsed.count()
<< " micro seconds and free: " << free_elapsed.count()
<< " micro seconds" << std::endl;
std::cout <<"Time taken to alloc : " << alloc_elapsed.count()
<< " micro seconds and release: "<< release_elapsed.count()
<< " micro seconds" << std::endl;
<< " micro seconds and free: " << free_elapsed.count() << " micro seconds"
<< std::endl;
std::cout << "Time taken to alloc : " << alloc_elapsed.count()
<< " micro seconds and release: " << release_elapsed.count() << " micro seconds"
<< std::endl;
std::cout << "Time taken to map : " << map_elapsed.count()
<< " micro seconds and unmap: " << unmap_elapsed.count()
<< " micro seconds" << std::endl;
<< " micro seconds and unmap: " << unmap_elapsed.count() << " micro seconds"
<< std::endl;
std::cout << "Time taken to H2D : " << h2d_elapsed.count()
<< " micro seconds and D2H: " << d2h_elapsed.count() << " micro seconds" << std::endl;
std::cout << "-------------------------/hipMallocPerf------------------------" << std::endl;
@@ -269,20 +268,20 @@ bool TestOnDevice(int deviceId) {
start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipMalloc(&dev_ptr_legacy, size_idx));
end = std::chrono::high_resolution_clock::now();
std::chrono::microseconds hm_elapsed
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
std::chrono::microseconds hm_elapsed =
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
start = std::chrono::high_resolution_clock::now();
HIP_CHECK(hipFree(dev_ptr_legacy));
end = std::chrono::high_resolution_clock::now();
std::chrono::microseconds hf_elapsed
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
std::chrono::microseconds hf_elapsed =
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
std::cout << "Time taken for hipMalloc : " << hm_elapsed.count()
<< " micro seconds and hipFree: " << hf_elapsed.count()
<< " micro seconds" << std::endl;
<< " micro seconds and hipFree: " << hf_elapsed.count() << " micro seconds"
<< std::endl;
std::cout << "---------------------------------------------------------------" << std::endl;
std::cout << std::endl;
}
return true;
}
@@ -21,9 +21,9 @@ THE SOFTWARE.
#include <hip_test_checkers.hh>
#include <unistd.h>
// Size Macros
#define MEMORY_CHUNK_SIZE (1024*1024)
#define MEMORY_CHUNK_SIZE_ODD (1025*1025)
#define MAXIMUM_CHUNKS (256*1024)
#define MEMORY_CHUNK_SIZE (1024 * 1024)
#define MEMORY_CHUNK_SIZE_ODD (1025 * 1025)
#define MAXIMUM_CHUNKS (256 * 1024)
// Subtest Macros
#define NO_ALLOCATION_ONHOST 0
#define ALLOCATE_ONHOST_HIPMALLOCMANAGED 1
@@ -57,14 +57,12 @@ __device__ static int* dev_common_ptr;
* This kernel checks kernel allocation of size more than available
* memory.
*/
static __global__ void kerTestDynamicAllocNeg(int test_type,
size_t perThreadSize,
int *ret) {
static __global__ void kerTestDynamicAllocNeg(int test_type, size_t perThreadSize, int* ret) {
// Allocate
char* ptr = nullptr;
printf("Memory to allocate in GPU = %zu \n", perThreadSize);
if (test_type == TEST_MALLOC_FREE) {
ptr = reinterpret_cast<char*> (malloc(perThreadSize));
ptr = reinterpret_cast<char*>(malloc(perThreadSize));
} else {
ptr = new char[perThreadSize];
}
@@ -87,9 +85,8 @@ static __global__ void kerTestDynamicAllocNeg(int test_type,
/**
* This kernel allocates memory till nullptr is returned.
*/
static __global__ void kerAllocTillExhaust(int test_type,
size_t *total_allocated_mem,
size_t mem_chunk_size) {
static __global__ void kerAllocTillExhaust(int test_type, size_t* total_allocated_mem,
size_t mem_chunk_size) {
int myId = threadIdx.x + blockDim.x * blockIdx.x;
// Allocate memory in thread 0 of block 0
if (0 == myId) {
@@ -99,16 +96,14 @@ static __global__ void kerAllocTillExhaust(int test_type,
int idx = 0;
if (test_type == TEST_MALLOC_FREE) {
do {
dev_mem_glob[idx] =
reinterpret_cast<char*> (malloc(mem_chunk_size));
dev_mem_glob[idx] = reinterpret_cast<char*>(malloc(mem_chunk_size));
if (idx >= MAXIMUM_CHUNKS) {
break;
}
} while (dev_mem_glob[idx++] != nullptr);
} else {
do {
dev_mem_glob[idx] =
reinterpret_cast<char*> (new char[mem_chunk_size]);
dev_mem_glob[idx] = reinterpret_cast<char*>(new char[mem_chunk_size]);
if (idx >= MAXIMUM_CHUNKS) {
break;
}
@@ -116,8 +111,7 @@ static __global__ void kerAllocTillExhaust(int test_type,
}
idx = 0;
*total_allocated_mem = 0;
while ((dev_mem_glob[idx] != nullptr) &&
(idx < MAXIMUM_CHUNKS)) {
while ((dev_mem_glob[idx] != nullptr) && (idx < MAXIMUM_CHUNKS)) {
*total_allocated_mem = *total_allocated_mem + mem_chunk_size;
idx++;
}
@@ -155,18 +149,15 @@ static __global__ void kerFreeAll(int test_type) {
* access this memory in all threads of the block. The memory is
* finally deleted in last thread of each block.
*/
static __global__ void kerBlockLevelMemoryAllocation(int *outputBuf,
int test_type) {
static __global__ void kerBlockLevelMemoryAllocation(int* outputBuf, int test_type) {
int myThreadId = threadIdx.x, lastThreadId = (blockDim.x - 1);
int myId = threadIdx.x + blockDim.x * blockIdx.x;
// Allocate memory in thread 0
if (0 == myThreadId) {
if (test_type == TEST_MALLOC_FREE) {
dev_mem[blockIdx.x] =
reinterpret_cast<int*> (malloc(blockDim.x*sizeof(int)));
dev_mem[blockIdx.x] = reinterpret_cast<int*>(malloc(blockDim.x * sizeof(int)));
} else {
dev_mem[blockIdx.x] =
reinterpret_cast<int*> (new int[blockDim.x]);
dev_mem[blockIdx.x] = reinterpret_cast<int*>(new int[blockDim.x]);
}
}
// All threads wait at this barrier
@@ -176,7 +167,7 @@ static __global__ void kerBlockLevelMemoryAllocation(int *outputBuf,
printf("Device Allocation Failed in thread = %d \n", myId);
return;
}
int *ptr = reinterpret_cast<int*> (dev_mem[blockIdx.x]);
int* ptr = reinterpret_cast<int*>(dev_mem[blockIdx.x]);
// Copy to buffer
ptr[myThreadId] = myId;
// All threads wait
@@ -202,11 +193,9 @@ static __global__ void kerAlloc(int test_type) {
// Allocate memory in thread 0 of block 0
if (0 == myId) {
if (test_type == TEST_MALLOC_FREE) {
dev_common_ptr =
reinterpret_cast<int*> (malloc(blockDim.x*gridDim.x*sizeof(int)));
dev_common_ptr = reinterpret_cast<int*>(malloc(blockDim.x * gridDim.x * sizeof(int)));
} else {
dev_common_ptr =
reinterpret_cast<int*> (new int[blockDim.x*gridDim.x]);
dev_common_ptr = reinterpret_cast<int*>(new int[blockDim.x * gridDim.x]);
}
}
}
@@ -229,7 +218,7 @@ static __global__ void kerWrite() {
* This kernel copies the contents of memory allocated in <kerAlloc>
* to host and deletes the memory from thread 0.
*/
static __global__ void kerFree(int *outputBuf, int test_type) {
static __global__ void kerFree(int* outputBuf, int test_type) {
int myId = threadIdx.x + blockDim.x * blockIdx.x;
// Check allocated memory in all threads in block before access
if (dev_common_ptr == nullptr) {
@@ -237,7 +226,7 @@ static __global__ void kerFree(int *outputBuf, int test_type) {
return;
}
if (0 == myId) {
for (size_t idx = 0; idx < (blockDim.x*gridDim.x); idx++) {
for (size_t idx = 0; idx < (blockDim.x * gridDim.x); idx++) {
outputBuf[idx] = dev_common_ptr[idx];
}
if (test_type == TEST_MALLOC_FREE) {
@@ -254,8 +243,7 @@ static __global__ void kerFree(int *outputBuf, int test_type) {
* kerFreeAll<<<>>> to test memory allocation till all device
* memory is exhausted.
*/
static bool TestAllocationOfAllAvailableMemory(int test_type,
int category, size_t mem_chunk_size) {
static bool TestAllocationOfAllAvailableMemory(int test_type, int category, size_t mem_chunk_size) {
size_t avail1 = 0, avail2 = 0, tot = 0;
constexpr size_t host_alloc = 2147483648; // 2 GB
HIP_CHECK(hipMemGetInfo(&avail1, &tot));
@@ -263,12 +251,11 @@ static bool TestAllocationOfAllAvailableMemory(int test_type,
HIP_CHECK(hipDeviceSetLimit(hipLimitMallocHeapSize, avail1));
#endif
size_t *tot_alloc_mem_d = nullptr, *tot_alloc_mem_h = nullptr;
tot_alloc_mem_h =
reinterpret_cast<size_t*> (malloc(sizeof(size_t)));
tot_alloc_mem_h = reinterpret_cast<size_t*>(malloc(sizeof(size_t)));
REQUIRE(nullptr != tot_alloc_mem_h);
HIP_CHECK(hipMalloc(&tot_alloc_mem_d, sizeof(size_t)));
REQUIRE(nullptr != tot_alloc_mem_d);
char *devptrHost = nullptr;
char* devptrHost = nullptr;
if (category == ALLOCATE_ONHOST_HIPMALLOCMANAGED) {
HIP_CHECK(hipMallocManaged(&devptrHost, host_alloc));
} else if (category == ALLOCATE_ONHOST_HIPMALLOC) {
@@ -278,12 +265,10 @@ static bool TestAllocationOfAllAvailableMemory(int test_type,
INFO("Total available memory " << tot);
INFO("Available memory before allocation " << avail1);
// Launch Test Kernel
kerAllocTillExhaust<<<1, 1>>>(test_type, tot_alloc_mem_d,
mem_chunk_size);
kerAllocTillExhaust<<<1, 1>>>(test_type, tot_alloc_mem_d, mem_chunk_size);
HIP_CHECK(hipDeviceSynchronize());
// Copy to host buffer
HIP_CHECK(hipMemcpy(tot_alloc_mem_h, tot_alloc_mem_d,
sizeof(size_t), hipMemcpyDefault));
HIP_CHECK(hipMemcpy(tot_alloc_mem_h, tot_alloc_mem_d, sizeof(size_t), hipMemcpyDefault));
HIP_CHECK(hipMemGetInfo(&avail2, &tot));
kerFreeAll<<<1, 1>>>(test_type);
HIP_CHECK(hipDeviceSynchronize());
@@ -312,11 +297,10 @@ static bool TestAllocationOfAllAvailableMemory(int test_type,
* Local function: Launch kerBlockLevelMemoryAllocation<<<>>>
* in a loop to stress test allocation and deallocation.
*/
static bool TestMemoryAllocationInLoop(int test_type,
bool isMultikernel = false) {
static bool TestMemoryAllocationInLoop(int test_type, bool isMultikernel = false) {
int *outputVec_d{nullptr}, *outputVec_h{nullptr};
int arraysize = (BLOCKSIZE * GRIDSIZE);
outputVec_h = reinterpret_cast<int*> (malloc(sizeof(int) * arraysize));
outputVec_h = reinterpret_cast<int*>(malloc(sizeof(int) * arraysize));
REQUIRE(outputVec_h != nullptr);
HIP_CHECK(hipMalloc(&outputVec_d, (sizeof(int) * arraysize)));
bool bPassed = true;
@@ -333,13 +317,11 @@ static bool TestMemoryAllocationInLoop(int test_type,
kerWrite<<<GRIDSIZE, BLOCKSIZE>>>();
kerFree<<<GRIDSIZE, BLOCKSIZE>>>(outputVec_d, test_type);
} else {
kerBlockLevelMemoryAllocation<<<GRIDSIZE, BLOCKSIZE>>>(outputVec_d,
test_type);
kerBlockLevelMemoryAllocation<<<GRIDSIZE, BLOCKSIZE>>>(outputVec_d, test_type);
}
HIP_CHECK(hipDeviceSynchronize());
// Copy to host buffer
HIP_CHECK(hipMemcpy(outputVec_h, outputVec_d, sizeof(int) * arraysize,
hipMemcpyDefault));
HIP_CHECK(hipMemcpy(outputVec_h, outputVec_d, sizeof(int) * arraysize, hipMemcpyDefault));
bPassed = true;
for (int idx = 0; idx < arraysize; idx++) {
if (outputVec_h[idx] != idx) {
@@ -359,32 +341,36 @@ static bool TestMemoryAllocationInLoop(int test_type,
* Scenario: Test malloc till nullptr is returned using even chunksize.
*/
TEST_CASE("Stress_deviceAllocation_malloc_Even") {
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE,
NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE));
REQUIRE(true ==
TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE, NO_ALLOCATION_ONHOST,
MEMORY_CHUNK_SIZE));
}
/**
* Scenario: Test malloc till nullptr is returned using odd chunksize.
*/
TEST_CASE("Stress_deviceAllocation_malloc_Odd") {
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE,
NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE_ODD));
REQUIRE(true ==
TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE, NO_ALLOCATION_ONHOST,
MEMORY_CHUNK_SIZE_ODD));
}
/**
* Scenario: Test new till nullptr is returned using even chunksize.
*/
TEST_CASE("Stress_deviceAllocation_new_Even") {
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE,
NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE));
REQUIRE(
true ==
TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE, NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE));
}
/**
* Scenario: Test new till nullptr is returned using odd chunksize.
*/
TEST_CASE("Stress_deviceAllocation_new_Odd") {
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE,
NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE_ODD));
REQUIRE(true ==
TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE, NO_ALLOCATION_ONHOST,
MEMORY_CHUNK_SIZE_ODD));
}
/**
@@ -393,8 +379,9 @@ TEST_CASE("Stress_deviceAllocation_new_Odd") {
* from host.
*/
TEST_CASE("Stress_deviceAllocation_malloc_hipmallocmanaged") {
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE,
ALLOCATE_ONHOST_HIPMALLOCMANAGED, MEMORY_CHUNK_SIZE));
REQUIRE(true ==
TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE, ALLOCATE_ONHOST_HIPMALLOCMANAGED,
MEMORY_CHUNK_SIZE));
}
/**
@@ -403,8 +390,9 @@ TEST_CASE("Stress_deviceAllocation_malloc_hipmallocmanaged") {
* from host.
*/
TEST_CASE("Stress_deviceAllocation_new_hipmallocmanaged") {
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE,
ALLOCATE_ONHOST_HIPMALLOCMANAGED, MEMORY_CHUNK_SIZE));
REQUIRE(true ==
TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE, ALLOCATE_ONHOST_HIPMALLOCMANAGED,
MEMORY_CHUNK_SIZE));
}
/**
@@ -412,8 +400,9 @@ TEST_CASE("Stress_deviceAllocation_new_hipmallocmanaged") {
* is returned. Device memory is also allocated using hipmalloc from host.
*/
TEST_CASE("Stress_deviceAllocation_malloc_hipmalloc") {
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE,
ALLOCATE_ONHOST_HIPMALLOC, MEMORY_CHUNK_SIZE));
REQUIRE(true ==
TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE, ALLOCATE_ONHOST_HIPMALLOC,
MEMORY_CHUNK_SIZE));
}
/**
@@ -421,8 +410,9 @@ TEST_CASE("Stress_deviceAllocation_malloc_hipmalloc") {
* is returned. Device memory is also allocated using hipmalloc from host.
*/
TEST_CASE("Stress_deviceAllocation_new_hipmalloc") {
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE,
ALLOCATE_ONHOST_HIPMALLOC, MEMORY_CHUNK_SIZE));
REQUIRE(true ==
TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE, ALLOCATE_ONHOST_HIPMALLOC,
MEMORY_CHUNK_SIZE));
}
/**
@@ -434,7 +424,7 @@ TEST_CASE("Stress_deviceAllocation_Negative") {
size_t avail = 0, tot = 0;
HIP_CHECK(hipMemGetInfo(&avail, &tot));
printf("Available Memory in GPU = %zu \n", avail);
ret_h = reinterpret_cast<int*> (malloc(sizeof(int)));
ret_h = reinterpret_cast<int*>(malloc(sizeof(int)));
REQUIRE(ret_h != nullptr);
HIP_CHECK(hipMalloc(&ret_d, (sizeof(int))));
SECTION("Test allocation with malloc") {
@@ -41,8 +41,7 @@ TEST_CASE("Stress_hipHostMalloc_MaxAllocation") {
INFO("Max Allocation of " << memFree << " bytes!");
while (hipHostMalloc(&d_ptr, memFree) != hipSuccess && memFree > 1) {
counter++;
INFO("Attempt to allocate " << memFree << \
" bytes out of " << devMemFree << "bytes Failed!");
INFO("Attempt to allocate " << memFree << " bytes out of " << devMemFree << "bytes Failed!");
memFree >>= 1; // reduce the memory to be allocated by half
REQUIRE(counter <= 2); // Make sure that we are atleast able to allocate
// 1/4th of max memory
@@ -50,8 +49,7 @@ TEST_CASE("Stress_hipHostMalloc_MaxAllocation") {
HIP_CHECK(hipMemset(d_ptr, 1, memFree));
HIP_CHECK(hipDeviceSynchronize()); // Flush caches
REQUIRE(std::all_of(d_ptr, d_ptr + memFree,
[](unsigned char n) { return n == 1; }));
REQUIRE(std::all_of(d_ptr, d_ptr + memFree, [](unsigned char n) { return n == 1; }));
HIP_CHECK(hipHostFree(d_ptr));
}
@@ -67,8 +65,7 @@ TEST_CASE("Stress_hipHostMalloc_MaxAllocation_AllGpu") {
// Get available GPU memory and total GPU memory
HIP_CHECK(hipSetDevice(dev));
HIP_CHECK(hipMemGetInfo(&availableMem, &maxGpuMem));
size_t allocsize = maxGpuMem +
((maxGpuMem*ADDITIONAL_MEMORY_PERCENT)/100);
size_t allocsize = maxGpuMem + ((maxGpuMem * ADDITIONAL_MEMORY_PERCENT) / 100);
// Get free host In bytes
size_t hostMemFree = HipTest::getMemoryAmount() * 1024 * 1024;
if (allocsize < hostMemFree) {

Some files were not shown because too many files have changed in this diff Show More