SWDEV-470698 - fix formatting, add format check workflow (#657)
This commit is contained in:
committed by
GitHub
parent
5840940caa
commit
f7338717ae
@@ -25,20 +25,17 @@
|
||||
#include <hip_test_checkers.hh>
|
||||
|
||||
#define N 1048576
|
||||
__managed__ float A[N]; // Accessible by ALL CPU and GPU functions !!!
|
||||
__managed__ float A[N]; // Accessible by ALL CPU and GPU functions !!!
|
||||
__managed__ float B[N];
|
||||
__managed__ int x = 0;
|
||||
__managed__ int x = 0;
|
||||
|
||||
__global__ void add(const float *A, float *B) {
|
||||
__global__ void add(const float* A, float* B) {
|
||||
int index = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
int stride = blockDim.x * gridDim.x;
|
||||
for (int i = index; i < N; i += stride)
|
||||
B[i] = A[i] + B[i];
|
||||
for (int i = index; i < N; i += stride) B[i] = A[i] + B[i];
|
||||
}
|
||||
|
||||
__global__ void GPU_func() {
|
||||
x++;
|
||||
}
|
||||
__global__ void GPU_func() { x++; }
|
||||
|
||||
TEST_CASE("Unit_hipManagedKeyword_SingleGpu") {
|
||||
for (int i = 0; i < N; i++) {
|
||||
@@ -57,8 +54,7 @@ TEST_CASE("Unit_hipManagedKeyword_SingleGpu") {
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
float maxError = 0.0f;
|
||||
for (int i = 0; i < N; i++)
|
||||
maxError = fmax(maxError, fabs(B[i]-3.0f));
|
||||
for (int i = 0; i < N; i++) maxError = fmax(maxError, fabs(B[i] - 3.0f));
|
||||
|
||||
REQUIRE(maxError == 0.0f);
|
||||
}
|
||||
@@ -67,11 +63,9 @@ TEST_CASE("Unit_hipManagedKeyword_MultiGpu") {
|
||||
int numDevices = 0;
|
||||
HIP_CHECK(hipGetDeviceCount(&numDevices));
|
||||
|
||||
for (int i = 0; i < numDevices; i++){
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
int managed_memory = 0;
|
||||
HIPCHECK(hipDeviceGetAttribute(&managed_memory,
|
||||
hipDeviceAttributeManagedMemory,
|
||||
i));
|
||||
HIPCHECK(hipDeviceGetAttribute(&managed_memory, hipDeviceAttributeManagedMemory, i));
|
||||
if (!managed_memory) {
|
||||
HipTest::HIP_SKIP_TEST("managed memory access not supported on device");
|
||||
return;
|
||||
@@ -80,7 +74,7 @@ TEST_CASE("Unit_hipManagedKeyword_MultiGpu") {
|
||||
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
GPU_func<<< 1, 1 >>>();
|
||||
GPU_func<<<1, 1>>>();
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
}
|
||||
REQUIRE(x == numDevices);
|
||||
|
||||
+12722
-13226
File diff suppressed because it is too large
Load Diff
+344
-401
File diff suppressed because it is too large
Load Diff
@@ -164,11 +164,10 @@ void TestContext::getConfigFiles() {
|
||||
}
|
||||
|
||||
std::string env_config = TestContext::getEnvVar("HIP_CATCH_EXCLUDE_FILE");
|
||||
LogPrintf("Env Config file: %s",
|
||||
(!env_config.empty()) ? env_config.c_str() : "Not found");
|
||||
LogPrintf("Env Config file: %s", (!env_config.empty()) ? env_config.c_str() : "Not found");
|
||||
// HIP_CATCH_EXCLUDE_FILE is set for custom file path
|
||||
if (!env_config.empty()) {
|
||||
if(fs::exists(env_config)) {
|
||||
if (fs::exists(env_config)) {
|
||||
config_.json_files.push_back(env_config);
|
||||
}
|
||||
} else {
|
||||
@@ -236,7 +235,8 @@ bool TestContext::parseJsonFiles() {
|
||||
}
|
||||
// Open the file
|
||||
std::ifstream js_file(fl);
|
||||
std::string json_str((std::istreambuf_iterator<char>(js_file)), std::istreambuf_iterator<char>());
|
||||
std::string json_str((std::istreambuf_iterator<char>(js_file)),
|
||||
std::istreambuf_iterator<char>());
|
||||
LogPrintf("Json contents:: %s", json_str.data());
|
||||
|
||||
picojson::value v;
|
||||
|
||||
@@ -6,9 +6,9 @@
|
||||
#include "hip_test_context.hh"
|
||||
|
||||
std::vector<std::unordered_set<std::string>> GCNArchFeatMap = {
|
||||
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_FINEGRAIN_HWSUPPORT
|
||||
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_HMM
|
||||
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_TEXTURES_NOT_SUPPORTED
|
||||
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_FINEGRAIN_HWSUPPORT
|
||||
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_HMM
|
||||
{"gfx90a", "gfx942", "gfx950"}, // CT_FEATURE_TEXTURES_NOT_SUPPORTED
|
||||
};
|
||||
|
||||
#if HT_AMD
|
||||
@@ -24,7 +24,7 @@ std::string TrimAndGetGFXName(const std::string& full_gfx_name) {
|
||||
gfx_name = full_gfx_name.substr(0, pos);
|
||||
}
|
||||
|
||||
assert(gfx_name.substr(0,3) == "gfx");
|
||||
assert(gfx_name.substr(0, 3) == "gfx");
|
||||
return gfx_name;
|
||||
}
|
||||
#endif
|
||||
@@ -32,14 +32,14 @@ std::string TrimAndGetGFXName(const std::string& full_gfx_name) {
|
||||
// Check if the GCN Maps
|
||||
bool CheckIfFeatSupported(enum CTFeatures test_feat, std::string gcn_arch) {
|
||||
#if HT_NVIDIA
|
||||
return true; // returning true since feature check does not exist for NV.
|
||||
return true; // returning true since feature check does not exist for NV.
|
||||
#elif HT_AMD
|
||||
assert(test_feat >= 0 && test_feat < CTFeatures::CT_FEATURE_LAST);
|
||||
gcn_arch = TrimAndGetGFXName(gcn_arch);
|
||||
assert(gcn_arch != "");
|
||||
return (GCNArchFeatMap[test_feat].find(gcn_arch) != GCNArchFeatMap[test_feat].cend());
|
||||
#else
|
||||
std::cout<<"Platform has to be either AMD or NVIDIA, asserting..."<<std::endl;
|
||||
std::cout << "Platform has to be either AMD or NVIDIA, asserting..." << std::endl;
|
||||
assert(false);
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -122,8 +122,8 @@ inline dim3 GenerateThreadDimensions() {
|
||||
map([max = props.maxThreadsDim[2], warp_size = props.warpSize](
|
||||
double i) { return dim3(1, 1, std::min(static_cast<int>(i * warp_size), max)); },
|
||||
values(multipliers)),
|
||||
dim3(16, 8, 8), dim3(32, 32, 1), dim3(64, 8, 2), dim3(16, 16, 3), dim3(props.warpSize - 1, 3, 3),
|
||||
dim3(props.warpSize + 1, 3, 3));
|
||||
dim3(16, 8, 8), dim3(32, 32, 1), dim3(64, 8, 2), dim3(16, 16, 3),
|
||||
dim3(props.warpSize - 1, 3, 3), dim3(props.warpSize + 1, 3, 3));
|
||||
}
|
||||
|
||||
/* Generate dimensions for 1D, 2D and 3D grids of blocks */
|
||||
@@ -161,8 +161,8 @@ inline dim3 GenerateThreadDimensionsForShuffle() {
|
||||
map([max = props.maxThreadsDim[2], warp_size = props.warpSize](
|
||||
double i) { return dim3(1, 1, std::min(static_cast<int>(i * warp_size), max)); },
|
||||
values(multipliers)),
|
||||
dim3(16, 8, 8), dim3(32, 32, 1), dim3(64, 8, 2), dim3(16, 16, 3), dim3(props.warpSize - 1, 3, 3),
|
||||
dim3(props.warpSize + 1, 3, 3));
|
||||
dim3(16, 8, 8), dim3(32, 32, 1), dim3(64, 8, 2), dim3(16, 16, 3),
|
||||
dim3(props.warpSize - 1, 3, 3), dim3(props.warpSize + 1, 3, 3));
|
||||
}
|
||||
|
||||
/* Generate dimensions for 1D, 2D and 3D grids of blocks - reduced set */
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#include <kernels.hh>
|
||||
|
||||
__global__ void Set(int* Ad, int val) {
|
||||
int tx = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
Ad[tx] = val;
|
||||
int tx = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
Ad[tx] = val;
|
||||
}
|
||||
@@ -41,7 +41,7 @@ static __global__ void kerTestDeviceMalloc(size_t size) {
|
||||
int myId = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
// Allocate
|
||||
if (myId == 0) {
|
||||
dev_common_ptr = reinterpret_cast<char*> (malloc(size));
|
||||
dev_common_ptr = reinterpret_cast<char*>(malloc(size));
|
||||
if (dev_common_ptr == nullptr) {
|
||||
printf("Device Allocation Failed! \n");
|
||||
return;
|
||||
@@ -67,13 +67,13 @@ static __global__ void kerTestDeviceWrite() {
|
||||
* This kernel frees the memory chunk allocated in kernel
|
||||
* kerTestDeviceMalloc using free().
|
||||
*/
|
||||
static __global__ void kerTestDeviceFree(int *result) {
|
||||
static __global__ void kerTestDeviceFree(int* result) {
|
||||
int myId = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
// Allocate
|
||||
if (myId == 0) {
|
||||
if (dev_common_ptr != nullptr) {
|
||||
*result = 1;
|
||||
for (int idx = 0; idx < (BLOCKSIZE*GRIDSIZE); idx++) {
|
||||
for (int idx = 0; idx < (BLOCKSIZE * GRIDSIZE); idx++) {
|
||||
if (*(dev_common_ptr + myId) != SCHAR_MAX) {
|
||||
*result = 0;
|
||||
break;
|
||||
@@ -105,13 +105,13 @@ static __global__ void kerTestDeviceNew(size_t size) {
|
||||
* This kernel frees the memory chunk allocated in kernel
|
||||
* kerTestDeviceNew using delete operator.
|
||||
*/
|
||||
static __global__ void kerTestDeviceDelete(int *result) {
|
||||
static __global__ void kerTestDeviceDelete(int* result) {
|
||||
int myId = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
// Allocate
|
||||
if (myId == 0) {
|
||||
if (dev_common_ptr != nullptr) {
|
||||
*result = 1;
|
||||
for (int idx = 0; idx < (BLOCKSIZE*GRIDSIZE); idx++) {
|
||||
for (int idx = 0; idx < (BLOCKSIZE * GRIDSIZE); idx++) {
|
||||
if (*(dev_common_ptr + myId) != SCHAR_MAX) {
|
||||
*result = 0;
|
||||
break;
|
||||
@@ -140,7 +140,7 @@ static bool testDeviceAllocMulProc(bool testmalloc) {
|
||||
childpid = fork();
|
||||
if (childpid > 0) { // Parent
|
||||
close(fd[1]);
|
||||
int *result_d{nullptr};
|
||||
int* result_d{nullptr};
|
||||
HIP_CHECK(hipMalloc(&result_d, sizeof(int)));
|
||||
// Allocate in parent
|
||||
if (testmalloc) {
|
||||
@@ -185,7 +185,7 @@ static bool testDeviceAllocMulProc(bool testmalloc) {
|
||||
HIP_CHECK(hipFree(result_d));
|
||||
} else if (!childpid) { // Child
|
||||
// Wait for hipDeviceSetLimit() completion in parent.
|
||||
int *result_d{nullptr};
|
||||
int* result_d{nullptr};
|
||||
HIP_CHECK(hipMalloc(&result_d, sizeof(int)));
|
||||
close(fd[0]);
|
||||
// Allocate in child
|
||||
@@ -230,7 +230,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
|
||||
bool testResult = false;
|
||||
pid_t childpid;
|
||||
int testResultChild = 0;
|
||||
size_t size = BLOCKSIZE*GRIDSIZE;
|
||||
size_t size = BLOCKSIZE * GRIDSIZE;
|
||||
// create pipe descriptors
|
||||
pipe(fd);
|
||||
// fork process
|
||||
@@ -239,7 +239,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
|
||||
close(fd[1]);
|
||||
int *result_d{nullptr}, *result_h{nullptr};
|
||||
HIP_CHECK(hipMalloc(&result_d, sizeof(int)));
|
||||
result_h = reinterpret_cast<int*> (malloc(sizeof(int)));
|
||||
result_h = reinterpret_cast<int*>(malloc(sizeof(int)));
|
||||
REQUIRE(result_h != nullptr);
|
||||
// Allocate in parent
|
||||
if (testmalloc) {
|
||||
@@ -257,8 +257,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
*result_h = 0;
|
||||
HIP_CHECK(hipMemcpy(result_h, result_d, sizeof(int),
|
||||
hipMemcpyDefault));
|
||||
HIP_CHECK(hipMemcpy(result_h, result_d, sizeof(int), hipMemcpyDefault));
|
||||
if (*result_h == 0) {
|
||||
testResult = false;
|
||||
} else {
|
||||
@@ -282,7 +281,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
|
||||
close(fd[0]);
|
||||
int *result_d{nullptr}, *result_h{nullptr};
|
||||
HIP_CHECK(hipMalloc(&result_d, sizeof(int)));
|
||||
result_h = reinterpret_cast<int*> (malloc(sizeof(int)));
|
||||
result_h = reinterpret_cast<int*>(malloc(sizeof(int)));
|
||||
REQUIRE(result_h != nullptr);
|
||||
// Allocate in child
|
||||
if (testmalloc) {
|
||||
@@ -300,8 +299,7 @@ static bool testDeviceMemMulProc(bool testmalloc) {
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
*result_h = 0;
|
||||
HIP_CHECK(hipMemcpy(result_h, result_d, sizeof(int),
|
||||
hipMemcpyDefault));
|
||||
HIP_CHECK(hipMemcpy(result_h, result_d, sizeof(int), hipMemcpyDefault));
|
||||
// send the value on the write-descriptor:
|
||||
write(fd[1], result_h, sizeof(int));
|
||||
// close the write descriptor:
|
||||
|
||||
@@ -22,5 +22,4 @@ THE SOFTWARE.
|
||||
|
||||
#include "hip/hip_runtime.h"
|
||||
|
||||
extern "C" __global__ void dummy_ker() {
|
||||
}
|
||||
extern "C" __global__ void dummy_ker() {}
|
||||
|
||||
@@ -34,7 +34,7 @@ THE SOFTWARE.
|
||||
/**
|
||||
* Fetches Gpu device count
|
||||
*/
|
||||
static void getDeviceCount(int *pdevCnt) {
|
||||
static void getDeviceCount(int* pdevCnt) {
|
||||
int fd[2], val = 0;
|
||||
pid_t childpid;
|
||||
|
||||
@@ -107,15 +107,14 @@ bool runMaskedDeviceTest(int actualNumGPUs) {
|
||||
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
|
||||
#endif
|
||||
|
||||
for (int count = 1;
|
||||
count < actualNumGPUs; count++) {
|
||||
for (int count = 1; count < actualNumGPUs; count++) {
|
||||
int major, minor;
|
||||
err = hipDeviceComputeCapability(&major, &minor, count);
|
||||
if (err == hipSuccess) {
|
||||
testResult = false;
|
||||
} else {
|
||||
printf("hipDeviceComputeCapability: Error Code Returned: '%s'(%d)\n",
|
||||
hipGetErrorString(err), err);
|
||||
hipGetErrorString(err), err);
|
||||
}
|
||||
}
|
||||
close(fd[0]);
|
||||
|
||||
@@ -38,7 +38,7 @@ namespace hipDeviceGetPCIBusIdTests {
|
||||
/**
|
||||
* Fetches Gpu device count
|
||||
*/
|
||||
void getDeviceCount(int *pdevCnt) {
|
||||
void getDeviceCount(int* pdevCnt) {
|
||||
int fd[2], val = 0;
|
||||
pid_t childpid;
|
||||
|
||||
@@ -112,14 +112,13 @@ bool testWithMaskedDevices(int actualNumGPUs) {
|
||||
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
|
||||
#endif
|
||||
|
||||
for (int count = 1;
|
||||
count < actualNumGPUs; count++) {
|
||||
for (int count = 1; count < actualNumGPUs; count++) {
|
||||
err = hipDeviceGetPCIBusId(pciBusId, MAX_DEVICE_LENGTH, count);
|
||||
if (err == hipSuccess) {
|
||||
testResult &= false;
|
||||
} else {
|
||||
printf("hipGetDeviceProperties: Error Code Returned: '%s'(%d)\n",
|
||||
hipGetErrorString(err), err);
|
||||
printf("hipGetDeviceProperties: Error Code Returned: '%s'(%d)\n", hipGetErrorString(err),
|
||||
err);
|
||||
}
|
||||
}
|
||||
close(fd[0]);
|
||||
@@ -143,8 +142,7 @@ bool testWithMaskedDevices(int actualNumGPUs) {
|
||||
}
|
||||
|
||||
|
||||
bool getPciBusId(int deviceCount,
|
||||
char **hipDeviceList) {
|
||||
bool getPciBusId(int deviceCount, char** hipDeviceList) {
|
||||
for (int i = 0; i < deviceCount; i++) {
|
||||
HIP_CHECK(hipDeviceGetPCIBusId(hipDeviceList[i], MAX_DEVICE_LENGTH, i));
|
||||
}
|
||||
@@ -175,11 +173,11 @@ TEST_CASE("Unit_hipDeviceGetPCIBusId_MaskedDevices") {
|
||||
* hipDeviceGetPCIBusId vs lspci
|
||||
*/
|
||||
TEST_CASE("Unit_hipDeviceGetPCIBusId_CheckPciBusIDWithLspci") {
|
||||
FILE *fpipe;
|
||||
FILE* fpipe;
|
||||
{
|
||||
// Check if lspci is installed, if not, don't proceed
|
||||
char const *cmd = "lspci --version";
|
||||
char *lspciCheck{nullptr};
|
||||
char const* cmd = "lspci --version";
|
||||
char* lspciCheck{nullptr};
|
||||
constexpr auto MaxLen = 50;
|
||||
char temp[MaxLen]{};
|
||||
|
||||
@@ -199,9 +197,9 @@ TEST_CASE("Unit_hipDeviceGetPCIBusId_CheckPciBusIDWithLspci") {
|
||||
HIP_CHECK(hipGetDeviceCount(&deviceCount));
|
||||
REQUIRE_FALSE(deviceCount == 0);
|
||||
// Allocate an array of pointer to characters
|
||||
char **hipDeviceList = new char*[deviceCount];
|
||||
char** hipDeviceList = new char*[deviceCount];
|
||||
REQUIRE_FALSE(hipDeviceList == nullptr);
|
||||
char **pciDeviceList = new char*[deviceCount];
|
||||
char** pciDeviceList = new char*[deviceCount];
|
||||
REQUIRE_FALSE(pciDeviceList == nullptr);
|
||||
for (int i = 0; i < deviceCount; i++) {
|
||||
hipDeviceList[i] = new char[MAX_DEVICE_LENGTH];
|
||||
@@ -211,14 +209,16 @@ TEST_CASE("Unit_hipDeviceGetPCIBusId_CheckPciBusIDWithLspci") {
|
||||
}
|
||||
|
||||
hipDeviceGetPCIBusIdTests::getPciBusId(deviceCount, hipDeviceList);
|
||||
char const *command = nullptr;
|
||||
char const* command = nullptr;
|
||||
// Get lspci device list and compare with hip device list
|
||||
if ((TestContext::get()).isNvidia()) {
|
||||
command = "lspci -D | grep controller | grep NVIDIA | "
|
||||
"cut -d ' ' -f 1";
|
||||
command =
|
||||
"lspci -D | grep controller | grep NVIDIA | "
|
||||
"cut -d ' ' -f 1";
|
||||
} else {
|
||||
command = "lspci -D | grep -e controller -e accelerator | grep AMD/ATI | "
|
||||
"cut -d ' ' -f 1";
|
||||
command =
|
||||
"lspci -D | grep -e controller -e accelerator | grep AMD/ATI | "
|
||||
"cut -d ' ' -f 1";
|
||||
}
|
||||
fpipe = popen(command, "r");
|
||||
REQUIRE_FALSE(fpipe == nullptr);
|
||||
@@ -229,15 +229,13 @@ TEST_CASE("Unit_hipDeviceGetPCIBusId_CheckPciBusIDWithLspci") {
|
||||
while (fgets(pciDeviceList[index], MAX_DEVICE_LENGTH, fpipe)) {
|
||||
bool bMatchFound = false;
|
||||
for (int deviceNo = 0; deviceNo < deviceCount; deviceNo++) {
|
||||
if (!strncasecmp(pciDeviceList[index], hipDeviceList[deviceNo],
|
||||
cmpLen)) {
|
||||
if (!strncasecmp(pciDeviceList[index], hipDeviceList[deviceNo], cmpLen)) {
|
||||
deviceMatchCount++;
|
||||
bMatchFound = true;
|
||||
}
|
||||
}
|
||||
if (bMatchFound == false) {
|
||||
printf("PCI device: %s is not reported by HIP\n",
|
||||
pciDeviceList[index]);
|
||||
printf("PCI device: %s is not reported by HIP\n", pciDeviceList[index]);
|
||||
}
|
||||
index++;
|
||||
if (index >= deviceCount) break;
|
||||
|
||||
@@ -34,7 +34,7 @@ THE SOFTWARE.
|
||||
/**
|
||||
* Fetches Gpu device count
|
||||
*/
|
||||
static void getDeviceCount(int *pdevCnt) {
|
||||
static void getDeviceCount(int* pdevCnt) {
|
||||
int fd[2], val = 0;
|
||||
pid_t childpid;
|
||||
|
||||
@@ -109,15 +109,13 @@ static bool getTotalMemoryOfMaskedDevices(int actualNumGPUs) {
|
||||
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
|
||||
#endif
|
||||
|
||||
for (int count = 1;
|
||||
count < actualNumGPUs; count++) {
|
||||
for (int count = 1; count < actualNumGPUs; count++) {
|
||||
size_t totMem;
|
||||
err = hipDeviceTotalMem(&totMem, count);
|
||||
if (err == hipSuccess) {
|
||||
testResult &= false;
|
||||
} else {
|
||||
printf("hipDeviceTotalMem: Error Code Returned: '%s'(%d)\n",
|
||||
hipGetErrorString(err), err);
|
||||
printf("hipDeviceTotalMem: Error Code Returned: '%s'(%d)\n", hipGetErrorString(err), err);
|
||||
}
|
||||
}
|
||||
close(fd[0]);
|
||||
|
||||
@@ -37,7 +37,7 @@ THE SOFTWARE.
|
||||
/**
|
||||
* Fetches Gpu device count
|
||||
*/
|
||||
static void getDeviceCount(int *pdevCnt) {
|
||||
static void getDeviceCount(int* pdevCnt) {
|
||||
int fd[2], val = 0;
|
||||
pid_t childpid;
|
||||
|
||||
@@ -112,15 +112,14 @@ static bool validateGetAttributeOfMaskedDevices(int actualNumGPUs) {
|
||||
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
|
||||
#endif
|
||||
|
||||
for (int count = 1;
|
||||
count < actualNumGPUs; count++) {
|
||||
for (int count = 1; count < actualNumGPUs; count++) {
|
||||
int pi = -1;
|
||||
err = hipDeviceGetAttribute(&pi, hipDeviceAttributePciBusId, count);
|
||||
if (err == hipSuccess) {
|
||||
testResult &= false;
|
||||
} else {
|
||||
printf("hipDeviceGetAttribute: Error Code Returned: '%s'(%d)\n",
|
||||
hipGetErrorString(err), err);
|
||||
printf("hipDeviceGetAttribute: Error Code Returned: '%s'(%d)\n", hipGetErrorString(err),
|
||||
err);
|
||||
}
|
||||
}
|
||||
close(fd[0]);
|
||||
|
||||
@@ -36,7 +36,7 @@ THE SOFTWARE.
|
||||
/**
|
||||
* Fetches Gpu device count
|
||||
*/
|
||||
static void getDeviceCount(int *pdevCnt) {
|
||||
static void getDeviceCount(int* pdevCnt) {
|
||||
int fd[2], val = 0;
|
||||
pid_t childpid;
|
||||
|
||||
@@ -84,7 +84,6 @@ static void getDeviceCount(int *pdevCnt) {
|
||||
}
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* Tries to fetch device properties of masked devices and returns pass/fail.
|
||||
*/
|
||||
@@ -112,15 +111,14 @@ static bool validateGetPropsOfMaskedDevices(int actualNumGPUs) {
|
||||
setenv("HIP_VISIBLE_DEVICES", visibleDeviceString, 1);
|
||||
#endif
|
||||
|
||||
for (int count = 1;
|
||||
count < actualNumGPUs; count++) {
|
||||
for (int count = 1; count < actualNumGPUs; count++) {
|
||||
hipDeviceProp_t prop;
|
||||
err = hipGetDeviceProperties(&prop, count);
|
||||
if (err == hipSuccess) {
|
||||
testResult &= false;
|
||||
} else {
|
||||
printf("hipGetDeviceProperties: Error Code Returned: '%s'(%d)\n",
|
||||
hipGetErrorString(err), err);
|
||||
printf("hipGetDeviceProperties: Error Code Returned: '%s'(%d)\n", hipGetErrorString(err),
|
||||
err);
|
||||
}
|
||||
}
|
||||
close(fd[0]);
|
||||
@@ -144,7 +142,6 @@ static bool validateGetPropsOfMaskedDevices(int actualNumGPUs) {
|
||||
}
|
||||
|
||||
|
||||
|
||||
/**
|
||||
* Scenario: Validate behavior of hipGetDeviceProperties for masked devices.
|
||||
*/
|
||||
|
||||
@@ -34,8 +34,8 @@ THE SOFTWARE.
|
||||
* This opaque handle may be copied into other processes and opened with hipIpcOpenEventHandle.
|
||||
*/
|
||||
|
||||
#define BUF_SIZE 4096
|
||||
#define MAX_DEVICES 16
|
||||
#define BUF_SIZE 4096
|
||||
#define MAX_DEVICES 16
|
||||
|
||||
|
||||
typedef struct ipcEventInfo {
|
||||
@@ -60,7 +60,7 @@ typedef struct ipcBarrier {
|
||||
Get device count and list down devices with
|
||||
P2P access with Device 0.
|
||||
*/
|
||||
void getDevices(ipcDevices_t *devices) {
|
||||
void getDevices(ipcDevices_t* devices) {
|
||||
pid_t pid = fork();
|
||||
|
||||
if (!pid) {
|
||||
@@ -70,9 +70,9 @@ void getDevices(ipcDevices_t *devices) {
|
||||
HIP_CHECK(hipGetDeviceCount(&devCnt));
|
||||
|
||||
if (devCnt < 2) {
|
||||
devices->count = 0;
|
||||
WARN("Count less than expected number of devices");
|
||||
exit(EXIT_SUCCESS);
|
||||
devices->count = 0;
|
||||
WARN("Count less than expected number of devices");
|
||||
exit(EXIT_SUCCESS);
|
||||
}
|
||||
|
||||
// Device 0
|
||||
@@ -85,27 +85,26 @@ void getDevices(ipcDevices_t *devices) {
|
||||
|
||||
int canPeerAccess_0i, canPeerAccess_i0;
|
||||
for (i = 1; i < devCnt; i++) {
|
||||
HIP_CHECK(hipDeviceCanAccessPeer(&canPeerAccess_0i, 0, i));
|
||||
HIP_CHECK(hipDeviceCanAccessPeer(&canPeerAccess_i0, i, 0));
|
||||
HIP_CHECK(hipDeviceCanAccessPeer(&canPeerAccess_0i, 0, i));
|
||||
HIP_CHECK(hipDeviceCanAccessPeer(&canPeerAccess_i0, i, 0));
|
||||
|
||||
if (canPeerAccess_0i * canPeerAccess_i0) {
|
||||
devices->ordinals[i] = i;
|
||||
INFO("Two-way peer access is available between GPU"
|
||||
<< devices->ordinals[0] <<" and GPU"
|
||||
<< devices->ordinals[devices->count]);
|
||||
devices->count += 1;
|
||||
}
|
||||
if (canPeerAccess_0i * canPeerAccess_i0) {
|
||||
devices->ordinals[i] = i;
|
||||
INFO("Two-way peer access is available between GPU" << devices->ordinals[0] << " and GPU"
|
||||
<< devices->ordinals[devices->count]);
|
||||
devices->count += 1;
|
||||
}
|
||||
}
|
||||
|
||||
exit(EXIT_SUCCESS);
|
||||
} else {
|
||||
int status;
|
||||
waitpid(pid, &status, 0);
|
||||
HIP_ASSERT(!status);
|
||||
int status;
|
||||
waitpid(pid, &status, 0);
|
||||
HIP_ASSERT(!status);
|
||||
}
|
||||
}
|
||||
|
||||
static ipcBarrier_t *g_Barrier{};
|
||||
static ipcBarrier_t* g_Barrier{};
|
||||
static bool g_procSense;
|
||||
static int g_processCnt;
|
||||
|
||||
@@ -121,11 +120,11 @@ void processBarrier() {
|
||||
|
||||
} else {
|
||||
while (g_Barrier->sense == g_procSense) {
|
||||
if (!g_Barrier->allExit) {
|
||||
sched_yield();
|
||||
} else {
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
if (!g_Barrier->allExit) {
|
||||
sched_yield();
|
||||
} else {
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -133,9 +132,9 @@ void processBarrier() {
|
||||
}
|
||||
|
||||
|
||||
__global__ void computeKernel(int *dst, int *src, int num) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
dst[idx] = src[idx] / num;
|
||||
__global__ void computeKernel(int* dst, int* src, int num) {
|
||||
int idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
dst[idx] = src[idx] / num;
|
||||
}
|
||||
|
||||
/*
|
||||
@@ -144,14 +143,14 @@ __global__ void computeKernel(int *dst, int *src, int num) {
|
||||
* and records event.
|
||||
* 3) Process 0 synchronizes event and validates the resulting buffer.
|
||||
*/
|
||||
void runMultiProcKernel(ipcEventInfo_t *shmEventInfo, int index) {
|
||||
int *d_ptr;
|
||||
void runMultiProcKernel(ipcEventInfo_t* shmEventInfo, int index) {
|
||||
int* d_ptr;
|
||||
int hData[BUF_SIZE]{};
|
||||
unsigned int seed = time(nullptr);
|
||||
|
||||
// Randomize data before computation
|
||||
for (int i = 0; i < BUF_SIZE; i++) {
|
||||
hData[i] = rand_r(&seed);
|
||||
hData[i] = rand_r(&seed);
|
||||
}
|
||||
|
||||
HIP_CHECK(hipSetDevice(shmEventInfo[index].device));
|
||||
@@ -162,8 +161,7 @@ void runMultiProcKernel(ipcEventInfo_t *shmEventInfo, int index) {
|
||||
|
||||
HIP_CHECK(hipMalloc(&d_ptr, BUF_SIZE * g_processCnt * sizeof(int)));
|
||||
HIP_CHECK(hipIpcGetMemHandle(&shmEventInfo[0].memHandle, d_ptr));
|
||||
HIP_CHECK(hipMemcpy(d_ptr, hData,
|
||||
BUF_SIZE * sizeof(int), hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(d_ptr, hData, BUF_SIZE * sizeof(int), hipMemcpyHostToDevice));
|
||||
|
||||
// Barrier 1: Process0 will wait for all processes to create event handles,
|
||||
// signals device memory creation.
|
||||
@@ -181,40 +179,38 @@ void runMultiProcKernel(ipcEventInfo_t *shmEventInfo, int index) {
|
||||
HIP_CHECK(hipEventSynchronize(event[i]));
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemcpy(h_results, d_ptr + BUF_SIZE,
|
||||
BUF_SIZE * (g_processCnt - 1) * sizeof(int), hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipMemcpy(h_results, d_ptr + BUF_SIZE, BUF_SIZE * (g_processCnt - 1) * sizeof(int),
|
||||
hipMemcpyDeviceToHost));
|
||||
|
||||
// Barrier 3: Process0 signals event usage is done.
|
||||
processBarrier();
|
||||
HIP_CHECK(hipFree(d_ptr));
|
||||
for (int n = 1; n < g_processCnt; n++) {
|
||||
for (int i = 0; i < BUF_SIZE; i++) {
|
||||
if (hData[i]/(n + 1) != h_results[(n-1) * BUF_SIZE + i]) {
|
||||
WARN("Data validation error at index " << i << " n" << n);
|
||||
g_Barrier->allExit = true;
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
for (int i = 0; i < BUF_SIZE; i++) {
|
||||
if (hData[i] / (n + 1) != h_results[(n - 1) * BUF_SIZE + i]) {
|
||||
WARN("Data validation error at index " << i << " n" << n);
|
||||
g_Barrier->allExit = true;
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int i = 1; i < g_processCnt; i++) {
|
||||
HIP_CHECK(hipEventDestroy(event[i]));
|
||||
}
|
||||
} else {
|
||||
hipEvent_t event;
|
||||
HIP_CHECK(hipEventCreateWithFlags(&event,
|
||||
hipEventDisableTiming | hipEventInterprocess));
|
||||
HIP_CHECK(hipEventCreateWithFlags(&event, hipEventDisableTiming | hipEventInterprocess));
|
||||
HIP_CHECK(hipIpcGetEventHandle(&shmEventInfo[index].eventHandle, event));
|
||||
|
||||
// Barrier 1 : wait until proc 0 initializes device memory,
|
||||
// signals event creation.
|
||||
processBarrier();
|
||||
HIP_CHECK(hipIpcOpenMemHandle(reinterpret_cast<void **>(&d_ptr),
|
||||
shmEventInfo[0].memHandle,
|
||||
hipIpcMemLazyEnablePeerAccess));
|
||||
HIP_CHECK(hipIpcOpenMemHandle(reinterpret_cast<void**>(&d_ptr), shmEventInfo[0].memHandle,
|
||||
hipIpcMemLazyEnablePeerAccess));
|
||||
const dim3 threads(512, 1);
|
||||
const dim3 blocks(BUF_SIZE / threads.x, 1);
|
||||
hipLaunchKernelGGL(computeKernel, dim3(blocks), dim3(threads), 0, 0,
|
||||
d_ptr + index *BUF_SIZE, d_ptr, index + 1);
|
||||
hipLaunchKernelGGL(computeKernel, dim3(blocks), dim3(threads), 0, 0, d_ptr + index * BUF_SIZE,
|
||||
d_ptr, index + 1);
|
||||
HIP_CHECK(hipGetLastError());
|
||||
HIP_CHECK(hipEventRecord(event));
|
||||
|
||||
@@ -243,10 +239,10 @@ void runMultiProcKernel(ipcEventInfo_t *shmEventInfo, int index) {
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_hipIpcEventHandle_Functional") {
|
||||
ipcDevices_t *shmDevices;
|
||||
ipcEventInfo_t *shmEventInfo;
|
||||
shmDevices = reinterpret_cast<ipcDevices_t *> (mmap(NULL, sizeof(*shmDevices),
|
||||
PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
|
||||
ipcDevices_t* shmDevices;
|
||||
ipcEventInfo_t* shmEventInfo;
|
||||
shmDevices = reinterpret_cast<ipcDevices_t*>(
|
||||
mmap(NULL, sizeof(*shmDevices), PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
|
||||
REQUIRE(MAP_FAILED != shmDevices);
|
||||
|
||||
getDevices(shmDevices);
|
||||
@@ -259,8 +255,8 @@ TEST_CASE("Unit_hipIpcEventHandle_Functional") {
|
||||
g_processCnt = (shmDevices->count > MAX_DEVICES) ? MAX_DEVICES : shmDevices->count;
|
||||
|
||||
// Barrier is used to synchronize processes created.
|
||||
g_Barrier = reinterpret_cast<ipcBarrier_t *> (mmap(NULL, sizeof(*g_Barrier),
|
||||
PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
|
||||
g_Barrier = reinterpret_cast<ipcBarrier_t*>(
|
||||
mmap(NULL, sizeof(*g_Barrier), PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
|
||||
REQUIRE(MAP_FAILED != g_Barrier);
|
||||
memset(g_Barrier, 0, sizeof(*g_Barrier));
|
||||
|
||||
@@ -268,9 +264,9 @@ TEST_CASE("Unit_hipIpcEventHandle_Functional") {
|
||||
g_procSense = 0;
|
||||
|
||||
// shared memory for Event and memHandle Info
|
||||
shmEventInfo = reinterpret_cast<ipcEventInfo_t *>(mmap(NULL,
|
||||
g_processCnt * sizeof(*shmEventInfo),
|
||||
PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
|
||||
shmEventInfo = reinterpret_cast<ipcEventInfo_t*>(mmap(NULL, g_processCnt * sizeof(*shmEventInfo),
|
||||
PROT_READ | PROT_WRITE,
|
||||
MAP_SHARED | MAP_ANONYMOUS, 0, 0));
|
||||
REQUIRE(MAP_FAILED != shmEventInfo);
|
||||
|
||||
// initialize shared memory
|
||||
@@ -279,14 +275,14 @@ TEST_CASE("Unit_hipIpcEventHandle_Functional") {
|
||||
int index = 0;
|
||||
|
||||
for (int i = 1; i < g_processCnt; i++) {
|
||||
int pid = fork();
|
||||
int pid = fork();
|
||||
|
||||
if (!pid) {
|
||||
index = i;
|
||||
break;
|
||||
} else {
|
||||
shmEventInfo[i].pid = pid;
|
||||
}
|
||||
if (!pid) {
|
||||
index = i;
|
||||
break;
|
||||
} else {
|
||||
shmEventInfo[i].pid = pid;
|
||||
}
|
||||
}
|
||||
|
||||
shmEventInfo[index].device = shmDevices->ordinals[index];
|
||||
@@ -297,9 +293,9 @@ TEST_CASE("Unit_hipIpcEventHandle_Functional") {
|
||||
// Cleanup
|
||||
if (index == 0) {
|
||||
for (int i = 1; i < g_processCnt; i++) {
|
||||
int status;
|
||||
waitpid(shmEventInfo[i].pid, &status, 0);
|
||||
HIP_ASSERT(WIFEXITED(status));
|
||||
int status;
|
||||
waitpid(shmEventInfo[i].pid, &status, 0);
|
||||
HIP_ASSERT(WIFEXITED(status));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -341,8 +337,7 @@ TEST_CASE("Unit_hipIpcEventHandle_ParameterValidation") {
|
||||
hipEvent_t event;
|
||||
hipIpcEventHandle_t eventHandle;
|
||||
hipError_t ret;
|
||||
HIP_CHECK(hipEventCreateWithFlags(&event,
|
||||
hipEventDisableTiming | hipEventInterprocess));
|
||||
HIP_CHECK(hipEventCreateWithFlags(&event, hipEventDisableTiming | hipEventInterprocess));
|
||||
#if HT_AMD
|
||||
// Test disabled for nvidia due to segfault with cuda api
|
||||
SECTION("Get event handle with eventHandle(nullptr)") {
|
||||
@@ -371,8 +366,7 @@ TEST_CASE("Unit_hipIpcEventHandle_ParameterValidation") {
|
||||
HIP_CHECK(hipEventCreateWithFlags(&eventNoIpc, hipEventDisableTiming));
|
||||
|
||||
ret = hipIpcGetEventHandle(&eventHandle, eventNoIpc);
|
||||
if ((ret != hipErrorInvalidResourceHandle) &&
|
||||
(ret != hipErrorInvalidConfiguration)) {
|
||||
if ((ret != hipErrorInvalidResourceHandle) && (ret != hipErrorInvalidConfiguration)) {
|
||||
INFO("Error returned : " << ret);
|
||||
REQUIRE(false);
|
||||
}
|
||||
|
||||
@@ -33,7 +33,7 @@ THE SOFTWARE.
|
||||
* @{
|
||||
* @ingroup DeviceTest
|
||||
* `hipIpcOpenMemHandle(void** devPtr, hipIpcMemHandle_t handle, unsigned int flags)` -
|
||||
* Opens an interprocess memory handle exported from another process
|
||||
* Opens an interprocess memory handle exported from another process
|
||||
* and returns a device pointer usable in the local process.
|
||||
*/
|
||||
|
||||
@@ -48,7 +48,6 @@ typedef struct mem_handle {
|
||||
} hip_ipc_t;
|
||||
|
||||
|
||||
|
||||
// This testcase verifies the hipIpcMemAccess APIs as follows
|
||||
// The following program spawns a child process and does the following
|
||||
// Parent iterate through each device, create memory -- create hipIpcMemhandle
|
||||
@@ -78,7 +77,7 @@ typedef struct mem_handle {
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
|
||||
hip_ipc_t *shrd_mem = NULL;
|
||||
hip_ipc_t* shrd_mem = NULL;
|
||||
pid_t pid;
|
||||
size_t N = 1024;
|
||||
size_t Nbytes = N * sizeof(int);
|
||||
@@ -90,19 +89,16 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
|
||||
std::string cmd_line = "rm -rf /dev/shm/sem.my-sem-object*";
|
||||
int res = system(cmd_line.c_str());
|
||||
REQUIRE(res != -1);
|
||||
sem_ob1 = sem_open("/my-sem-object1", O_CREAT|O_EXCL, 0660, 0);
|
||||
sem_ob2 = sem_open("/my-sem-object2", O_CREAT|O_EXCL, 0660, 0);
|
||||
sem_ob1 = sem_open("/my-sem-object1", O_CREAT | O_EXCL, 0660, 0);
|
||||
sem_ob2 = sem_open("/my-sem-object2", O_CREAT | O_EXCL, 0660, 0);
|
||||
REQUIRE(sem_ob1 != SEM_FAILED);
|
||||
REQUIRE(sem_ob2 != SEM_FAILED);
|
||||
|
||||
shrd_mem = reinterpret_cast<hip_ipc_t *>(mmap(NULL, sizeof(hip_ipc_t),
|
||||
PROT_READ | PROT_WRITE,
|
||||
MAP_SHARED | MAP_ANONYMOUS,
|
||||
0, 0));
|
||||
shrd_mem = reinterpret_cast<hip_ipc_t*>(
|
||||
mmap(NULL, sizeof(hip_ipc_t), PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, 0, 0));
|
||||
REQUIRE(shrd_mem != NULL);
|
||||
shrd_mem->IfTestPassed = true;
|
||||
HipTest::initArrays<int>(nullptr, nullptr, nullptr,
|
||||
&A_h, nullptr, &C_h, N, false);
|
||||
HipTest::initArrays<int>(nullptr, nullptr, nullptr, &A_h, nullptr, &C_h, N, false);
|
||||
pid = fork();
|
||||
if (pid != 0) {
|
||||
// Parent process
|
||||
@@ -111,9 +107,8 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
|
||||
if (shrd_mem->IfTestPassed == true) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
HIP_CHECK(hipMalloc(&A_d, Nbytes));
|
||||
HIP_CHECK(hipIpcGetMemHandle(reinterpret_cast<hipIpcMemHandle_t *>
|
||||
(&shrd_mem->memHandle),
|
||||
A_d));
|
||||
HIP_CHECK(
|
||||
hipIpcGetMemHandle(reinterpret_cast<hipIpcMemHandle_t*>(&shrd_mem->memHandle), A_d));
|
||||
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
|
||||
shrd_mem->device = i;
|
||||
if ((sem_post(sem_ob1)) == -1) {
|
||||
@@ -132,7 +127,7 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
|
||||
// Child process
|
||||
HIP_CHECK(hipGetDeviceCount(&Num_devices));
|
||||
for (int j = 0; j < Num_devices; ++j) {
|
||||
HIP_CHECK(hipSetDevice(j));
|
||||
HIP_CHECK(hipSetDevice(j));
|
||||
if ((sem_wait(sem_ob1)) == -1) {
|
||||
shrd_mem->IfTestPassed = false;
|
||||
WARN("sem_wait() call failed in child process.");
|
||||
@@ -147,8 +142,7 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
|
||||
HIP_CHECK(hipDeviceCanAccessPeer(&CanAccessPeer, i, shrd_mem->device));
|
||||
if (CanAccessPeer == 1) {
|
||||
HIP_CHECK(hipMalloc(&C_d, Nbytes));
|
||||
HIP_CHECK(hipIpcOpenMemHandle(reinterpret_cast<void **>(&B_d),
|
||||
shrd_mem->memHandle,
|
||||
HIP_CHECK(hipIpcOpenMemHandle(reinterpret_cast<void**>(&B_d), shrd_mem->memHandle,
|
||||
hipIpcMemLazyEnablePeerAccess));
|
||||
HIP_CHECK(hipMemcpy(C_d, B_d, Nbytes, hipMemcpyDeviceToDevice));
|
||||
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
|
||||
@@ -157,8 +151,8 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
|
||||
// Checking if the data obtained from Ipc shared memory is consistent
|
||||
HIP_CHECK(hipMemcpy(C_h, B_d, Nbytes, hipMemcpyDeviceToHost));
|
||||
HipTest::checkTest<int>(A_h, C_h, N);
|
||||
HIP_CHECK(hipIpcCloseMemHandle(reinterpret_cast<void*>(B_d)));
|
||||
HIP_CHECK(hipFree(C_d));
|
||||
HIP_CHECK(hipIpcCloseMemHandle(reinterpret_cast<void*>(B_d)));
|
||||
HIP_CHECK(hipFree(C_d));
|
||||
}
|
||||
}
|
||||
if ((sem_post(sem_ob2)) == -1) {
|
||||
@@ -178,8 +172,7 @@ TEST_CASE("Unit_hipIpcMemAccess_Semaphores") {
|
||||
int rFlag = 0;
|
||||
waitpid(pid, &rFlag, 0);
|
||||
REQUIRE(shrd_mem->IfTestPassed == true);
|
||||
HipTest::freeArrays<int>(nullptr, nullptr, nullptr,
|
||||
A_h, nullptr, C_h, false);
|
||||
HipTest::freeArrays<int>(nullptr, nullptr, nullptr, A_h, nullptr, C_h, false);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -246,14 +239,12 @@ TEST_CASE("Unit_hipIpcMemAccess_ParameterValidation") {
|
||||
}
|
||||
|
||||
SECTION("Open mem handle with devptr as nullptr") {
|
||||
ret = hipIpcOpenMemHandle(nullptr, MemHandle,
|
||||
hipIpcMemLazyEnablePeerAccess);
|
||||
ret = hipIpcOpenMemHandle(nullptr, MemHandle, hipIpcMemLazyEnablePeerAccess);
|
||||
REQUIRE(ret == hipErrorInvalidValue);
|
||||
}
|
||||
|
||||
SECTION("Open mem handle with handle as un-initialized") {
|
||||
ret = hipIpcOpenMemHandle(&Ad2, MemHandleUninit,
|
||||
hipIpcMemLazyEnablePeerAccess);
|
||||
ret = hipIpcOpenMemHandle(&Ad2, MemHandleUninit, hipIpcMemLazyEnablePeerAccess);
|
||||
REQUIRE((ret == hipErrorInvalidValue || ret == hipErrorInvalidDevicePointer));
|
||||
}
|
||||
#if HT_AMD
|
||||
|
||||
@@ -106,9 +106,11 @@ static bool validateMemoryOnGPU(int gpu, bool concurOnOneGPU = false) {
|
||||
HIP_CHECK(hipMemGetInfo(&curAvl, &curTot));
|
||||
|
||||
if (!concurOnOneGPU && (prevAvl < curAvl || prevTot != curTot)) {
|
||||
//In concurrent calls on one GPU, we cannot verify leaking in this way
|
||||
printf("%s : Memory allocation mismatch observed."
|
||||
"Possible memory leak.\n", __func__);
|
||||
// In concurrent calls on one GPU, we cannot verify leaking in this way
|
||||
printf(
|
||||
"%s : Memory allocation mismatch observed."
|
||||
"Possible memory leak.\n",
|
||||
__func__);
|
||||
TestPassed &= false;
|
||||
}
|
||||
|
||||
@@ -117,9 +119,8 @@ static bool validateMemoryOnGPU(int gpu, bool concurOnOneGPU = false) {
|
||||
HIP_CHECK(hipMemcpy(A_d, A_h, Nbytes, hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(B_d, B_h, Nbytes, hipMemcpyHostToDevice));
|
||||
|
||||
hipLaunchKernelGGL(HipTest::vectorADD, dim3(blocks), dim3(threadsPerBlock),
|
||||
0, 0, static_cast<const int*>(A_d),
|
||||
static_cast<const int*>(B_d), C_d, N);
|
||||
hipLaunchKernelGGL(HipTest::vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, 0,
|
||||
static_cast<const int*>(A_d), static_cast<const int*>(B_d), C_d, N);
|
||||
HIP_CHECK(hipGetLastError());
|
||||
HIP_CHECK(hipMemcpy(C_h, C_d, Nbytes, hipMemcpyDeviceToHost));
|
||||
|
||||
@@ -189,8 +190,7 @@ TEST_CASE("Unit_hipMalloc_ChildConcurrencyDefaultGpu") {
|
||||
|
||||
// Wait and get result from child
|
||||
pid = wait(&exitStatus);
|
||||
if ((WEXITSTATUS(exitStatus) == resFailure) || (pid < 0))
|
||||
TestPassed = false;
|
||||
if ((WEXITSTATUS(exitStatus) == resFailure) || (pid < 0)) TestPassed = false;
|
||||
}
|
||||
|
||||
REQUIRE(TestPassed == true);
|
||||
|
||||
@@ -41,7 +41,7 @@
|
||||
#include <chrono>
|
||||
#include "../unit/memory/hipSVMCommon.h"
|
||||
|
||||
__global__ void CoherentTst(int *ptr, volatile unsigned int *expired) {
|
||||
__global__ void CoherentTst(int* ptr, volatile unsigned int* expired) {
|
||||
// Incrementing the value by 1
|
||||
atomicAdd_system(ptr, 1);
|
||||
// The following while loop checks the value until expiration.
|
||||
@@ -50,43 +50,42 @@ __global__ void CoherentTst(int *ptr, volatile unsigned int *expired) {
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void SquareKrnl(int *ptr) {
|
||||
__global__ void SquareKrnl(int* ptr) {
|
||||
// ptr value squared here
|
||||
*ptr = (*ptr) * (*ptr);
|
||||
}
|
||||
|
||||
// The function tests the coherency of allocated memory
|
||||
// Return false on failure, true on success.
|
||||
bool static TstCoherency(int *Ptr, bool HmmMem) {
|
||||
bool static TstCoherency(int* Ptr, bool HmmMem) {
|
||||
using namespace std::chrono_literals;
|
||||
int *Dptr = nullptr;
|
||||
int* Dptr = nullptr;
|
||||
hipStream_t strm;
|
||||
HIP_CHECK(hipStreamCreate(&strm));
|
||||
// storing value 1 in the memory created above
|
||||
*Ptr = 1;
|
||||
|
||||
unsigned int *expired = nullptr;
|
||||
HIP_CHECK(hipHostMalloc(&expired, sizeof(unsigned int))); // hipHostMallocCoherent by defaut
|
||||
unsigned int* expired = nullptr;
|
||||
HIP_CHECK(hipHostMalloc(&expired, sizeof(unsigned int))); // hipHostMallocCoherent by defaut
|
||||
*expired = 0;
|
||||
|
||||
if (!HmmMem) {
|
||||
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void **>(&Dptr), Ptr, 0));
|
||||
HIP_CHECK(hipHostGetDevicePointer(reinterpret_cast<void**>(&Dptr), Ptr, 0));
|
||||
CoherentTst<<<1, 1, 0, strm>>>(Dptr, expired);
|
||||
} else {
|
||||
CoherentTst<<<1, 1, 0, strm>>>(Ptr, expired);
|
||||
}
|
||||
// looping until the value is 2 for 3 seconds
|
||||
std::chrono::steady_clock::time_point start =
|
||||
std::chrono::steady_clock::now();
|
||||
while (std::chrono::duration_cast<std::chrono::seconds>(
|
||||
std::chrono::steady_clock::now() - start).count() < 3) {
|
||||
std::chrono::steady_clock::time_point start = std::chrono::steady_clock::now();
|
||||
while (std::chrono::duration_cast<std::chrono::seconds>(std::chrono::steady_clock::now() - start)
|
||||
.count() < 3) {
|
||||
if (*Ptr == 2) {
|
||||
*Ptr += 1;
|
||||
std::this_thread::sleep_for(200ms); // Make sure kernel gets updated Dptr
|
||||
std::this_thread::sleep_for(200ms); // Make sure kernel gets updated Dptr
|
||||
break;
|
||||
}
|
||||
}
|
||||
*expired = 1; // Notify kernel loop to exit
|
||||
*expired = 1; // Notify kernel loop to exit
|
||||
HIP_CHECK(hipStreamSynchronize(strm));
|
||||
HIP_CHECK(hipStreamDestroy(strm));
|
||||
HIP_CHECK(hipHostFree(expired));
|
||||
@@ -106,13 +105,12 @@ TEST_CASE("Unit_malloc_CoherentTst") {
|
||||
CHECK_PCIE_ATOMICS_SUPPORT
|
||||
hipDeviceProp_t prop;
|
||||
HIPCHECK(hipGetDeviceProperties(&prop, 0));
|
||||
char *p = NULL;
|
||||
char* p = NULL;
|
||||
p = strstr(prop.gcnArchName, "xnack+");
|
||||
if (p) {
|
||||
// Test Case execution begins from here
|
||||
int managed = 0;
|
||||
HIPCHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory,
|
||||
0));
|
||||
HIPCHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory, 0));
|
||||
if (managed == 1) {
|
||||
int *Ptr = nullptr, SIZE = sizeof(int);
|
||||
bool HmmMem = true;
|
||||
@@ -122,7 +120,7 @@ TEST_CASE("Unit_malloc_CoherentTst") {
|
||||
auto ret = TstCoherency(Ptr, HmmMem);
|
||||
free(Ptr);
|
||||
REQUIRE(ret);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
HipTest::HIP_SKIP_TEST("GPU is not xnack enabled hence skipping the test...\n");
|
||||
}
|
||||
@@ -137,12 +135,11 @@ TEST_CASE("Unit_malloc_CoherentTst") {
|
||||
TEST_CASE("Unit_malloc_CoherentTstWthAdvise") {
|
||||
hipDeviceProp_t prop;
|
||||
HIPCHECK(hipGetDeviceProperties(&prop, 0));
|
||||
char *p = NULL;
|
||||
char* p = NULL;
|
||||
p = strstr(prop.gcnArchName, "xnack+");
|
||||
if (p) {
|
||||
int managed = 0;
|
||||
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory,
|
||||
0));
|
||||
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory, 0));
|
||||
if (managed == 1) {
|
||||
int *Ptr = nullptr, SIZE = sizeof(int);
|
||||
|
||||
@@ -154,7 +151,7 @@ TEST_CASE("Unit_malloc_CoherentTstWthAdvise") {
|
||||
SquareKrnl<<<1, 1, 0, strm>>>(Ptr);
|
||||
HIP_CHECK(hipStreamSynchronize(strm));
|
||||
HIP_CHECK(hipStreamDestroy(strm));
|
||||
REQUIRE (*Ptr == 16);
|
||||
REQUIRE(*Ptr == 16);
|
||||
}
|
||||
} else {
|
||||
HipTest::HIP_SKIP_TEST("GPU is not xnack enabled hence skipping the test...\n");
|
||||
@@ -170,17 +167,15 @@ TEST_CASE("Unit_mmap_CoherentTst") {
|
||||
CHECK_PCIE_ATOMICS_SUPPORT
|
||||
hipDeviceProp_t prop;
|
||||
HIPCHECK(hipGetDeviceProperties(&prop, 0));
|
||||
char *p = NULL;
|
||||
char* p = NULL;
|
||||
p = strstr(prop.gcnArchName, "xnack+");
|
||||
if (p) {
|
||||
int managed = 0;
|
||||
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory,
|
||||
0));
|
||||
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory, 0));
|
||||
if (managed == 1) {
|
||||
bool HmmMem = true;
|
||||
int *Ptr = reinterpret_cast<int*>(mmap(NULL, sizeof(int),
|
||||
PROT_READ | PROT_WRITE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS, 0, 0));
|
||||
int* Ptr = reinterpret_cast<int*>(
|
||||
mmap(NULL, sizeof(int), PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, 0, 0));
|
||||
if (Ptr == MAP_FAILED) {
|
||||
WARN("Mapping Failed\n");
|
||||
REQUIRE(false);
|
||||
@@ -191,7 +186,7 @@ TEST_CASE("Unit_mmap_CoherentTst") {
|
||||
WARN("munmap failed\n");
|
||||
}
|
||||
REQUIRE(ret);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
HipTest::HIP_SKIP_TEST("GPU is not xnack enabled hence skipping the test...\n");
|
||||
}
|
||||
@@ -205,17 +200,15 @@ TEST_CASE("Unit_mmap_CoherentTst") {
|
||||
TEST_CASE("Unit_mmap_CoherentTstWthAdvise") {
|
||||
hipDeviceProp_t prop;
|
||||
HIPCHECK(hipGetDeviceProperties(&prop, 0));
|
||||
char *p = NULL;
|
||||
char* p = NULL;
|
||||
p = strstr(prop.gcnArchName, "xnack+");
|
||||
if (p) {
|
||||
int managed = 0;
|
||||
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory,
|
||||
0));
|
||||
HIP_CHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeManagedMemory, 0));
|
||||
if (managed == 1) {
|
||||
int SIZE = sizeof(int);
|
||||
int *Ptr = reinterpret_cast<int*>(mmap(NULL, SIZE,
|
||||
PROT_READ | PROT_WRITE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS, 0, 0));
|
||||
int* Ptr = reinterpret_cast<int*>(
|
||||
mmap(NULL, SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, 0, 0));
|
||||
if (Ptr == MAP_FAILED) {
|
||||
WARN("Mapping Failed\n");
|
||||
REQUIRE(false);
|
||||
@@ -230,13 +223,13 @@ TEST_CASE("Unit_mmap_CoherentTstWthAdvise") {
|
||||
bool IfTstPassed = false;
|
||||
if (*Ptr == 81) {
|
||||
IfTstPassed = true;
|
||||
}
|
||||
}
|
||||
int err = munmap(Ptr, SIZE);
|
||||
if (err != 0) {
|
||||
WARN("munmap failed\n");
|
||||
}
|
||||
REQUIRE(IfTstPassed);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
HipTest::HIP_SKIP_TEST("GPU is not xnack enabled hence skipping the test...\n");
|
||||
}
|
||||
@@ -249,8 +242,8 @@ TEST_CASE("Unit_mmap_CoherentTstWthAdvise") {
|
||||
#if HT_AMD
|
||||
TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg1") {
|
||||
if ((setenv("HIP_HOST_COHERENT", "0", 1)) != 0) {
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
}
|
||||
int stat = 0;
|
||||
if (fork() == 0) {
|
||||
@@ -289,8 +282,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg1") {
|
||||
#if HT_AMD
|
||||
TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg2") {
|
||||
if ((setenv("HIP_HOST_COHERENT", "0", 1)) != 0) {
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
}
|
||||
int stat = 0;
|
||||
if (fork() == 0) {
|
||||
@@ -329,8 +322,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg2") {
|
||||
#if HT_AMD
|
||||
TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg3") {
|
||||
if ((setenv("HIP_HOST_COHERENT", "0", 1)) != 0) {
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
}
|
||||
int stat = 0;
|
||||
if (fork() == 0) {
|
||||
@@ -369,8 +362,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg3") {
|
||||
#if HT_AMD
|
||||
TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg4") {
|
||||
if ((setenv("HIP_HOST_COHERENT", "0", 1)) != 0) {
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
}
|
||||
int stat = 0;
|
||||
if (fork() == 0) {
|
||||
@@ -410,8 +403,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv0Flg4") {
|
||||
#if HT_AMD
|
||||
TEST_CASE("Unit_hipHostMalloc_WthEnv1") {
|
||||
if ((setenv("HIP_HOST_COHERENT", "1", 1)) != 0) {
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
}
|
||||
int stat = 0;
|
||||
if (fork() == 0) { // child process
|
||||
@@ -439,8 +432,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv1") {
|
||||
#if HT_AMD
|
||||
TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg1") {
|
||||
if ((setenv("HIP_HOST_COHERENT", "1", 1)) != 0) {
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
}
|
||||
int stat = 0;
|
||||
if (fork() == 0) { // child process
|
||||
@@ -467,8 +460,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg1") {
|
||||
#if HT_AMD
|
||||
TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg2") {
|
||||
if ((setenv("HIP_HOST_COHERENT", "1", 1)) != 0) {
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
}
|
||||
int stat = 0;
|
||||
if (fork() == 0) { // child process
|
||||
@@ -495,8 +488,8 @@ TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg2") {
|
||||
#if HT_AMD
|
||||
TEST_CASE("Unit_hipHostMalloc_WthEnv1Flg3") {
|
||||
if ((setenv("HIP_HOST_COHERENT", "1", 1)) != 0) {
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
WARN("Unable to turn on HIP_HOST_COHERENT, hence terminating the Test case!");
|
||||
REQUIRE(false);
|
||||
}
|
||||
int stat = 0;
|
||||
if (fork() == 0) { // child process
|
||||
|
||||
@@ -24,16 +24,16 @@ THE SOFTWARE.
|
||||
#include <sys/wait.h>
|
||||
#include <sys/types.h>
|
||||
|
||||
#define ReadEnd 0
|
||||
#define ReadEnd 0
|
||||
#define WriteEnd 1
|
||||
#define MAX_SIZE 32
|
||||
#define FREE_MEM_TO_HIDE 4294967296
|
||||
#define SIZE_TO_ALLOCATE 2147483648
|
||||
/*
|
||||
* In main process allocate 2 GB of device memory.
|
||||
* Fork() a child process and verify that 2 GB has been
|
||||
* allocated in parent process.
|
||||
*/
|
||||
* In main process allocate 2 GB of device memory.
|
||||
* Fork() a child process and verify that 2 GB has been
|
||||
* allocated in parent process.
|
||||
*/
|
||||
TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario1") {
|
||||
constexpr size_t size = 2147483648; // 2GB
|
||||
int fd[2], fd1[2], status;
|
||||
@@ -42,10 +42,10 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario1") {
|
||||
status = pipe(fd1);
|
||||
REQUIRE(status == 0);
|
||||
pid_t child_pid;
|
||||
child_pid = fork(); // Create a new child process
|
||||
child_pid = fork(); // Create a new child process
|
||||
if (child_pid < 0) {
|
||||
WARN("Fork failed!!!!");
|
||||
} else if (child_pid == 0) { // child
|
||||
} else if (child_pid == 0) { // child
|
||||
close(fd1[WriteEnd]);
|
||||
close(fd[ReadEnd]);
|
||||
int result;
|
||||
@@ -67,7 +67,7 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario1") {
|
||||
REQUIRE(status != -1);
|
||||
close(fd[WriteEnd]);
|
||||
exit(0);
|
||||
} else { // Parent
|
||||
} else { // Parent
|
||||
close(fd1[ReadEnd]);
|
||||
close(fd[WriteEnd]);
|
||||
// Allocate memory
|
||||
@@ -90,10 +90,10 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario1") {
|
||||
}
|
||||
}
|
||||
/**
|
||||
* From main process Fork() a child process. In the child process allocate
|
||||
* 2 GB of device memory. Signal the parent process. Verify from the parent
|
||||
* process that 2 GB is allocated in the child process.
|
||||
*/
|
||||
* From main process Fork() a child process. In the child process allocate
|
||||
* 2 GB of device memory. Signal the parent process. Verify from the parent
|
||||
* process that 2 GB is allocated in the child process.
|
||||
*/
|
||||
TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
|
||||
constexpr size_t size = 2147483648; // 2GB
|
||||
int fd[2], fd2[2], status;
|
||||
@@ -102,10 +102,10 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
|
||||
status = pipe(fd2);
|
||||
REQUIRE(status == 0);
|
||||
pid_t child_pid;
|
||||
child_pid = fork(); // Create a new child process
|
||||
child_pid = fork(); // Create a new child process
|
||||
if (child_pid < 0) {
|
||||
WARN("Fork failed!!!!");
|
||||
} else if (child_pid == 0) { // Child
|
||||
} else if (child_pid == 0) { // Child
|
||||
close(fd[ReadEnd]);
|
||||
close(fd2[WriteEnd]);
|
||||
// Allocate memory
|
||||
@@ -124,7 +124,7 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
|
||||
// Free allocated device memory
|
||||
HIP_CHECK(hipFree(A_d));
|
||||
exit(0);
|
||||
} else { // Parent
|
||||
} else { // Parent
|
||||
size_t free = 0, total = 0;
|
||||
close(fd[WriteEnd]);
|
||||
close(fd2[ReadEnd]);
|
||||
@@ -134,7 +134,7 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
|
||||
REQUIRE(status != -1);
|
||||
close(fd[ReadEnd]);
|
||||
// Verify the memory
|
||||
HIP_CHECK(hipMemGetInfo(&free , &total));
|
||||
HIP_CHECK(hipMemGetInfo(&free, &total));
|
||||
REQUIRE((total - free) >= size);
|
||||
// Signal child that validation is over and child can free memory
|
||||
int valid = 0;
|
||||
@@ -146,21 +146,21 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario2") {
|
||||
}
|
||||
}
|
||||
/*
|
||||
* From main process Fork() a child process. In the child process
|
||||
* allocate 2 GB of device memory. Free the memory and exit from
|
||||
* child process. Verify from the parent process that 2 GB is
|
||||
* freed in the child process.
|
||||
*/
|
||||
* From main process Fork() a child process. In the child process
|
||||
* allocate 2 GB of device memory. Free the memory and exit from
|
||||
* child process. Verify from the parent process that 2 GB is
|
||||
* freed in the child process.
|
||||
*/
|
||||
TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario3") {
|
||||
constexpr size_t size = 2147483648; // 2GB
|
||||
int fd[2], status;
|
||||
status = pipe(fd);
|
||||
REQUIRE(status == 0);
|
||||
pid_t child_pid;
|
||||
child_pid = fork(); // Create a new child process
|
||||
child_pid = fork(); // Create a new child process
|
||||
if (child_pid < 0) {
|
||||
WARN("Fork failed!!!!");
|
||||
} else if (child_pid == 0) { // Child
|
||||
} else if (child_pid == 0) { // Child
|
||||
close(fd[ReadEnd]);
|
||||
// Allocate the memory
|
||||
void* A_d = nullptr;
|
||||
@@ -173,7 +173,7 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario3") {
|
||||
REQUIRE(status != -1);
|
||||
close(fd[WriteEnd]);
|
||||
exit(0);
|
||||
} else { // Parent
|
||||
} else { // Parent
|
||||
close(fd[WriteEnd]);
|
||||
// Wait for the signal from child about memory free
|
||||
int check_parent;
|
||||
@@ -182,43 +182,43 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_Scenario3") {
|
||||
close(fd[ReadEnd]);
|
||||
size_t free = 0, total = 0;
|
||||
// Verify the memory
|
||||
HIP_CHECK(hipMemGetInfo(&free , &total));
|
||||
HIP_CHECK(hipMemGetInfo(&free, &total));
|
||||
REQUIRE((total - free) >= 0);
|
||||
// wait for child exit
|
||||
wait(NULL);
|
||||
}
|
||||
}
|
||||
/*
|
||||
* From main process Fork() a child process. In the child process allocate
|
||||
* 2 GB of device memory. Exit from child process. Verify from the parent
|
||||
* process that 2 GB is freed in the child process.
|
||||
*/
|
||||
* From main process Fork() a child process. In the child process allocate
|
||||
* 2 GB of device memory. Exit from child process. Verify from the parent
|
||||
* process that 2 GB is freed in the child process.
|
||||
*/
|
||||
TEST_CASE("Unit_hipMemGetInfo_Functional_scenario4") {
|
||||
constexpr size_t size = 2147483648; // 2GB
|
||||
pid_t child_pid;
|
||||
child_pid = fork(); // Create a new child process
|
||||
child_pid = fork(); // Create a new child process
|
||||
if (child_pid < 0) {
|
||||
WARN("Fork failed!!!!");
|
||||
} else if (child_pid == 0) { // Child
|
||||
} else if (child_pid == 0) { // Child
|
||||
// Allocate the memory
|
||||
void* A_d = nullptr;
|
||||
HIP_CHECK(hipMalloc(&A_d, size));
|
||||
exit(0);
|
||||
} else { // Parent
|
||||
} else { // Parent
|
||||
// wait for child exit
|
||||
wait(NULL);
|
||||
size_t free = 0, total = 0;
|
||||
// Verify the memory
|
||||
HIP_CHECK(hipMemGetInfo(&free , &total));
|
||||
REQUIRE((total-free) >= 0);
|
||||
HIP_CHECK(hipMemGetInfo(&free, &total));
|
||||
REQUIRE((total - free) >= 0);
|
||||
}
|
||||
}
|
||||
/*
|
||||
* Multidevice Scenario: In main process allocate 2 GB of device memory
|
||||
* in every device. Verify that 2 GB is allocated using hipMemGetInfo.
|
||||
* Fork() a child process and verify that 2 GB has been allocated from
|
||||
* parent process in every device.
|
||||
*/
|
||||
* Multidevice Scenario: In main process allocate 2 GB of device memory
|
||||
* in every device. Verify that 2 GB is allocated using hipMemGetInfo.
|
||||
* Fork() a child process and verify that 2 GB has been allocated from
|
||||
* parent process in every device.
|
||||
*/
|
||||
TEST_CASE("Unit_hipMemGetInfo_Functional_MultiDevice_Scenario5") {
|
||||
constexpr size_t size = 2147483648; // 2GB
|
||||
size_t free = 0, total = 0;
|
||||
@@ -228,29 +228,29 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_MultiDevice_Scenario5") {
|
||||
status = pipe(fd2);
|
||||
REQUIRE(status == 0);
|
||||
pid_t child_pid;
|
||||
child_pid = fork(); // Create a new child process
|
||||
child_pid = fork(); // Create a new child process
|
||||
if (child_pid < 0) {
|
||||
WARN("Fork failed!!!!");
|
||||
} else if (child_pid == 0) { // Child
|
||||
close(fd1[WriteEnd]);
|
||||
close(fd2[ReadEnd]);
|
||||
// Wait for the signal from parent after memory allocatoin
|
||||
int check_child;
|
||||
status = read(fd1[ReadEnd], &check_child, sizeof(check_child));
|
||||
REQUIRE(status != -1);
|
||||
close(fd1[ReadEnd]);
|
||||
int num_devices, result, count = 0;
|
||||
// Get the device count
|
||||
HIP_CHECK(hipGetDeviceCount(&num_devices));
|
||||
for (int i = 0; i < num_devices; i++) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
// Check the memory
|
||||
HIP_CHECK(hipMemGetInfo(&free , &total));
|
||||
if ((total - free) >= size) {
|
||||
count+=1;
|
||||
}
|
||||
} else if (child_pid == 0) { // Child
|
||||
close(fd1[WriteEnd]);
|
||||
close(fd2[ReadEnd]);
|
||||
// Wait for the signal from parent after memory allocatoin
|
||||
int check_child;
|
||||
status = read(fd1[ReadEnd], &check_child, sizeof(check_child));
|
||||
REQUIRE(status != -1);
|
||||
close(fd1[ReadEnd]);
|
||||
int num_devices, result, count = 0;
|
||||
// Get the device count
|
||||
HIP_CHECK(hipGetDeviceCount(&num_devices));
|
||||
for (int i = 0; i < num_devices; i++) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
// Check the memory
|
||||
HIP_CHECK(hipMemGetInfo(&free, &total));
|
||||
if ((total - free) >= size) {
|
||||
count += 1;
|
||||
}
|
||||
}
|
||||
if ( count == num_devices ) {
|
||||
if (count == num_devices) {
|
||||
result = 1;
|
||||
} else {
|
||||
result = 0;
|
||||
@@ -260,21 +260,21 @@ TEST_CASE("Unit_hipMemGetInfo_Functional_MultiDevice_Scenario5") {
|
||||
REQUIRE(status != -1);
|
||||
close(fd2[WriteEnd]);
|
||||
exit(0);
|
||||
} else { // Parent
|
||||
} else { // Parent
|
||||
close(fd1[ReadEnd]);
|
||||
close(fd2[WriteEnd]);
|
||||
int num_devices;
|
||||
// Get the device count
|
||||
HIP_CHECK(hipGetDeviceCount(&num_devices));
|
||||
std::vector<void*>v(num_devices, nullptr);
|
||||
std::vector<void*> v(num_devices, nullptr);
|
||||
for (int i = 0; i < num_devices; i++) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
// verify the memory
|
||||
HIP_CHECK(hipMemGetInfo(&free , &total));
|
||||
HIP_CHECK(hipMemGetInfo(&free, &total));
|
||||
// Allocate memory
|
||||
HIP_CHECK(hipMalloc(&v[i], size));
|
||||
// Verify the memory
|
||||
HIP_CHECK(hipMemGetInfo(&free , &total));
|
||||
HIP_CHECK(hipMemGetInfo(&free, &total));
|
||||
}
|
||||
// Signal the child about memory allocation
|
||||
int check = 0;
|
||||
@@ -310,7 +310,7 @@ static bool testHiddenFreeMemFromChild() {
|
||||
size_t free = 0, total = 0, min_size = 0;
|
||||
close(fd_c2p[ReadEnd]);
|
||||
close(fd_p2c[WriteEnd]);
|
||||
int64_t size_tohide = (FREE_MEM_TO_HIDE/(1024*1024)); // in MB
|
||||
int64_t size_tohide = (FREE_MEM_TO_HIDE / (1024 * 1024)); // in MB
|
||||
// set environment variable from shell
|
||||
unsetenv("HIP_HIDDEN_FREE_MEM");
|
||||
setenv("HIP_HIDDEN_FREE_MEM", std::to_string(size_tohide).c_str(), 1);
|
||||
@@ -377,7 +377,7 @@ TEST_CASE("Unit_hipMemGetInfo_SetHiddenFreeMemFromChild") {
|
||||
*/
|
||||
TEST_CASE("Unit_hipMemGetInfo_VerifyHiddenFreeMemForAllGpu") {
|
||||
int numDevices = 0;
|
||||
int64_t size_tohide = (FREE_MEM_TO_HIDE/(1024*1024)); // in MB
|
||||
int64_t size_tohide = (FREE_MEM_TO_HIDE / (1024 * 1024)); // in MB
|
||||
// set environment variable from shell
|
||||
unsetenv("HIP_HIDDEN_FREE_MEM");
|
||||
setenv("HIP_HIDDEN_FREE_MEM", std::to_string(size_tohide).c_str(), 1);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -36,7 +36,7 @@
|
||||
/**
|
||||
* Fetches Gpu device count
|
||||
*/
|
||||
static void getDeviceCount(int *pdevCnt) {
|
||||
static void getDeviceCount(int* pdevCnt) {
|
||||
int fd[2], val = 0;
|
||||
pid_t childpid;
|
||||
|
||||
@@ -82,8 +82,7 @@ static void getDeviceCount(int *pdevCnt) {
|
||||
|
||||
|
||||
// Pass either -1 in deviceNumber or invalid device number
|
||||
static void testInvalidDevice(int numDevices, bool useRocrEnv,
|
||||
int deviceNumber) {
|
||||
static void testInvalidDevice(int numDevices, bool useRocrEnv, int deviceNumber) {
|
||||
bool testResult = true;
|
||||
int device;
|
||||
int tempCount = 0;
|
||||
@@ -117,24 +116,24 @@ static void testInvalidDevice(int numDevices, bool useRocrEnv,
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
err = hipSetDevice(i);
|
||||
if (err != hipSuccess) {
|
||||
setDeviceErrorCheck+= 1;
|
||||
setDeviceErrorCheck += 1;
|
||||
}
|
||||
|
||||
err = hipGetDevice(&device);
|
||||
if (err != hipSuccess) {
|
||||
getDeviceErrorCheck+= 1;
|
||||
getDeviceErrorCheck += 1;
|
||||
}
|
||||
}
|
||||
|
||||
if ((getDeviceCountErrorCheck == 1) && (setDeviceErrorCheck == numDevices)
|
||||
&& (getDeviceErrorCheck == numDevices)) {
|
||||
if ((getDeviceCountErrorCheck == 1) && (setDeviceErrorCheck == numDevices) &&
|
||||
(getDeviceErrorCheck == numDevices)) {
|
||||
testResult = true;
|
||||
|
||||
} else {
|
||||
printf("Test failed for invalid device, getDeviceCountErrorCheck %d,"
|
||||
"setDeviceErrorCheck %d, getDeviceErrorCheck %d\n",
|
||||
getDeviceCountErrorCheck, setDeviceErrorCheck,
|
||||
getDeviceErrorCheck);
|
||||
printf(
|
||||
"Test failed for invalid device, getDeviceCountErrorCheck %d,"
|
||||
"setDeviceErrorCheck %d, getDeviceErrorCheck %d\n",
|
||||
getDeviceCountErrorCheck, setDeviceErrorCheck, getDeviceErrorCheck);
|
||||
|
||||
testResult = false;
|
||||
}
|
||||
@@ -159,19 +158,18 @@ static void testInvalidDevice(int numDevices, bool useRocrEnv,
|
||||
}
|
||||
|
||||
|
||||
static void testValidDevices(int numDevices, bool useRocrEnv, int *deviceList,
|
||||
int deviceListLength) {
|
||||
static void testValidDevices(int numDevices, bool useRocrEnv, int* deviceList,
|
||||
int deviceListLength) {
|
||||
bool testResult = true;
|
||||
int tempCount = 0;
|
||||
int device;
|
||||
int setDeviceErrorCheck = 0;
|
||||
int getDeviceErrorCheck = 0;
|
||||
int getDeviceCountErrorCheck = 0;
|
||||
int *deviceListPtr = deviceList;
|
||||
int* deviceListPtr = deviceList;
|
||||
std::string visibleDeviceString;
|
||||
|
||||
if ((NULL == deviceList) || ((deviceListLength < 1) ||
|
||||
deviceListLength > numDevices)) {
|
||||
if ((NULL == deviceList) || ((deviceListLength < 1) || deviceListLength > numDevices)) {
|
||||
INFO("Invalid argument for number of devices. Skipping current test");
|
||||
REQUIRE(false);
|
||||
}
|
||||
@@ -213,17 +211,17 @@ static void testValidDevices(int numDevices, bool useRocrEnv, int *deviceList,
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
err = hipSetDevice(i);
|
||||
if (err != hipSuccess) {
|
||||
setDeviceErrorCheck+= 1;
|
||||
setDeviceErrorCheck += 1;
|
||||
}
|
||||
|
||||
err = hipGetDevice(&device);
|
||||
if (err != hipSuccess) {
|
||||
getDeviceErrorCheck+= 1;
|
||||
getDeviceErrorCheck += 1;
|
||||
}
|
||||
}
|
||||
|
||||
if ((getDeviceCountErrorCheck == 1) && (setDeviceErrorCheck ==
|
||||
(numDevices-deviceListLength)) && (getDeviceErrorCheck == 0)) {
|
||||
if ((getDeviceCountErrorCheck == 1) &&
|
||||
(setDeviceErrorCheck == (numDevices - deviceListLength)) && (getDeviceErrorCheck == 0)) {
|
||||
testResult = true;
|
||||
|
||||
} else {
|
||||
@@ -251,19 +249,19 @@ static void testValidDevices(int numDevices, bool useRocrEnv, int *deviceList,
|
||||
}
|
||||
|
||||
|
||||
static void Initialize(int *deviceList, int numDevices, int count,
|
||||
std::string& min_visibleDeviceString, std::string& max_visibleDeviceString) {
|
||||
int *deviceListPtr = deviceList;
|
||||
for (int i =0; i < count; i++) {
|
||||
if (i == count-1) {
|
||||
static void Initialize(int* deviceList, int numDevices, int count,
|
||||
std::string& min_visibleDeviceString, std::string& max_visibleDeviceString) {
|
||||
int* deviceListPtr = deviceList;
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (i == count - 1) {
|
||||
min_visibleDeviceString.append(std::to_string(*deviceListPtr++));
|
||||
} else {
|
||||
min_visibleDeviceString.append(std::to_string(*deviceListPtr++) + ",");
|
||||
}
|
||||
}
|
||||
|
||||
for (int i =0; i < numDevices; i++) {
|
||||
if (i == numDevices-1) {
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
if (i == numDevices - 1) {
|
||||
max_visibleDeviceString.append(std::to_string(i));
|
||||
} else {
|
||||
max_visibleDeviceString.append(std::to_string(i) + ",");
|
||||
@@ -271,7 +269,7 @@ static void Initialize(int *deviceList, int numDevices, int count,
|
||||
}
|
||||
}
|
||||
|
||||
static void testMaxRvdMinHvd(int numDevices, int *deviceList, int count) {
|
||||
static void testMaxRvdMinHvd(int numDevices, int* deviceList, int count) {
|
||||
bool testResult = true;
|
||||
int device;
|
||||
int validateCount = 0;
|
||||
@@ -282,8 +280,7 @@ static void testMaxRvdMinHvd(int numDevices, int *deviceList, int count) {
|
||||
pid_t cPid;
|
||||
cPid = fork();
|
||||
if (cPid == 0) { // child
|
||||
Initialize(deviceList, numDevices,
|
||||
count, min_visibleDeviceString, max_visibleDeviceString);
|
||||
Initialize(deviceList, numDevices, count, min_visibleDeviceString, max_visibleDeviceString);
|
||||
unsetenv("ROCR_VISIBLE_DEVICES");
|
||||
unsetenv("HIP_VISIBLE_DEVICES");
|
||||
setenv("ROCR_VISIBLE_DEVICES", max_visibleDeviceString.c_str(), 1);
|
||||
@@ -293,7 +290,7 @@ static void testMaxRvdMinHvd(int numDevices, int *deviceList, int count) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
HIP_CHECK(hipGetDevice(&device));
|
||||
if (device == i) {
|
||||
validateCount+= 1;
|
||||
validateCount += 1;
|
||||
}
|
||||
}
|
||||
if (count != validateCount) {
|
||||
@@ -312,19 +309,19 @@ static void testMaxRvdMinHvd(int numDevices, int *deviceList, int count) {
|
||||
REQUIRE(testResult == true);
|
||||
}
|
||||
|
||||
static void testRvdCvd(int numDevices, int *deviceList, int count) {
|
||||
static void testRvdCvd(int numDevices, int* deviceList, int count) {
|
||||
bool testResult = true;
|
||||
int device;
|
||||
int validateCount = 0;
|
||||
std::string min_visibleDeviceString;
|
||||
std::string max_visibleDeviceString;;
|
||||
std::string max_visibleDeviceString;
|
||||
;
|
||||
int fd[2];
|
||||
pipe(fd);
|
||||
pid_t cPid;
|
||||
cPid = fork();
|
||||
if (cPid == 0) { // child
|
||||
Initialize(deviceList, numDevices, count,
|
||||
min_visibleDeviceString, max_visibleDeviceString);
|
||||
Initialize(deviceList, numDevices, count, min_visibleDeviceString, max_visibleDeviceString);
|
||||
unsetenv("ROCR_VISIBLE_DEVICES");
|
||||
unsetenv("HIP_VISIBLE_DEVICES");
|
||||
setenv("ROCR_VISIBLE_DEVICES", max_visibleDeviceString.c_str(), 1);
|
||||
@@ -334,7 +331,7 @@ static void testRvdCvd(int numDevices, int *deviceList, int count) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
HIP_CHECK(hipGetDevice(&device));
|
||||
if (device == i) {
|
||||
validateCount+= 1;
|
||||
validateCount += 1;
|
||||
}
|
||||
}
|
||||
if (count != validateCount) {
|
||||
@@ -353,7 +350,7 @@ static void testRvdCvd(int numDevices, int *deviceList, int count) {
|
||||
REQUIRE(testResult == true);
|
||||
}
|
||||
|
||||
static void testMinRvdMaxHvd(int numDevices, int *deviceList, int count) {
|
||||
static void testMinRvdMaxHvd(int numDevices, int* deviceList, int count) {
|
||||
bool testResult = true;
|
||||
int device;
|
||||
int validateCount = 0;
|
||||
@@ -364,8 +361,7 @@ static void testMinRvdMaxHvd(int numDevices, int *deviceList, int count) {
|
||||
pid_t cPid;
|
||||
cPid = fork();
|
||||
if (cPid == 0) { // child
|
||||
Initialize(deviceList, numDevices, count,
|
||||
min_visibleDeviceString, max_visibleDeviceString);
|
||||
Initialize(deviceList, numDevices, count, min_visibleDeviceString, max_visibleDeviceString);
|
||||
unsetenv("ROCR_VISIBLE_DEVICES");
|
||||
unsetenv("HIP_VISIBLE_DEVICES");
|
||||
setenv("ROCR_VISIBLE_DEVICES", min_visibleDeviceString.c_str(), 1);
|
||||
@@ -375,7 +371,7 @@ static void testMinRvdMaxHvd(int numDevices, int *deviceList, int count) {
|
||||
HIP_CHECK(hipSetDevice(i));
|
||||
HIP_CHECK(hipGetDevice(&device));
|
||||
if (device == i) {
|
||||
validateCount+= 1;
|
||||
validateCount += 1;
|
||||
}
|
||||
}
|
||||
if (count != validateCount) {
|
||||
@@ -407,17 +403,13 @@ TEST_CASE("Unit_hipSetDevice_InvalidVisibleDeviceList") {
|
||||
getDeviceCount(&numDevices);
|
||||
REQUIRE(numDevices != 0);
|
||||
|
||||
SECTION("Test setting -1 to HIP_VISIBLE_DEVICES") {
|
||||
testInvalidDevice(numDevices, false, -1);
|
||||
}
|
||||
SECTION("Test setting -1 to HIP_VISIBLE_DEVICES") { testInvalidDevice(numDevices, false, -1); }
|
||||
|
||||
SECTION("Test setting invalid device to HIP_VISIBLE_DEVICES") {
|
||||
testInvalidDevice(numDevices, false, numDevices);
|
||||
}
|
||||
#ifndef __HIP_PLATFORM_NVIDIA__
|
||||
SECTION("Test setting -1 to ROCR_VISIBLE_DEVICES") {
|
||||
testInvalidDevice(numDevices, true, -1);
|
||||
}
|
||||
SECTION("Test setting -1 to ROCR_VISIBLE_DEVICES") { testInvalidDevice(numDevices, true, -1); }
|
||||
|
||||
SECTION("Test setting invalid device to ROCR_VISIBLE_DEVICES") {
|
||||
testInvalidDevice(numDevices, true, numDevices);
|
||||
@@ -462,16 +454,14 @@ TEST_CASE("Unit_hipSetDevice_SubsetOfAvailableDevices") {
|
||||
REQUIRE(numDevices != 0);
|
||||
|
||||
// Test for subset of available gpus
|
||||
for (int i=0; i < deviceListLength; i++) {
|
||||
deviceList[i] = deviceListLength-1-i;
|
||||
for (int i = 0; i < deviceListLength; i++) {
|
||||
deviceList[i] = deviceListLength - 1 - i;
|
||||
}
|
||||
|
||||
#ifndef __HIP_PLATFORM_NVIDIA__
|
||||
testValidDevices(numDevices, true, deviceList,
|
||||
deviceListLength);
|
||||
testValidDevices(numDevices, true, deviceList, deviceListLength);
|
||||
#endif
|
||||
testValidDevices(numDevices, false, deviceList,
|
||||
deviceListLength);
|
||||
testValidDevices(numDevices, false, deviceList, deviceListLength);
|
||||
}
|
||||
|
||||
#ifndef __HIP_PLATFORM_NVIDIA__
|
||||
@@ -494,8 +484,8 @@ TEST_CASE("Unit_hipSetDevice_MinRvdMaxHvdDevicesList") {
|
||||
deviceList.push_back(0);
|
||||
count = 1;
|
||||
} else {
|
||||
for (int i=0; i < numDevices; i++) {
|
||||
if (i%2 == 0) {
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
if (i % 2 == 0) {
|
||||
deviceList.push_back(i);
|
||||
count++;
|
||||
}
|
||||
@@ -520,8 +510,8 @@ TEST_CASE("Unit_hipSetDevice_MaxRvdMinHvdDevicesList") {
|
||||
if (numDevices == 1) {
|
||||
deviceList.push_back(0);
|
||||
} else {
|
||||
for (int i=0; i < numDevices; i++) {
|
||||
if (i%2 == 0) {
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
if (i % 2 == 0) {
|
||||
deviceList.push_back(i);
|
||||
}
|
||||
}
|
||||
@@ -546,8 +536,8 @@ TEST_CASE("Unit_hipSetDevice_RvdCvdDevicesList") {
|
||||
deviceList[0] = 0;
|
||||
count = 1;
|
||||
} else {
|
||||
for (int i=0; i < numDevices; i++) {
|
||||
if (i%2 == 0) {
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
if (i % 2 == 0) {
|
||||
deviceList[count] = i;
|
||||
count++;
|
||||
}
|
||||
|
||||
@@ -56,6 +56,6 @@ TEST_CASE("Performance_hipEventCreate") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -83,6 +83,6 @@ TEST_CASE("Performance_hipEventCreateWithFlags") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -21,15 +21,15 @@ THE SOFTWARE.
|
||||
#include <hip_test_checkers.hh>
|
||||
#include <hip/hip_ext.h>
|
||||
#include <hip_array_common.hh>
|
||||
#define THREADS_PER_BLOCK 64 // 64 threads per wave on Mi300, 32 threads per wave on Navi31
|
||||
#define MAXITERS 100000 // maximum iteration number in a thread
|
||||
#define FUNC1(K, I) (K + K * I / 37 - I) / 1091; // Shared by kernel and host
|
||||
#define THREADS_PER_BLOCK 64 // 64 threads per wave on Mi300, 32 threads per wave on Navi31
|
||||
#define MAXITERS 100000 // maximum iteration number in a thread
|
||||
#define FUNC1(K, I) (K + K * I / 37 - I) / 1091; // Shared by kernel and host
|
||||
|
||||
using namespace std;
|
||||
constexpr int w1 = 34;
|
||||
|
||||
//#define SHOW_DETAILS // Show statics details
|
||||
//#define VERIFY // Verify data
|
||||
// #define SHOW_DETAILS // Show statics details
|
||||
// #define VERIFY // Verify data
|
||||
|
||||
/**
|
||||
* @addtogroup kernelLaunch clock
|
||||
@@ -37,8 +37,8 @@ constexpr int w1 = 34;
|
||||
* @ingroup PerformanceTest
|
||||
* Contains unit tests for clock, clock64, wall_clock64 and hipExtLaunchKernelGGL APIs
|
||||
*/
|
||||
__global__ void
|
||||
kernel1(uint64_t* out, size_t maxIter, uint64_t* clockCount, uint64_t* wallClockCount) {
|
||||
__global__ void kernel1(uint64_t* out, size_t maxIter, uint64_t* clockCount,
|
||||
uint64_t* wallClockCount) {
|
||||
uint64_t wallClock = wall_clock64();
|
||||
uint64_t clock = clock64();
|
||||
size_t tid = hipBlockDim_x * hipBlockIdx_x + hipThreadIdx_x;
|
||||
@@ -47,8 +47,8 @@ kernel1(uint64_t* out, size_t maxIter, uint64_t* clockCount, uint64_t* wallClock
|
||||
k = FUNC1(k, i);
|
||||
}
|
||||
out[tid] = k;
|
||||
clockCount[tid] = clock64() - clock; // GPU cycle count
|
||||
wallClockCount[tid] = wall_clock64() - wallClock; // Wall clock cycle count
|
||||
clockCount[tid] = clock64() - clock; // GPU cycle count
|
||||
wallClockCount[tid] = wall_clock64() - wallClock; // Wall clock cycle count
|
||||
}
|
||||
|
||||
#ifdef VERIFY
|
||||
@@ -66,17 +66,17 @@ static void host1(uint64_t* out, size_t maxIter, size_t totalThreadsSize) {
|
||||
/*
|
||||
* Roughly query the variable gpu frequency
|
||||
*/
|
||||
static bool query_gpu_frequency(
|
||||
void (*kernel)(uint64_t* out, size_t maxIter, uint64_t* clockCount, uint64_t* wallClockCount),
|
||||
const int wall_clock_rate, const uint32_t blocksMax, const uint32_t blockSizeMax,
|
||||
const int CUs) {
|
||||
static bool query_gpu_frequency(void (*kernel)(uint64_t* out, size_t maxIter, uint64_t* clockCount,
|
||||
uint64_t* wallClockCount),
|
||||
const int wall_clock_rate, const uint32_t blocksMax,
|
||||
const uint32_t blockSizeMax, const int CUs) {
|
||||
hipStream_t stream;
|
||||
hipEvent_t start_event, end_event;
|
||||
|
||||
const size_t totalThreadsSize = static_cast<size_t>(blockSizeMax) * blocksMax;
|
||||
const size_t totalBytesSize = totalThreadsSize * sizeof(uint64_t);
|
||||
const size_t maxIter = MAXITERS;
|
||||
uint64_t* out; // Data to verify kernel rightness
|
||||
uint64_t* out; // Data to verify kernel rightness
|
||||
uint64_t* clockCount;
|
||||
uint64_t* wallClockCount;
|
||||
|
||||
@@ -98,58 +98,52 @@ static bool query_gpu_frequency(
|
||||
#endif
|
||||
for (uint32_t blocks = 1; blocks <= blocksMax; blocks++) {
|
||||
for (uint32_t blockSize = THREADS_PER_BLOCK; blockSize <= blockSizeMax; blockSize *= 2) {
|
||||
hipExtLaunchKernelGGL(kernel, dim3(blocks), dim3(blockSize), 0, stream, start_event,
|
||||
end_event, 0, out, maxIter, clockCount, wallClockCount);
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
float totalGpuTime = 0; // Total GPU time
|
||||
HIP_CHECK(hipEventElapsedTime(&totalGpuTime, start_event, end_event));
|
||||
hipExtLaunchKernelGGL(kernel, dim3(blocks), dim3(blockSize), 0, stream, start_event,
|
||||
end_event, 0, out, maxIter, clockCount, wallClockCount);
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
float totalGpuTime = 0; // Total GPU time
|
||||
HIP_CHECK(hipEventElapsedTime(&totalGpuTime, start_event, end_event));
|
||||
|
||||
const size_t curThreadsSize = static_cast<size_t>(blockSize) * blocks;
|
||||
const size_t curBytesSize = curThreadsSize * sizeof(uint64_t);
|
||||
HIP_CHECK(hipMemcpy(hostClockCount.data(), clockCount, curBytesSize,
|
||||
hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipMemcpy(hostWallClockCount.data(), wallClockCount, curBytesSize,
|
||||
hipMemcpyDeviceToHost));
|
||||
const size_t curThreadsSize = static_cast<size_t>(blockSize) * blocks;
|
||||
const size_t curBytesSize = curThreadsSize * sizeof(uint64_t);
|
||||
HIP_CHECK(hipMemcpy(hostClockCount.data(), clockCount, curBytesSize, hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipMemcpy(hostWallClockCount.data(), wallClockCount, curBytesSize,
|
||||
hipMemcpyDeviceToHost));
|
||||
|
||||
double clockMean = 0;
|
||||
double wallClockMean = 0;
|
||||
#ifdef SHOW_DETAILS
|
||||
double clockDeviation = 0;
|
||||
double wallClockDeviation = 0;
|
||||
getStatics(hostClockCount.data(), curThreadsSize, clockMean, &clockDeviation);
|
||||
getStatics(hostWallClockCount.data(), curThreadsSize, wallClockMean, &wallClockDeviation);
|
||||
#else
|
||||
getStatics(hostClockCount.data(), curThreadsSize, clockMean);
|
||||
getStatics(hostWallClockCount.data(), curThreadsSize, wallClockMean);
|
||||
#endif
|
||||
double clockMean = 0;
|
||||
double wallClockMean = 0;
|
||||
#ifdef SHOW_DETAILS
|
||||
double clockDeviation = 0;
|
||||
double wallClockDeviation = 0;
|
||||
getStatics(hostClockCount.data(), curThreadsSize, clockMean, &clockDeviation);
|
||||
getStatics(hostWallClockCount.data(), curThreadsSize, wallClockMean, &wallClockDeviation);
|
||||
#else
|
||||
getStatics(hostClockCount.data(), curThreadsSize, clockMean);
|
||||
getStatics(hostWallClockCount.data(), curThreadsSize, wallClockMean);
|
||||
#endif
|
||||
|
||||
double aveGpuTime = wallClockMean / wall_clock_rate; // in ms, should be < totalGpuTime
|
||||
double avgGpuFrequency = clockMean / aveGpuTime; // in KHz
|
||||
double aveGpuTime = wallClockMean / wall_clock_rate; // in ms, should be < totalGpuTime
|
||||
double avgGpuFrequency = clockMean / aveGpuTime; // in KHz
|
||||
|
||||
cout <<
|
||||
setw(8) << blocks <<
|
||||
setw(11) << blockSize <<
|
||||
setw(22) << fixed << setprecision(3) << avgGpuFrequency / 1000. <<
|
||||
setw(20) << fixed << setprecision(3) << aveGpuTime <<
|
||||
setw(20) << fixed << setprecision(3) << totalGpuTime <<
|
||||
setw(26) << fixed << setprecision(6) << curBytesSize / totalGpuTime / 1000. <<
|
||||
setw(31) << fixed << setprecision(6) << curBytesSize / totalGpuTime / 1000. / CUs;
|
||||
cout << setw(8) << blocks << setw(11) << blockSize << setw(22) << fixed << setprecision(3)
|
||||
<< avgGpuFrequency / 1000. << setw(20) << fixed << setprecision(3) << aveGpuTime
|
||||
<< setw(20) << fixed << setprecision(3) << totalGpuTime << setw(26) << fixed
|
||||
<< setprecision(6) << curBytesSize / totalGpuTime / 1000. << setw(31) << fixed
|
||||
<< setprecision(6) << curBytesSize / totalGpuTime / 1000. / CUs;
|
||||
|
||||
#ifdef SHOW_DETAILS
|
||||
cout <<
|
||||
setw(15) << fixed << setprecision(3) << wallClockMean <<
|
||||
setw(15) << fixed << setprecision(3) << wallClockDeviation <<
|
||||
setw(15) << fixed << setprecision(3) << clockMean <<
|
||||
setw(15) << fixed << setprecision(3) << clockDeviation;
|
||||
#endif
|
||||
cout << endl;
|
||||
#ifdef SHOW_DETAILS
|
||||
cout << setw(15) << fixed << setprecision(3) << wallClockMean << setw(15) << fixed
|
||||
<< setprecision(3) << wallClockDeviation << setw(15) << fixed << setprecision(3)
|
||||
<< clockMean << setw(15) << fixed << setprecision(3) << clockDeviation;
|
||||
#endif
|
||||
cout << endl;
|
||||
|
||||
#ifdef VERIFY
|
||||
HIP_CHECK(hipMemcpy(hostOut.data(), out, curBytesSize, hipMemcpyDeviceToHost));
|
||||
host1(hostOutExpected.data(), maxIter, curThreadsSize);
|
||||
verified = verify(hostOutExpected.data(), hostOut.data(), curThreadsSize);
|
||||
HIP_CHECK(hipMemcpy(hostOut.data(), out, curBytesSize, hipMemcpyDeviceToHost));
|
||||
host1(hostOutExpected.data(), maxIter, curThreadsSize);
|
||||
verified = verify(hostOutExpected.data(), hostOut.data(), curThreadsSize);
|
||||
#endif
|
||||
if(!verified) {
|
||||
if (!verified) {
|
||||
cout << "Failed" << endl;
|
||||
break;
|
||||
}
|
||||
@@ -178,7 +172,7 @@ static bool query_gpu_frequency(
|
||||
*/
|
||||
TEST_CASE("Performance_hipExtLaunchKernelGGL_QueryGPUFrequency") {
|
||||
HIP_CHECK(hipSetDevice(0));
|
||||
int clock_rate = 0; // in kHz
|
||||
int clock_rate = 0; // in kHz
|
||||
int wall_clock_rate = 0; // in kHz
|
||||
int occupancyBlocks = 0;
|
||||
int occupancyBlockSize = 0;
|
||||
@@ -190,42 +184,31 @@ TEST_CASE("Performance_hipExtLaunchKernelGGL_QueryGPUFrequency") {
|
||||
|
||||
cout << left;
|
||||
cout << setw(w1)
|
||||
<< "--------------------------------------------------------------------------------"
|
||||
<< endl;
|
||||
<< "--------------------------------------------------------------------------------"
|
||||
<< endl;
|
||||
cout << setw(w1) << "device#" << 0 << endl;
|
||||
cout << setw(w1) << "Name: " << props.name << endl;
|
||||
cout << setw(w1) << "gcnArchName: " << props.gcnArchName << endl;
|
||||
cout << setw(w1) << "multiProcessorCount: " << props.multiProcessorCount << endl;
|
||||
cout << setw(w1) << "maxThreadsPerMultiProcessor: " << props.maxThreadsPerMultiProcessor
|
||||
<< endl;
|
||||
cout << setw(w1) << "maxThreadsPerMultiProcessor: " << props.maxThreadsPerMultiProcessor << endl;
|
||||
cout << setw(w1) << "maxThreadsPerBlock: " << props.maxThreadsPerBlock << endl;
|
||||
cout << setw(w1) << "occupancyBlocks: " << occupancyBlocks << endl;
|
||||
cout << setw(w1) << "occupancyBlockSize: " << occupancyBlockSize << endl;
|
||||
cout << setw(w1) << "waveSize: " << props.warpSize << endl;
|
||||
cout << setw(w1) << "clockRate: " << clock_rate / 1000.0 << " Mhz" << endl;
|
||||
cout << setw(w1) << "wallClockRate: " << wall_clock_rate / 1000.0 << " Mhz" << endl;
|
||||
cout << setw(w1) << "memoryClockRate: " << props.memoryClockRate / 1000.0 << " Mhz"
|
||||
<< endl;
|
||||
cout << setw(w1) << "memoryClockRate: " << props.memoryClockRate / 1000.0 << " Mhz" << endl;
|
||||
cout << setw(w1) << "totalGlobalMem: " << fixed << setprecision(2)
|
||||
<< props.totalGlobalMem / 1000000000. << " GB" << endl;
|
||||
cout << setw(w1) << "sharedMemPerBlock: " << props.sharedMemPerBlock / 1024.0 << " KiB"
|
||||
<< endl;
|
||||
<< props.totalGlobalMem / 1000000000. << " GB" << endl;
|
||||
cout << setw(w1) << "sharedMemPerBlock: " << props.sharedMemPerBlock / 1024.0 << " KiB" << endl;
|
||||
cout << setw(w1) << "l2CacheSize: " << props.l2CacheSize << endl;
|
||||
|
||||
cout <<
|
||||
setw(8) << "Blocks " <<
|
||||
setw(11) << "BlockSize" <<
|
||||
setw(22) << "avgGpuFrequency(MHz)" <<
|
||||
setw(20) << "aveGpuTime(ms)" <<
|
||||
setw(20) << "totalGpuTime(ms)" <<
|
||||
setw(26) << "processCapacity(Mbytes/s)" <<
|
||||
setw(31) << "processCapacityPerCU(Mbytes/s)";
|
||||
cout << setw(8) << "Blocks " << setw(11) << "BlockSize" << setw(22) << "avgGpuFrequency(MHz)"
|
||||
<< setw(20) << "aveGpuTime(ms)" << setw(20) << "totalGpuTime(ms)" << setw(26)
|
||||
<< "processCapacity(Mbytes/s)" << setw(31) << "processCapacityPerCU(Mbytes/s)";
|
||||
#ifdef SHOW_DETAILS
|
||||
cout <<
|
||||
setw(15) << "mean wallClock" <<
|
||||
setw(15) << "deviation" <<
|
||||
setw(15) << "mean clock" <<
|
||||
setw(15) << "deviation";
|
||||
cout << setw(15) << "mean wallClock" << setw(15) << "deviation" << setw(15) << "mean clock"
|
||||
<< setw(15) << "deviation";
|
||||
#endif
|
||||
cout << endl;
|
||||
|
||||
|
||||
@@ -29,14 +29,12 @@ THE SOFTWARE.
|
||||
class MemcpyBenchmark : public Benchmark<MemcpyBenchmark> {
|
||||
public:
|
||||
void operator()(void* dst, const void* src, size_t size, hipMemcpyKind kind) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpy(dst, src, size, kind));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpy(dst, src, size, kind)); }
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allocation_type,
|
||||
size_t size, hipMemcpyKind kind, bool enable_peer_access=false) {
|
||||
size_t size, hipMemcpyKind kind, bool enable_peer_access = false) {
|
||||
MemcpyBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(GetAllocationSectionName(src_allocation_type));
|
||||
@@ -49,7 +47,9 @@ static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allo
|
||||
} else {
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard<int> src_allocation(src_allocation_type, size);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
@@ -162,7 +162,8 @@ TEST_CASE("Performance_hipMemcpy_DeviceToDevice_EnablePeerAccess") {
|
||||
const auto allocation_size = GENERATE(4_KB, 4_MB, 16_MB);
|
||||
const auto src_allocation_type = LinearAllocs::hipMalloc;
|
||||
const auto dst_allocation_type = LinearAllocs::hipMalloc;
|
||||
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice, true);
|
||||
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice,
|
||||
true);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -35,7 +35,8 @@ class Memcpy2DBenchmark : public Benchmark<Memcpy2DBenchmark> {
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool enable_peer_access=false) {
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
bool enable_peer_access = false) {
|
||||
Memcpy2DBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
|
||||
|
||||
@@ -43,17 +44,15 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool e
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * height);
|
||||
benchmark.Run(host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.ptr(), device_allocation.pitch(),
|
||||
device_allocation.width(), device_allocation.height(),
|
||||
benchmark.Run(host_allocation.ptr(), device_allocation.width(), device_allocation.ptr(),
|
||||
device_allocation.pitch(), device_allocation.width(), device_allocation.height(),
|
||||
hipMemcpyDeviceToHost);
|
||||
} else if (kind == hipMemcpyHostToDevice) {
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * height);
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
|
||||
host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.width(), device_allocation.height(),
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), host_allocation.ptr(),
|
||||
device_allocation.width(), device_allocation.width(), device_allocation.height(),
|
||||
hipMemcpyHostToDevice);
|
||||
} else if (kind == hipMemcpyHostToHost) {
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
|
||||
@@ -64,15 +63,16 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool e
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard2D<int> src_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
LinearAllocGuard2D<int> dst_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(src_device));
|
||||
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(),
|
||||
src_allocation.ptr(), src_allocation.pitch(),
|
||||
dst_allocation.width(), dst_allocation.height(),
|
||||
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(), src_allocation.ptr(),
|
||||
src_allocation.pitch(), dst_allocation.width(), dst_allocation.height(),
|
||||
hipMemcpyDeviceToDevice);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -36,7 +36,8 @@ class Memcpy2DAsyncBenchmark : public Benchmark<Memcpy2DAsyncBenchmark> {
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool enable_peer_access=false) {
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
bool enable_peer_access = false) {
|
||||
Memcpy2DAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
|
||||
|
||||
@@ -47,17 +48,15 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool e
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * height);
|
||||
benchmark.Run(host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.ptr(), device_allocation.pitch(),
|
||||
device_allocation.width(), device_allocation.height(),
|
||||
benchmark.Run(host_allocation.ptr(), device_allocation.width(), device_allocation.ptr(),
|
||||
device_allocation.pitch(), device_allocation.width(), device_allocation.height(),
|
||||
hipMemcpyDeviceToHost, stream);
|
||||
} else if (kind == hipMemcpyHostToDevice) {
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * height);
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
|
||||
host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.width(), device_allocation.height(),
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), host_allocation.ptr(),
|
||||
device_allocation.width(), device_allocation.width(), device_allocation.height(),
|
||||
hipMemcpyHostToDevice, stream);
|
||||
} else if (kind == hipMemcpyHostToHost) {
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
|
||||
@@ -68,16 +67,17 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind, bool e
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard2D<int> src_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
LinearAllocGuard2D<int> dst_allocation(width, height);
|
||||
|
||||
HIP_CHECK(hipSetDevice(src_device));
|
||||
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(),
|
||||
src_allocation.ptr(), src_allocation.pitch(),
|
||||
dst_allocation.width(), dst_allocation.height(),
|
||||
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(), src_allocation.ptr(),
|
||||
src_allocation.pitch(), dst_allocation.width(), dst_allocation.height(),
|
||||
hipMemcpyDeviceToDevice, stream);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +27,8 @@ THE SOFTWARE.
|
||||
|
||||
class Memcpy2DFromArrayBenchmark : public Benchmark<Memcpy2DFromArrayBenchmark> {
|
||||
public:
|
||||
void operator()(void* dst, size_t dst_pitch, hipArray_const_t src, size_t width, size_t height, hipMemcpyKind kind) {
|
||||
void operator()(void* dst, size_t dst_pitch, hipArray_const_t src, size_t width, size_t height,
|
||||
hipMemcpyKind kind) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpy2DFromArray(dst, dst_pitch, src, 0, 0, width, height, kind));
|
||||
}
|
||||
@@ -35,7 +36,7 @@ class Memcpy2DFromArrayBenchmark : public Benchmark<Memcpy2DFromArrayBenchmark>
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
bool enable_peer_access=false) {
|
||||
bool enable_peer_access = false) {
|
||||
Memcpy2DFromArrayBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
|
||||
|
||||
@@ -49,15 +50,16 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
|
||||
HIP_CHECK(hipSetDevice(src_device));
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
|
||||
array_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.height(), hipMemcpyDeviceToDevice);
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), array_allocation.ptr(),
|
||||
device_allocation.width(), device_allocation.height(), hipMemcpyDeviceToDevice);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -37,7 +37,7 @@ class Memcpy2DFromArrayAsyncBenchmark : public Benchmark<Memcpy2DFromArrayAsyncB
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
bool enable_peer_access=false) {
|
||||
bool enable_peer_access = false) {
|
||||
Memcpy2DFromArrayAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
|
||||
|
||||
@@ -48,22 +48,23 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
size_t allocation_size = width * height * sizeof(int);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, allocation_size);
|
||||
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
|
||||
benchmark.Run(host_allocation.ptr(), width * sizeof(int),
|
||||
array_allocation.ptr(), width * sizeof(int),
|
||||
height, hipMemcpyDeviceToHost, stream);
|
||||
benchmark.Run(host_allocation.ptr(), width * sizeof(int), array_allocation.ptr(),
|
||||
width * sizeof(int), height, hipMemcpyDeviceToHost, stream);
|
||||
} else {
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
|
||||
HIP_CHECK(hipSetDevice(src_device));
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
|
||||
array_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.height(), hipMemcpyDeviceToDevice, stream);
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), array_allocation.ptr(),
|
||||
device_allocation.width(), device_allocation.height(), hipMemcpyDeviceToDevice,
|
||||
stream);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -27,8 +27,8 @@ THE SOFTWARE.
|
||||
|
||||
class Memcpy2DToArrayBenchmark : public Benchmark<Memcpy2DToArrayBenchmark> {
|
||||
public:
|
||||
void operator()(hipArray_t dst, const void* src, size_t src_pitch, size_t width,
|
||||
size_t height, hipMemcpyKind kind) {
|
||||
void operator()(hipArray_t dst, const void* src, size_t src_pitch, size_t width, size_t height,
|
||||
hipMemcpyKind kind) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpy2DToArray(dst, 0, 0, src, src_pitch, width, height, kind));
|
||||
}
|
||||
@@ -36,7 +36,7 @@ class Memcpy2DToArrayBenchmark : public Benchmark<Memcpy2DToArrayBenchmark> {
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
bool enable_peer_access=false) {
|
||||
bool enable_peer_access = false) {
|
||||
Memcpy2DToArrayBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
|
||||
|
||||
@@ -50,7 +50,9 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
|
||||
@@ -27,8 +27,8 @@ THE SOFTWARE.
|
||||
|
||||
class Memcpy2DToArrayAsyncBenchmark : public Benchmark<Memcpy2DToArrayAsyncBenchmark> {
|
||||
public:
|
||||
void operator()(hipArray_t dst, const void* src, size_t src_pitch, size_t width,
|
||||
size_t height, hipMemcpyKind kind, const hipStream_t& stream) {
|
||||
void operator()(hipArray_t dst, const void* src, size_t src_pitch, size_t width, size_t height,
|
||||
hipMemcpyKind kind, const hipStream_t& stream) {
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
|
||||
HIP_CHECK(hipMemcpy2DToArrayAsync(dst, 0, 0, src, src_pitch, width, height, kind, stream));
|
||||
}
|
||||
@@ -37,7 +37,7 @@ class Memcpy2DToArrayAsyncBenchmark : public Benchmark<Memcpy2DToArrayAsyncBench
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
bool enable_peer_access=false) {
|
||||
bool enable_peer_access = false) {
|
||||
Memcpy2DToArrayAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
|
||||
|
||||
@@ -48,22 +48,23 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
size_t allocation_size = width * height * sizeof(int);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, allocation_size);
|
||||
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
|
||||
benchmark.Run(array_allocation.ptr(), host_allocation.ptr(),
|
||||
width * sizeof(int), width * sizeof(int), height,
|
||||
hipMemcpyHostToDevice, stream);
|
||||
benchmark.Run(array_allocation.ptr(), host_allocation.ptr(), width * sizeof(int),
|
||||
width * sizeof(int), height, hipMemcpyHostToDevice, stream);
|
||||
} else {
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
ArrayAllocGuard<int> array_allocation(make_hipExtent(width, height, 0), hipArrayDefault);
|
||||
HIP_CHECK(hipSetDevice(src_device));
|
||||
benchmark.Run(array_allocation.ptr(), device_allocation.ptr(), device_allocation.pitch(),
|
||||
device_allocation.width(), device_allocation.height(),
|
||||
hipMemcpyDeviceToDevice, stream);
|
||||
device_allocation.width(), device_allocation.height(), hipMemcpyDeviceToDevice,
|
||||
stream);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -29,49 +29,53 @@ class Memcpy3DBenchmark : public Benchmark<Memcpy3DBenchmark> {
|
||||
public:
|
||||
void operator()(const hipPitchedPtr& dst_ptr, const hipPitchedPtr& src_ptr,
|
||||
const hipExtent extent, hipMemcpyKind kind) {
|
||||
hipMemcpy3DParms params = CreateMemcpy3DParam(dst_ptr, make_hipPos(0, 0, 0),
|
||||
src_ptr, make_hipPos(0, 0, 0),
|
||||
extent, kind);
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpy3D(¶ms));
|
||||
}
|
||||
hipMemcpy3DParms params = CreateMemcpy3DParam(dst_ptr, make_hipPos(0, 0, 0), src_ptr,
|
||||
make_hipPos(0, 0, 0), extent, kind);
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpy3D(¶ms)); }
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(const hipExtent extent, hipMemcpyKind kind, bool enable_peer_access=false) {
|
||||
static void RunBenchmark(const hipExtent extent, hipMemcpyKind kind,
|
||||
bool enable_peer_access = false) {
|
||||
Memcpy3DBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(extent.width) + ", " + std::to_string(extent.height)
|
||||
+ ", " + std::to_string(extent.depth) + ")");
|
||||
benchmark.AddSectionName("(" + std::to_string(extent.width) + ", " +
|
||||
std::to_string(extent.height) + ", " + std::to_string(extent.depth) +
|
||||
")");
|
||||
|
||||
if (kind == hipMemcpyDeviceToHost) {
|
||||
LinearAllocGuard3D<int> device_allocation(extent);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() *
|
||||
device_allocation.height() * device_allocation.depth());
|
||||
benchmark.Run(make_hipPitchedPtr(host_allocation.ptr(), device_allocation.width(),
|
||||
LinearAllocGuard<int> host_allocation(
|
||||
LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * device_allocation.height() * device_allocation.depth());
|
||||
benchmark.Run(make_hipPitchedPtr(host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.width(), device_allocation.height()),
|
||||
device_allocation.pitched_ptr(), device_allocation.extent(), kind);
|
||||
} else if (kind == hipMemcpyHostToDevice) {
|
||||
LinearAllocGuard3D<int> device_allocation(extent);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.pitch() *
|
||||
device_allocation.height() * device_allocation.depth());
|
||||
LinearAllocGuard<int> host_allocation(
|
||||
LinearAllocs::hipHostMalloc,
|
||||
device_allocation.pitch() * device_allocation.height() * device_allocation.depth());
|
||||
benchmark.Run(device_allocation.pitched_ptr(),
|
||||
make_hipPitchedPtr(host_allocation.ptr(), device_allocation.pitch(),
|
||||
device_allocation.width(), device_allocation.height()),
|
||||
device_allocation.extent(), kind);
|
||||
} else if (kind == hipMemcpyHostToHost) {
|
||||
LinearAllocGuard3D<int> device_allocation(extent);
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, extent.width *
|
||||
extent.height * extent.depth);
|
||||
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc, extent.width *
|
||||
extent.height * extent.depth);
|
||||
benchmark.Run(make_hipPitchedPtr(dst_allocation.ptr(), extent.width, extent.width, extent.height),
|
||||
make_hipPitchedPtr(src_allocation.ptr(), extent.width, extent.width, extent.height),
|
||||
extent, kind);
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc,
|
||||
extent.width * extent.height * extent.depth);
|
||||
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc,
|
||||
extent.width * extent.height * extent.depth);
|
||||
benchmark.Run(
|
||||
make_hipPitchedPtr(dst_allocation.ptr(), extent.width, extent.width, extent.height),
|
||||
make_hipPitchedPtr(src_allocation.ptr(), extent.width, extent.width, extent.height), extent,
|
||||
kind);
|
||||
} else {
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard3D<int> src_allocation(extent);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
|
||||
@@ -29,55 +29,57 @@ class Memcpy3DAsyncBenchmark : public Benchmark<Memcpy3DAsyncBenchmark> {
|
||||
public:
|
||||
void operator()(const hipPitchedPtr& dst_ptr, const hipPitchedPtr& src_ptr,
|
||||
const hipExtent extent, hipMemcpyKind kind, const hipStream_t& stream) {
|
||||
hipMemcpy3DParms params = CreateMemcpy3DParam(dst_ptr, make_hipPos(0, 0, 0),
|
||||
src_ptr, make_hipPos(0, 0, 0),
|
||||
extent, kind);
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
|
||||
HIP_CHECK(hipMemcpy3DAsync(¶ms, stream));
|
||||
}
|
||||
hipMemcpy3DParms params = CreateMemcpy3DParam(dst_ptr, make_hipPos(0, 0, 0), src_ptr,
|
||||
make_hipPos(0, 0, 0), extent, kind);
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) { HIP_CHECK(hipMemcpy3DAsync(¶ms, stream)); }
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(const hipExtent extent, hipMemcpyKind kind, bool enable_peer_access=false) {
|
||||
static void RunBenchmark(const hipExtent extent, hipMemcpyKind kind,
|
||||
bool enable_peer_access = false) {
|
||||
Memcpy3DAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(extent.width) + ", " + std::to_string(extent.height)
|
||||
+ ", " + std::to_string(extent.depth) + ")");
|
||||
benchmark.AddSectionName("(" + std::to_string(extent.width) + ", " +
|
||||
std::to_string(extent.height) + ", " + std::to_string(extent.depth) +
|
||||
")");
|
||||
|
||||
const StreamGuard stream_guard(Streams::created);
|
||||
const hipStream_t stream = stream_guard.stream();
|
||||
|
||||
if (kind == hipMemcpyDeviceToHost) {
|
||||
LinearAllocGuard3D<int> device_allocation(extent);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() *
|
||||
device_allocation.height() * device_allocation.depth());
|
||||
LinearAllocGuard<int> host_allocation(
|
||||
LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * device_allocation.height() * device_allocation.depth());
|
||||
benchmark.Run(make_hipPitchedPtr(host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.width(), device_allocation.height()),
|
||||
device_allocation.pitched_ptr(), device_allocation.extent(), kind, stream);
|
||||
} else if (kind == hipMemcpyHostToDevice) {
|
||||
LinearAllocGuard3D<int> device_allocation(extent);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.pitch() *
|
||||
device_allocation.height() * device_allocation.depth());
|
||||
LinearAllocGuard<int> host_allocation(
|
||||
LinearAllocs::hipHostMalloc,
|
||||
device_allocation.pitch() * device_allocation.height() * device_allocation.depth());
|
||||
benchmark.Run(device_allocation.pitched_ptr(),
|
||||
make_hipPitchedPtr(host_allocation.ptr(),
|
||||
device_allocation.pitch(),
|
||||
device_allocation.width(),
|
||||
device_allocation.height()),
|
||||
make_hipPitchedPtr(host_allocation.ptr(), device_allocation.pitch(),
|
||||
device_allocation.width(), device_allocation.height()),
|
||||
device_allocation.extent(), kind, stream);
|
||||
} else if (kind == hipMemcpyHostToHost) {
|
||||
LinearAllocGuard3D<int> device_allocation(extent);
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, extent.width *
|
||||
extent.height * extent.depth);
|
||||
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc, extent.width *
|
||||
extent.height * extent.depth);
|
||||
benchmark.Run(make_hipPitchedPtr(dst_allocation.ptr(), extent.width, extent.width, extent.height),
|
||||
make_hipPitchedPtr(src_allocation.ptr(), extent.width, extent.width, extent.height),
|
||||
extent, kind, stream);
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc,
|
||||
extent.width * extent.height * extent.depth);
|
||||
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc,
|
||||
extent.width * extent.height * extent.depth);
|
||||
benchmark.Run(
|
||||
make_hipPitchedPtr(dst_allocation.ptr(), extent.width, extent.width, extent.height),
|
||||
make_hipPitchedPtr(src_allocation.ptr(), extent.width, extent.width, extent.height), extent,
|
||||
kind, stream);
|
||||
} else {
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard3D<int> src_allocation(extent);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
|
||||
@@ -27,7 +27,8 @@ THE SOFTWARE.
|
||||
|
||||
class MemcpyAsyncBenchmark : public Benchmark<MemcpyAsyncBenchmark> {
|
||||
public:
|
||||
void operator()(void* dst, const void* src, size_t size, hipMemcpyKind kind, const hipStream_t& stream) {
|
||||
void operator()(void* dst, const void* src, size_t size, hipMemcpyKind kind,
|
||||
const hipStream_t& stream) {
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
|
||||
HIP_CHECK(hipMemcpyAsync(dst, src, size, kind, stream));
|
||||
}
|
||||
@@ -36,7 +37,7 @@ class MemcpyAsyncBenchmark : public Benchmark<MemcpyAsyncBenchmark> {
|
||||
};
|
||||
|
||||
static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allocation_type,
|
||||
size_t size, hipMemcpyKind kind, bool enable_peer_access=false) {
|
||||
size_t size, hipMemcpyKind kind, bool enable_peer_access = false) {
|
||||
MemcpyAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(GetAllocationSectionName(src_allocation_type));
|
||||
@@ -51,7 +52,9 @@ static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allo
|
||||
} else {
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard<int> src_allocation(src_allocation_type, size);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
@@ -189,7 +192,8 @@ TEST_CASE("Performance_hipMemcpyAsync_DeviceToDevice_EnablePeerAccess") {
|
||||
const auto allocation_size = GENERATE(4_KB, 4_MB, 16_MB);
|
||||
const auto src_allocation_type = LinearAllocs::hipMalloc;
|
||||
const auto dst_allocation_type = LinearAllocs::hipMalloc;
|
||||
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice, true);
|
||||
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice,
|
||||
true);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -28,9 +28,7 @@ THE SOFTWARE.
|
||||
class MemcpyAtoHBenchmark : public Benchmark<MemcpyAtoHBenchmark> {
|
||||
public:
|
||||
void operator()(void* dst, hipArray_t src_array, size_t allocation_size) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpyAtoH(dst, src_array, 0, allocation_size));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyAtoH(dst, src_array, 0, allocation_size)); }
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -28,19 +28,19 @@ THE SOFTWARE.
|
||||
class MemcpyDtoDBenchmark : public Benchmark<MemcpyDtoDBenchmark> {
|
||||
public:
|
||||
void operator()(hipDeviceptr_t& dst, const hipDeviceptr_t& src, size_t size) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpyDtoD(dst, src, size));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyDtoD(dst, src, size)); }
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t size, bool enable_peer_access=false) {
|
||||
static void RunBenchmark(size_t size, bool enable_peer_access = false) {
|
||||
MemcpyDtoDBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipMalloc, size);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
|
||||
@@ -27,7 +27,8 @@ THE SOFTWARE.
|
||||
|
||||
class MemcpyDtoDAsyncBenchmark : public Benchmark<MemcpyDtoDAsyncBenchmark> {
|
||||
public:
|
||||
void operator()(hipDeviceptr_t& dst, const hipDeviceptr_t& src, size_t size, const hipStream_t& stream) {
|
||||
void operator()(hipDeviceptr_t& dst, const hipDeviceptr_t& src, size_t size,
|
||||
const hipStream_t& stream) {
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
|
||||
HIP_CHECK(hipMemcpyDtoDAsync(dst, src, size, stream));
|
||||
}
|
||||
@@ -35,7 +36,7 @@ class MemcpyDtoDAsyncBenchmark : public Benchmark<MemcpyDtoDAsyncBenchmark> {
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t size, bool enable_peer_access=false) {
|
||||
static void RunBenchmark(size_t size, bool enable_peer_access = false) {
|
||||
MemcpyDtoDAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
|
||||
@@ -43,15 +44,16 @@ static void RunBenchmark(size_t size, bool enable_peer_access=false) {
|
||||
const hipStream_t stream = stream_guard.stream();
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipMalloc, size);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipMalloc, size);
|
||||
HIP_CHECK(hipSetDevice(src_device));
|
||||
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(dst_allocation.ptr()),
|
||||
reinterpret_cast<hipDeviceptr_t>(src_allocation.ptr()),
|
||||
size, stream);
|
||||
reinterpret_cast<hipDeviceptr_t>(src_allocation.ptr()), size, stream);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -28,21 +28,19 @@ THE SOFTWARE.
|
||||
class MemcpyDtoHBenchmark : public Benchmark<MemcpyDtoHBenchmark> {
|
||||
public:
|
||||
void operator()(void* dst, const hipDeviceptr_t& src, size_t size) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpyDtoH(dst, src, size));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyDtoH(dst, src, size)); }
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type, size_t size) {
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
|
||||
size_t size) {
|
||||
MemcpyDtoHBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(GetAllocationSectionName(host_allocation_type));
|
||||
|
||||
LinearAllocGuard<int> device_allocation(device_allocation_type, size);
|
||||
LinearAllocGuard<int> host_allocation(host_allocation_type, size);
|
||||
benchmark.Run(host_allocation.ptr(),
|
||||
reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
|
||||
benchmark.Run(host_allocation.ptr(), reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
|
||||
size);
|
||||
}
|
||||
|
||||
|
||||
@@ -35,7 +35,8 @@ class MemcpyDtoHAsyncBenchmark : public Benchmark<MemcpyDtoHAsyncBenchmark> {
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type, size_t size) {
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
|
||||
size_t size) {
|
||||
MemcpyDtoHAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(GetAllocationSectionName(host_allocation_type));
|
||||
@@ -44,8 +45,7 @@ static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_
|
||||
const hipStream_t stream = stream_guard.stream();
|
||||
LinearAllocGuard<int> device_allocation(device_allocation_type, size);
|
||||
LinearAllocGuard<int> host_allocation(host_allocation_type, size);
|
||||
benchmark.Run(host_allocation.ptr(),
|
||||
reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
|
||||
benchmark.Run(host_allocation.ptr(), reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
|
||||
size, stream);
|
||||
}
|
||||
|
||||
|
||||
@@ -38,7 +38,7 @@ class MemcpyFromSymbolBenchmark : public Benchmark<MemcpyFromSymbolBenchmark> {
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(const void* source, void* result, size_t size=1, size_t offset=0) {
|
||||
static void RunBenchmark(const void* source, void* result, size_t size = 1, size_t offset = 0) {
|
||||
MemcpyFromSymbolBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(std::to_string(offset));
|
||||
@@ -112,7 +112,8 @@ TEST_CASE("Performance_hipMemcpyFromSymbol_WithOffset") {
|
||||
std::fill_n(result.data(), size, 0);
|
||||
|
||||
size_t offset = GENERATE_REF(0, size / 2);
|
||||
RunBenchmark(array.data() + offset, result.data() + offset, sizeof(int) * (size - offset), offset * sizeof(int));
|
||||
RunBenchmark(array.data() + offset, result.data() + offset, sizeof(int) * (size - offset),
|
||||
offset * sizeof(int));
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -30,7 +30,8 @@ __device__ int devSymbol[1_MB];
|
||||
|
||||
class MemcpyFromSymbolAsyncBenchmark : public Benchmark<MemcpyFromSymbolAsyncBenchmark> {
|
||||
public:
|
||||
void operator()(const void* source, void* result, size_t size, size_t offset, const hipStream_t& stream) {
|
||||
void operator()(const void* source, void* result, size_t size, size_t offset,
|
||||
const hipStream_t& stream) {
|
||||
HIP_CHECK(hipMemcpyToSymbolAsync(HIP_SYMBOL(devSymbol), source, size, offset,
|
||||
hipMemcpyHostToDevice, stream));
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
|
||||
@@ -41,7 +42,7 @@ class MemcpyFromSymbolAsyncBenchmark : public Benchmark<MemcpyFromSymbolAsyncBen
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(const void* source, void* result, size_t size=1, size_t offset=0) {
|
||||
static void RunBenchmark(const void* source, void* result, size_t size = 1, size_t offset = 0) {
|
||||
MemcpyFromSymbolAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(std::to_string(offset));
|
||||
@@ -118,7 +119,8 @@ TEST_CASE("Performance_hipMemcpyFromSymbolAsync_WithOffset") {
|
||||
std::fill_n(result.data(), size, 0);
|
||||
|
||||
size_t offset = GENERATE_REF(0, size / 2);
|
||||
RunBenchmark(array.data() + offset, result.data() + offset, sizeof(int) * (size - offset), offset * sizeof(int));
|
||||
RunBenchmark(array.data() + offset, result.data() + offset, sizeof(int) * (size - offset),
|
||||
offset * sizeof(int));
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -28,9 +28,7 @@ THE SOFTWARE.
|
||||
class MemcpyHtoABenchmark : public Benchmark<MemcpyHtoABenchmark> {
|
||||
public:
|
||||
void operator()(hipArray_t dst_array, const void* src, size_t allocation_size) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpyHtoA(dst_array, 0, src, allocation_size));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyHtoA(dst_array, 0, src, allocation_size)); }
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -28,20 +28,20 @@ THE SOFTWARE.
|
||||
class MemcpyHtoDBenchmark : public Benchmark<MemcpyHtoDBenchmark> {
|
||||
public:
|
||||
void operator()(hipDeviceptr_t& dst, void* src, size_t size) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpyHtoD(dst, src, size));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyHtoD(dst, src, size)); }
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type, size_t size) {
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
|
||||
size_t size) {
|
||||
MemcpyHtoDBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(GetAllocationSectionName(host_allocation_type));
|
||||
|
||||
LinearAllocGuard<int> device_allocation(device_allocation_type, size);
|
||||
LinearAllocGuard<int> host_allocation(host_allocation_type, size);
|
||||
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()), host_allocation.ptr(), size);
|
||||
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()), host_allocation.ptr(),
|
||||
size);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -35,7 +35,8 @@ class MemcpyHtoDAsyncBenchmark : public Benchmark<MemcpyHtoDAsyncBenchmark> {
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type, size_t size) {
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
|
||||
size_t size) {
|
||||
MemcpyHtoDAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(GetAllocationSectionName(host_allocation_type));
|
||||
@@ -44,8 +45,8 @@ static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_
|
||||
const hipStream_t stream = stream_guard.stream();
|
||||
LinearAllocGuard<int> device_allocation(device_allocation_type, size);
|
||||
LinearAllocGuard<int> host_allocation(host_allocation_type, size);
|
||||
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()),
|
||||
host_allocation.ptr(), size, stream);
|
||||
benchmark.Run(reinterpret_cast<hipDeviceptr_t>(device_allocation.ptr()), host_allocation.ptr(),
|
||||
size, stream);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -27,54 +27,52 @@ THE SOFTWARE.
|
||||
|
||||
class MemcpyParam2DBenchmark : public Benchmark<MemcpyParam2DBenchmark> {
|
||||
public:
|
||||
void operator()(void* dst, size_t dst_pitch, void* src, size_t src_pitch,
|
||||
size_t width, size_t height, hipMemcpyKind kind) {
|
||||
hip_Memcpy2D params = CreateMemcpy2DParam(dst, dst_pitch, src, src_pitch,
|
||||
width, height, kind);
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpyParam2D(¶ms));
|
||||
}
|
||||
void operator()(void* dst, size_t dst_pitch, void* src, size_t src_pitch, size_t width,
|
||||
size_t height, hipMemcpyKind kind) {
|
||||
hip_Memcpy2D params = CreateMemcpy2DParam(dst, dst_pitch, src, src_pitch, width, height, kind);
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyParam2D(¶ms)); }
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
bool enable_peer_access=false) {
|
||||
bool enable_peer_access = false) {
|
||||
MemcpyParam2DBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
|
||||
|
||||
if (kind == hipMemcpyDeviceToHost) {
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() * height);
|
||||
benchmark.Run(host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.ptr(), device_allocation.pitch(),
|
||||
device_allocation.width(), device_allocation.height(), kind);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * height);
|
||||
benchmark.Run(host_allocation.ptr(), device_allocation.width(), device_allocation.ptr(),
|
||||
device_allocation.pitch(), device_allocation.width(), device_allocation.height(),
|
||||
kind);
|
||||
} else if (kind == hipMemcpyHostToDevice) {
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() * height);
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
|
||||
host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.width(), device_allocation.height(), kind);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * height);
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), host_allocation.ptr(),
|
||||
device_allocation.width(), device_allocation.width(), device_allocation.height(),
|
||||
kind);
|
||||
} else if (kind == hipMemcpyHostToHost) {
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
|
||||
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
|
||||
benchmark.Run(dst_allocation.ptr(), width * sizeof(int),
|
||||
src_allocation.ptr(), width * sizeof(int),
|
||||
width * sizeof(int), height, kind);
|
||||
benchmark.Run(dst_allocation.ptr(), width * sizeof(int), src_allocation.ptr(),
|
||||
width * sizeof(int), width * sizeof(int), height, kind);
|
||||
} else {
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard2D<int> src_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
LinearAllocGuard2D<int> dst_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(src_device));
|
||||
|
||||
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(),
|
||||
src_allocation.ptr(), src_allocation.pitch(),
|
||||
dst_allocation.width(), dst_allocation.height(),
|
||||
kind);
|
||||
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(), src_allocation.ptr(),
|
||||
src_allocation.pitch(), dst_allocation.width(), dst_allocation.height(), kind);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -27,19 +27,16 @@ THE SOFTWARE.
|
||||
|
||||
class MemcpyParam2DBenchmark : public Benchmark<MemcpyParam2DBenchmark> {
|
||||
public:
|
||||
void operator()(void* dst, size_t dst_pitch, void* src, size_t src_pitch,
|
||||
size_t width, size_t height, hipMemcpyKind kind, const hipStream_t& stream) {
|
||||
hip_Memcpy2D params = CreateMemcpy2DParam(dst, dst_pitch, src, src_pitch,
|
||||
width, height, kind);
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpyParam2DAsync(¶ms, stream));
|
||||
}
|
||||
void operator()(void* dst, size_t dst_pitch, void* src, size_t src_pitch, size_t width,
|
||||
size_t height, hipMemcpyKind kind, const hipStream_t& stream) {
|
||||
hip_Memcpy2D params = CreateMemcpy2DParam(dst, dst_pitch, src, src_pitch, width, height, kind);
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyParam2DAsync(¶ms, stream)); }
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
bool enable_peer_access=false) {
|
||||
bool enable_peer_access = false) {
|
||||
MemcpyParam2DBenchmark benchmark;
|
||||
benchmark.AddSectionName("(" + std::to_string(width) + ", " + std::to_string(height) + ")");
|
||||
|
||||
@@ -48,38 +45,38 @@ static void RunBenchmark(size_t width, size_t height, hipMemcpyKind kind,
|
||||
|
||||
if (kind == hipMemcpyDeviceToHost) {
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() * height);
|
||||
benchmark.Run(host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.ptr(), device_allocation.pitch(),
|
||||
device_allocation.width(), device_allocation.height(),
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * height);
|
||||
benchmark.Run(host_allocation.ptr(), device_allocation.width(), device_allocation.ptr(),
|
||||
device_allocation.pitch(), device_allocation.width(), device_allocation.height(),
|
||||
kind, stream);
|
||||
} else if (kind == hipMemcpyHostToDevice) {
|
||||
LinearAllocGuard2D<int> device_allocation(width, height);
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc, device_allocation.width() * height);
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(),
|
||||
host_allocation.ptr(), device_allocation.width(),
|
||||
device_allocation.width(), device_allocation.height(),
|
||||
LinearAllocGuard<int> host_allocation(LinearAllocs::hipHostMalloc,
|
||||
device_allocation.width() * height);
|
||||
benchmark.Run(device_allocation.ptr(), device_allocation.pitch(), host_allocation.ptr(),
|
||||
device_allocation.width(), device_allocation.width(), device_allocation.height(),
|
||||
kind, stream);
|
||||
} else if (kind == hipMemcpyHostToHost) {
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
|
||||
LinearAllocGuard<int> dst_allocation(LinearAllocs::hipHostMalloc, width * sizeof(int) * height);
|
||||
benchmark.Run(dst_allocation.ptr(), width * sizeof(int),
|
||||
src_allocation.ptr(), width * sizeof(int),
|
||||
width * sizeof(int), height, kind, stream);
|
||||
benchmark.Run(dst_allocation.ptr(), width * sizeof(int), src_allocation.ptr(),
|
||||
width * sizeof(int), width * sizeof(int), height, kind, stream);
|
||||
} else {
|
||||
// hipMemcpyDeviceToDevice
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard2D<int> src_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
LinearAllocGuard2D<int> dst_allocation(width, height);
|
||||
HIP_CHECK(hipSetDevice(src_device));
|
||||
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(),
|
||||
src_allocation.ptr(), src_allocation.pitch(),
|
||||
dst_allocation.width(), dst_allocation.height(),
|
||||
kind, stream);
|
||||
benchmark.Run(dst_allocation.ptr(), dst_allocation.pitch(), src_allocation.ptr(),
|
||||
src_allocation.pitch(), dst_allocation.width(), dst_allocation.height(), kind,
|
||||
stream);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -36,7 +36,7 @@ class MemcpyToSymbolBenchmark : public Benchmark<MemcpyToSymbolBenchmark> {
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(const void* source, size_t size=1, size_t offset=0) {
|
||||
static void RunBenchmark(const void* source, size_t size = 1, size_t offset = 0) {
|
||||
MemcpyToSymbolBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(std::to_string(offset));
|
||||
|
||||
@@ -40,7 +40,7 @@ class MemcpyToSymbolAsyncBenchmark : public Benchmark<MemcpyToSymbolAsyncBenchma
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(const void* source, size_t size=1, size_t offset=0) {
|
||||
static void RunBenchmark(const void* source, size_t size = 1, size_t offset = 0) {
|
||||
MemcpyToSymbolAsyncBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(std::to_string(offset));
|
||||
|
||||
@@ -28,7 +28,8 @@ __global__ void Sum(void* ptr, size_t size) {
|
||||
atomicAdd(&((unsigned long long*)ptr)[0], ((unsigned long long*)ptr)[index]);
|
||||
}
|
||||
}
|
||||
class MemcpyHtoDKernelDtoHv1AsyncBenchmark : public Benchmark<MemcpyHtoDKernelDtoHv1AsyncBenchmark> {
|
||||
class MemcpyHtoDKernelDtoHv1AsyncBenchmark
|
||||
: public Benchmark<MemcpyHtoDKernelDtoHv1AsyncBenchmark> {
|
||||
public:
|
||||
void operator()(void* host_mem, void* device_mem, size_t size, const hipStream_t& stream) {
|
||||
size_t count = size / sizeof(size_t);
|
||||
@@ -36,19 +37,20 @@ class MemcpyHtoDKernelDtoHv1AsyncBenchmark : public Benchmark<MemcpyHtoDKernelDt
|
||||
((size_t*)host_mem)[i] = i;
|
||||
}
|
||||
TIMED_SECTION_STREAM(kTimerTypeCpu, stream) {
|
||||
HIP_CHECK(hipMemcpyHtoDAsync(reinterpret_cast<hipDeviceptr_t>(device_mem), host_mem,
|
||||
size, stream));
|
||||
HIP_CHECK(
|
||||
hipMemcpyHtoDAsync(reinterpret_cast<hipDeviceptr_t>(device_mem), host_mem, size, stream));
|
||||
int threads_num = 32;
|
||||
Sum<<<count / threads_num + 1, threads_num, 0, stream>>>(device_mem, count);
|
||||
HIP_CHECK(hipMemcpyDtoHAsync(host_mem, reinterpret_cast<hipDeviceptr_t>(device_mem),
|
||||
size, stream));
|
||||
HIP_CHECK(
|
||||
hipMemcpyDtoHAsync(host_mem, reinterpret_cast<hipDeviceptr_t>(device_mem), size, stream));
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
}
|
||||
size_t sum = ((size_t*)host_mem)[0];
|
||||
REQUIRE(sum == count * (count - 1) / 2);
|
||||
}
|
||||
};
|
||||
class MemcpyHtoDKernelDtoHv2AsyncBenchmark : public Benchmark<MemcpyHtoDKernelDtoHv2AsyncBenchmark> {
|
||||
class MemcpyHtoDKernelDtoHv2AsyncBenchmark
|
||||
: public Benchmark<MemcpyHtoDKernelDtoHv2AsyncBenchmark> {
|
||||
public:
|
||||
void operator()(void* host_mem, void* device_mem, size_t size, const hipStream_t& stream) {
|
||||
size_t count = size / sizeof(size_t);
|
||||
@@ -56,19 +58,17 @@ class MemcpyHtoDKernelDtoHv2AsyncBenchmark : public Benchmark<MemcpyHtoDKernelDt
|
||||
((size_t*)host_mem)[i] = i;
|
||||
}
|
||||
TIMED_SECTION_STREAM(kTimerTypeCpu, stream) {
|
||||
HIP_CHECK(hipMemcpyAsync(device_mem, host_mem, size, hipMemcpyHostToDevice,
|
||||
stream));
|
||||
HIP_CHECK(hipMemcpyAsync(device_mem, host_mem, size, hipMemcpyHostToDevice, stream));
|
||||
int threads_num = 32;
|
||||
Sum<<<count / threads_num + 1, threads_num, 0, stream>>>(device_mem, count);
|
||||
HIP_CHECK(hipMemcpyWithStream(host_mem, device_mem, size,
|
||||
hipMemcpyDeviceToHost, stream));
|
||||
HIP_CHECK(hipMemcpyWithStream(host_mem, device_mem, size, hipMemcpyDeviceToHost, stream));
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
}
|
||||
size_t sum = ((size_t*)host_mem)[0];
|
||||
REQUIRE(sum == count * (count - 1) / 2);
|
||||
}
|
||||
};
|
||||
template<typename BenchmarkType>
|
||||
template <typename BenchmarkType>
|
||||
static void RunBenchmark(LinearAllocs host_allocation_type, LinearAllocs device_allocation_type,
|
||||
size_t size) {
|
||||
BenchmarkType benchmark;
|
||||
|
||||
@@ -28,14 +28,12 @@ THE SOFTWARE.
|
||||
class MemcpyWithStreamBenchmark : public Benchmark<MemcpyWithStreamBenchmark> {
|
||||
public:
|
||||
void operator()(void* dst, const void* src, size_t size, hipMemcpyKind kind, hipStream_t stream) {
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemcpyWithStream(dst, src, size, kind, stream));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemcpyWithStream(dst, src, size, kind, stream)); }
|
||||
}
|
||||
};
|
||||
|
||||
static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allocation_type,
|
||||
size_t size, hipMemcpyKind kind, bool enable_peer_access=false) {
|
||||
size_t size, hipMemcpyKind kind, bool enable_peer_access = false) {
|
||||
MemcpyWithStreamBenchmark benchmark;
|
||||
benchmark.AddSectionName(std::to_string(size));
|
||||
benchmark.AddSectionName(GetAllocationSectionName(src_allocation_type));
|
||||
@@ -51,7 +49,9 @@ static void RunBenchmark(LinearAllocs dst_allocation_type, LinearAllocs src_allo
|
||||
} else {
|
||||
int src_device = std::get<0>(GetDeviceIds(enable_peer_access));
|
||||
int dst_device = std::get<1>(GetDeviceIds(enable_peer_access));
|
||||
if (src_device == -1 && dst_device == -1) { return; }
|
||||
if (src_device == -1 && dst_device == -1) {
|
||||
return;
|
||||
}
|
||||
|
||||
LinearAllocGuard<int> src_allocation(LinearAllocs::hipMalloc, size);
|
||||
HIP_CHECK(hipSetDevice(dst_device));
|
||||
@@ -189,7 +189,8 @@ TEST_CASE("Performance_hipMemcpyWithStream_DeviceToDevice_EnablePeerAccess") {
|
||||
const auto allocation_size = GENERATE(4_KB, 4_MB, 16_MB);
|
||||
const auto src_allocation_type = LinearAllocs::hipMalloc;
|
||||
const auto dst_allocation_type = LinearAllocs::hipMalloc;
|
||||
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice, true);
|
||||
RunBenchmark(dst_allocation_type, src_allocation_type, allocation_size, hipMemcpyDeviceToDevice,
|
||||
true);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -79,6 +79,6 @@ TEST_CASE("Performance_hipMemset") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -72,6 +72,6 @@ TEST_CASE("Performance_hipMemset2D") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -75,6 +75,6 @@ TEST_CASE("Performance_hipMemset2DAsync") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -73,6 +73,6 @@ TEST_CASE("Performance_hipMemset3D") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -75,6 +75,6 @@ TEST_CASE("Performance_hipMemset3DAsync") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -81,6 +81,6 @@ TEST_CASE("Performance_hipMemsetAsync") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -80,6 +80,6 @@ TEST_CASE("Performance_hipMemsetD16") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -82,6 +82,6 @@ TEST_CASE("Performance_hipMemsetD16Async") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -80,6 +80,6 @@ TEST_CASE("Performance_hipMemsetD32") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -82,6 +82,6 @@ TEST_CASE("Performance_hipMemsetD32Async") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -80,6 +80,6 @@ TEST_CASE("Performance_hipMemsetD8") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -82,6 +82,6 @@ TEST_CASE("Performance_hipMemsetD8Async") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -60,11 +60,9 @@ static void RunBenchmark() {
|
||||
* - Platform specific (AMD)
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Performance_hipExtStreamCreateWithCUMask") {
|
||||
RunBenchmark();
|
||||
}
|
||||
TEST_CASE("Performance_hipExtStreamCreateWithCUMask") { RunBenchmark(); }
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -62,11 +62,9 @@ static void RunBenchmark() {
|
||||
* - Platform specific (AMD)
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Performance_hipExtStreamGetCUMask") {
|
||||
RunBenchmark();
|
||||
}
|
||||
TEST_CASE("Performance_hipExtStreamGetCUMask") { RunBenchmark(); }
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -32,11 +32,10 @@ class FreeAsyncBenchmark : public Benchmark<FreeAsyncBenchmark> {
|
||||
const StreamGuard stream_guard{Streams::created};
|
||||
const hipStream_t stream = stream_guard.stream();
|
||||
float* dev_ptr{nullptr};
|
||||
HIP_CHECK(hipMallocAsync(reinterpret_cast<void**>(&dev_ptr), array_size * sizeof(float), stream));
|
||||
HIP_CHECK(
|
||||
hipMallocAsync(reinterpret_cast<void**>(&dev_ptr), array_size * sizeof(float), stream));
|
||||
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
|
||||
HIP_CHECK(hipFreeAsync(dev_ptr, stream));
|
||||
}
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) { HIP_CHECK(hipFreeAsync(dev_ptr, stream)); }
|
||||
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
}
|
||||
@@ -69,6 +68,6 @@ TEST_CASE("Performance_hipFreeAsync") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -34,7 +34,8 @@ class MallocAsyncBenchmark : public Benchmark<MallocAsyncBenchmark> {
|
||||
float* dev_ptr{nullptr};
|
||||
|
||||
TIMED_SECTION_STREAM(kTimerTypeEvent, stream) {
|
||||
HIP_CHECK(hipMallocAsync(reinterpret_cast<void**>(&dev_ptr), array_size * sizeof(float), stream));
|
||||
HIP_CHECK(
|
||||
hipMallocAsync(reinterpret_cast<void**>(&dev_ptr), array_size * sizeof(float), stream));
|
||||
}
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
HIP_CHECK(hipFree(dev_ptr));
|
||||
@@ -68,6 +69,6 @@ TEST_CASE("Performance_hipMallocAsync") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -73,8 +73,9 @@ static void RunBenchmark(const size_t array_size) {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMallocFromPoolAsync") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
size_t array_size = GENERATE(4_KB, 4_MB, 16_MB);
|
||||
@@ -82,6 +83,6 @@ TEST_CASE("Performance_hipMallocFromPoolAsync") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -31,9 +31,7 @@ class MemPoolCreateBenchmark : public Benchmark<MemPoolCreateBenchmark> {
|
||||
hipMemPool_t mem_pool{nullptr};
|
||||
hipMemPoolProps pool_props = CreateMemPoolProps(0, hipMemHandleTypeNone);
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props)); }
|
||||
|
||||
REQUIRE(mem_pool != nullptr);
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool));
|
||||
@@ -63,14 +61,15 @@ static void RunBenchmark() {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolCreate") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
RunBenchmark();
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -32,9 +32,7 @@ class MemPoolDestroyBenchmark : public Benchmark<MemPoolDestroyBenchmark> {
|
||||
hipMemPoolProps pool_props = CreateMemPoolProps(0, hipMemHandleTypeNone);
|
||||
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolDestroy(mem_pool)); }
|
||||
}
|
||||
};
|
||||
|
||||
@@ -62,14 +60,15 @@ static void RunBenchmark() {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolDestroy") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
RunBenchmark();
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -37,9 +37,7 @@ class MemPoolExportPointerBenchmark : public Benchmark<MemPoolExportPointerBench
|
||||
HIP_CHECK(hipMallocFromPoolAsync(&device_ptr, array_size * sizeof(float), mem_pool, nullptr));
|
||||
HIP_CHECK(hipStreamSynchronize(nullptr));
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolExportPointer(&exp_data, device_ptr));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolExportPointer(&exp_data, device_ptr)); }
|
||||
|
||||
HIP_CHECK(hipFreeAsync(device_ptr, nullptr));
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool));
|
||||
@@ -75,8 +73,9 @@ static void RunBenchmark(const size_t array_size) {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolExportPointer") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
size_t array_size = GENERATE(4_KB, 4_MB, 16_MB);
|
||||
@@ -84,6 +83,6 @@ TEST_CASE("Performance_hipMemPoolExportPointer") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -25,7 +25,8 @@ THE SOFTWARE.
|
||||
* @ingroup PerformanceTest
|
||||
*/
|
||||
|
||||
class MemPoolExportToShareableHandleBenchmark : public Benchmark<MemPoolExportToShareableHandleBenchmark> {
|
||||
class MemPoolExportToShareableHandleBenchmark
|
||||
: public Benchmark<MemPoolExportToShareableHandleBenchmark> {
|
||||
public:
|
||||
void operator()() {
|
||||
hipMemPool_t mem_pool{nullptr};
|
||||
@@ -66,14 +67,15 @@ static void RunBenchmark() {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolExportToShareableHandle") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
RunBenchmark();
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -33,13 +33,8 @@ class MemPoolGetAccessBenchmark : public Benchmark<MemPoolGetAccessBenchmark> {
|
||||
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
|
||||
|
||||
hipMemAccessFlags flags = hipMemAccessFlagsProtNone;
|
||||
hipMemLocation location = {
|
||||
hipMemLocationTypeDevice,
|
||||
0
|
||||
};
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolGetAccess(&flags, mem_pool, location));
|
||||
}
|
||||
hipMemLocation location = {hipMemLocationTypeDevice, 0};
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolGetAccess(&flags, mem_pool, location)); }
|
||||
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool));
|
||||
}
|
||||
@@ -68,14 +63,15 @@ static void RunBenchmark() {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolGetAccess") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
RunBenchmark();
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -34,9 +34,7 @@ class MemPoolGetAttributeBenchmark : public Benchmark<MemPoolGetAttributeBenchma
|
||||
|
||||
uint64_t value{0};
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolGetAttribute(mem_pool, attribute, &value));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolGetAttribute(mem_pool, attribute, &value)); }
|
||||
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool));
|
||||
}
|
||||
@@ -71,18 +69,18 @@ static void RunBenchmark(const hipMemPoolAttr attribute) {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolGetAttribute") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
hipMemPoolAttr attribute = GENERATE(hipMemPoolAttrReleaseThreshold,
|
||||
hipMemPoolReuseFollowEventDependencies,
|
||||
hipMemPoolReuseAllowOpportunistic,
|
||||
hipMemPoolReuseAllowInternalDependencies);
|
||||
hipMemPoolAttr attribute =
|
||||
GENERATE(hipMemPoolAttrReleaseThreshold, hipMemPoolReuseFollowEventDependencies,
|
||||
hipMemPoolReuseAllowOpportunistic, hipMemPoolReuseAllowInternalDependencies);
|
||||
RunBenchmark(attribute);
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -25,7 +25,8 @@ THE SOFTWARE.
|
||||
* @ingroup PerformanceTest
|
||||
*/
|
||||
|
||||
class MemPoolImportFromShareableHandleBenchmark : public Benchmark<MemPoolImportFromShareableHandleBenchmark> {
|
||||
class MemPoolImportFromShareableHandleBenchmark
|
||||
: public Benchmark<MemPoolImportFromShareableHandleBenchmark> {
|
||||
public:
|
||||
void operator()() {
|
||||
hipMemPool_t mem_pool{nullptr};
|
||||
@@ -41,8 +42,8 @@ class MemPoolImportFromShareableHandleBenchmark : public Benchmark<MemPoolImport
|
||||
HIP_CHECK(hipMemPoolExportToShareableHandle(&share_handle, mem_pool, kHandleType, 0));
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolImportFromShareableHandle(
|
||||
&mem_pool_shareable, (void*)share_handle, kHandleType, 0));
|
||||
HIP_CHECK(hipMemPoolImportFromShareableHandle(&mem_pool_shareable, (void*)share_handle,
|
||||
kHandleType, 0));
|
||||
}
|
||||
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool_shareable));
|
||||
@@ -74,14 +75,15 @@ static void RunBenchmark() {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolImportFromShareableHandle") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
RunBenchmark();
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -40,7 +40,8 @@ class MemPoolImportPointerBenchmark : public Benchmark<MemPoolImportPointerBench
|
||||
HIP_CHECK(hipMemPoolExportPointer(&exp_data, device_ptr));
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolImportPointer(reinterpret_cast<void**>(&device_ptr_import), mem_pool, &exp_data));
|
||||
HIP_CHECK(hipMemPoolImportPointer(reinterpret_cast<void**>(&device_ptr_import), mem_pool,
|
||||
&exp_data));
|
||||
}
|
||||
|
||||
HIP_CHECK(hipFree(device_ptr));
|
||||
@@ -78,8 +79,9 @@ static void RunBenchmark(const size_t array_size) {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolImportPointer") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
size_t array_size = GENERATE(4_KB, 4_MB, 16_MB);
|
||||
@@ -87,6 +89,6 @@ TEST_CASE("Performance_hipMemPoolImportPointer") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -32,17 +32,9 @@ class MemPoolSetAccessBenchmark : public Benchmark<MemPoolSetAccessBenchmark> {
|
||||
hipMemPoolProps pool_props = CreateMemPoolProps(0, hipMemHandleTypeNone);
|
||||
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
|
||||
|
||||
hipMemAccessDesc desc_list = {
|
||||
{
|
||||
hipMemLocationTypeDevice,
|
||||
0
|
||||
},
|
||||
hipMemAccessFlagsProtReadWrite
|
||||
};
|
||||
hipMemAccessDesc desc_list = {{hipMemLocationTypeDevice, 0}, hipMemAccessFlagsProtReadWrite};
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolSetAccess(mem_pool, &desc_list, 1));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolSetAccess(mem_pool, &desc_list, 1)); }
|
||||
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool));
|
||||
}
|
||||
@@ -71,14 +63,15 @@ static void RunBenchmark() {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolSetAccess") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
RunBenchmark();
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -34,9 +34,7 @@ class MemPoolSetAttributeBenchmark : public Benchmark<MemPoolSetAttributeBenchma
|
||||
|
||||
int value{0};
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolSetAttribute(mem_pool, attribute, &value));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolSetAttribute(mem_pool, attribute, &value)); }
|
||||
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool));
|
||||
}
|
||||
@@ -71,18 +69,18 @@ static void RunBenchmark(const hipMemPoolAttr attribute) {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolSetAttribute") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
hipMemPoolAttr attribute = GENERATE(hipMemPoolAttrReleaseThreshold,
|
||||
hipMemPoolReuseFollowEventDependencies,
|
||||
hipMemPoolReuseAllowOpportunistic,
|
||||
hipMemPoolReuseAllowInternalDependencies);
|
||||
hipMemPoolAttr attribute =
|
||||
GENERATE(hipMemPoolAttrReleaseThreshold, hipMemPoolReuseFollowEventDependencies,
|
||||
hipMemPoolReuseAllowOpportunistic, hipMemPoolReuseAllowInternalDependencies);
|
||||
RunBenchmark(attribute);
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -32,9 +32,7 @@ class MemPoolTrimToBenchmark : public Benchmark<MemPoolTrimToBenchmark> {
|
||||
hipMemPoolProps pool_props = CreateMemPoolProps(0, hipMemHandleTypeNone);
|
||||
HIP_CHECK(hipMemPoolCreate(&mem_pool, &pool_props));
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipMemPoolTrimTo(mem_pool, min_bytes_to_hold));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipMemPoolTrimTo(mem_pool, min_bytes_to_hold)); }
|
||||
|
||||
HIP_CHECK(hipMemPoolDestroy(mem_pool));
|
||||
}
|
||||
@@ -68,8 +66,9 @@ static void RunBenchmark(const size_t min_bytes_to_hold) {
|
||||
*/
|
||||
TEST_CASE("Performance_hipMemPoolTrimTo") {
|
||||
if (!AreMemPoolsSupported(0)) {
|
||||
HipTest::HIP_SKIP_TEST("GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
HipTest::HIP_SKIP_TEST(
|
||||
"GPU 0 doesn't support hipDeviceAttributeMemoryPoolsSupported "
|
||||
"attribute. Hence skipping the testing with Pass result.\n");
|
||||
return;
|
||||
}
|
||||
size_t min_bytes_to_hold = GENERATE(4_KB, 4_MB, 16_MB);
|
||||
@@ -77,6 +76,6 @@ TEST_CASE("Performance_hipMemPoolTrimTo") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -34,9 +34,7 @@ class StreamAddCallbackBenchmark : public Benchmark<StreamAddCallbackBenchmark>
|
||||
const StreamGuard stream_guard{Streams::created};
|
||||
const hipStream_t stream = stream_guard.stream();
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipStreamAddCallback(stream, Callback, nullptr, 0));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipStreamAddCallback(stream, Callback, nullptr, 0)); }
|
||||
}
|
||||
};
|
||||
|
||||
@@ -56,11 +54,9 @@ static void RunBenchmark() {
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.2
|
||||
*/
|
||||
TEST_CASE("Performance_hipStreamAddCallback") {
|
||||
RunBenchmark();
|
||||
}
|
||||
TEST_CASE("Performance_hipStreamAddCallback") { RunBenchmark(); }
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -27,12 +27,15 @@ THE SOFTWARE.
|
||||
* @ingroup PerformanceTest
|
||||
* Contains performance tests for all hipStream related APIs
|
||||
*/
|
||||
|
||||
class HipDeviceGetStreamPriorityRangeBenchmark : public Benchmark<HipDeviceGetStreamPriorityRangeBenchmark> {
|
||||
|
||||
class HipDeviceGetStreamPriorityRangeBenchmark
|
||||
: public Benchmark<HipDeviceGetStreamPriorityRangeBenchmark> {
|
||||
public:
|
||||
void operator()() {
|
||||
int priority_min, priority_max;
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipDeviceGetStreamPriorityRange(&priority_min, &priority_max)); }
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipDeviceGetStreamPriorityRange(&priority_min, &priority_max));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -42,19 +45,19 @@ class HipStreamQueryBenchmark : public Benchmark<HipStreamQueryBenchmark> {
|
||||
hipError_t error;
|
||||
hipStream_t stream;
|
||||
HIP_CHECK(hipStreamCreate(&stream));
|
||||
void *dptr;
|
||||
|
||||
if(perform_work) {
|
||||
void* dptr;
|
||||
|
||||
if (perform_work) {
|
||||
HIP_CHECK(hipMallocAsync(&dptr, 2048 * 4, stream));
|
||||
}
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) { error = hipStreamQuery(stream); }
|
||||
|
||||
if(perform_work) {
|
||||
|
||||
if (perform_work) {
|
||||
HIP_CHECK(hipFreeAsync(dptr, stream));
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
}
|
||||
|
||||
|
||||
HIP_CHECK(hipStreamDestroy(stream));
|
||||
}
|
||||
};
|
||||
@@ -65,9 +68,9 @@ class HipStreamSynchronizeBenchmark : public Benchmark<HipStreamSynchronizeBench
|
||||
hipError_t error;
|
||||
hipStream_t stream;
|
||||
HIP_CHECK(hipStreamCreate(&stream));
|
||||
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) { error = hipStreamSynchronize(stream); }
|
||||
|
||||
|
||||
HIP_CHECK(hipStreamDestroy(stream));
|
||||
}
|
||||
};
|
||||
@@ -93,23 +96,25 @@ class HipStreamCreateBenchmark : public Benchmark<HipStreamCreateBenchmark> {
|
||||
}
|
||||
};
|
||||
|
||||
class HipStreamCreateWithPriorityBenchmark : public Benchmark<HipStreamCreateWithPriorityBenchmark> {
|
||||
class HipStreamCreateWithPriorityBenchmark
|
||||
: public Benchmark<HipStreamCreateWithPriorityBenchmark> {
|
||||
public:
|
||||
void operator()(unsigned int flag) {
|
||||
hipStream_t stream;
|
||||
int priority_min, priority_max, priority_mid;
|
||||
|
||||
|
||||
HIP_CHECK(hipDeviceGetStreamPriorityRange(&priority_min, &priority_max));
|
||||
priority_mid = (priority_max + priority_min) / 2;
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipStreamCreateWithPriority(&stream, flag, priority_mid)); }
|
||||
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipStreamCreateWithPriority(&stream, flag, priority_mid));
|
||||
}
|
||||
|
||||
HIP_CHECK(hipStreamDestroy(stream));
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
static std::string GetStreamCreateFlagName(unsigned flag) {
|
||||
switch (flag) {
|
||||
case hipStreamDefault:
|
||||
@@ -244,7 +249,7 @@ TEST_CASE("Performance_hipDeviceGetStreamPriorityRange") {
|
||||
TEST_CASE("Performance_hipStreamQuery") {
|
||||
const auto perform_work = GENERATE(true, false);
|
||||
HipStreamQueryBenchmark benchmark;
|
||||
if(perform_work) {
|
||||
if (perform_work) {
|
||||
benchmark.AddSectionName("stream with work");
|
||||
} else {
|
||||
benchmark.AddSectionName("stream without work");
|
||||
@@ -269,6 +274,6 @@ TEST_CASE("Performance_hipStreamSynchronize") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -33,10 +33,8 @@ class StreamGetFlagsBenchmark : public Benchmark<StreamGetFlagsBenchmark> {
|
||||
hipStream_t stream;
|
||||
|
||||
HIP_CHECK(hipStreamCreateWithFlags(&stream, expected_flag));
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipStreamGetFlags(stream, &returned_flags))
|
||||
}
|
||||
HIP_CHECK(hipStreamDestroy(stream));
|
||||
TIMED_SECTION(kTimerTypeCpu){HIP_CHECK(hipStreamGetFlags(stream, &returned_flags))} HIP_CHECK(
|
||||
hipStreamDestroy(stream));
|
||||
}
|
||||
};
|
||||
|
||||
@@ -75,6 +73,6 @@ TEST_CASE("Performance_hipStreamGetFlags") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -33,9 +33,7 @@ class StreamGetPriorityBenchmark : public Benchmark<StreamGetPriorityBenchmark>
|
||||
const hipStream_t stream = stream_guard.stream();
|
||||
|
||||
int priority{};
|
||||
TIMED_SECTION(kTimerTypeCpu) {
|
||||
HIP_CHECK(hipStreamGetPriority(stream, &priority));
|
||||
}
|
||||
TIMED_SECTION(kTimerTypeCpu) { HIP_CHECK(hipStreamGetPriority(stream, &priority)); }
|
||||
}
|
||||
};
|
||||
|
||||
@@ -74,6 +72,6 @@ TEST_CASE("Performance_hipStreamGetPriority") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -80,6 +80,6 @@ TEST_CASE("Performance_hipStreamWaitEvent") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -172,6 +172,6 @@ TEST_CASE("Performance_hipStreamWaitValue64") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -123,6 +123,6 @@ TEST_CASE("Performance_hipStreamWriteValue64") {
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group PerformanceTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -33,73 +33,49 @@ THE SOFTWARE.
|
||||
#include <map>
|
||||
|
||||
/**
|
||||
* @addtogroup __reduce_op_sync __reduce_op_sync
|
||||
* @{
|
||||
* @ingroup WarpSyncPerformance
|
||||
* __reduce_op_sync(MaskT mask, T val)
|
||||
* Reduces the val as per the lanes described in mask and calculates the
|
||||
* aggregated result
|
||||
*/
|
||||
* @addtogroup __reduce_op_sync __reduce_op_sync
|
||||
* @{
|
||||
* @ingroup WarpSyncPerformance
|
||||
* __reduce_op_sync(MaskT mask, T val)
|
||||
* Reduces the val as per the lanes described in mask and calculates the
|
||||
* aggregated result
|
||||
*/
|
||||
|
||||
static constexpr int kBlockDim = 1024;
|
||||
|
||||
template <class T>
|
||||
struct AtomicAddOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs)
|
||||
{
|
||||
return atomicAdd(lhs, rhs);
|
||||
}
|
||||
template <class T> struct AtomicAddOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs) { return atomicAdd(lhs, rhs); }
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct AtomicMinOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs)
|
||||
{
|
||||
return atomicMin(lhs, rhs);
|
||||
}
|
||||
template <class T> struct AtomicMinOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs) { return atomicMin(lhs, rhs); }
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct AtomicMaxOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs)
|
||||
{
|
||||
return atomicMax(lhs, rhs);
|
||||
}
|
||||
template <class T> struct AtomicMaxOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs) { return atomicMax(lhs, rhs); }
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct AtomicAndOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs)
|
||||
{
|
||||
return atomicAnd(lhs, rhs);
|
||||
}
|
||||
template <class T> struct AtomicAndOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs) { return atomicAnd(lhs, rhs); }
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct AtomicOrOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs)
|
||||
{
|
||||
return atomicOr(lhs, rhs);
|
||||
}
|
||||
template <class T> struct AtomicOrOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs) { return atomicOr(lhs, rhs); }
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct AtomicXorOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs)
|
||||
{
|
||||
return atomicXor(lhs, rhs);
|
||||
}
|
||||
template <class T> struct AtomicXorOp {
|
||||
__device__ T operator()(T* lhs, const T& rhs) { return atomicXor(lhs, rhs); }
|
||||
};
|
||||
|
||||
// uses atomics to reduce the whole warp; depending on the mask our reduce should be faster
|
||||
// @output to store the result, one per warp
|
||||
// @numItems must be a multiple of warpSize
|
||||
template <class T, template <typename> class Op>
|
||||
__global__ void reduceAllAtomics(T* __restrict__ output, const T* __restrict__ input, unsigned long long mask)
|
||||
{
|
||||
__global__ void reduceAllAtomics(T* __restrict__ output, const T* __restrict__ input,
|
||||
unsigned long long mask) {
|
||||
int idx = threadIdx.x + blockIdx.x * kBlockDim;
|
||||
extern __shared__ uint8_t shared_mem[];
|
||||
T* result = reinterpret_cast<T*>(shared_mem); // one per warp
|
||||
T* result = reinterpret_cast<T*>(shared_mem); // one per warp
|
||||
Op<T> op;
|
||||
int numWarp = threadIdx.x / warpSize;
|
||||
|
||||
@@ -115,18 +91,16 @@ __global__ void reduceAllAtomics(T* __restrict__ output, const T* __restrict__ i
|
||||
|
||||
__syncthreads();
|
||||
|
||||
if (mask & (1ul << __ockl_lane_u32()))
|
||||
op(&result[numWarp], input[idx]);
|
||||
if (mask & (1ul << __ockl_lane_u32())) op(&result[numWarp], input[idx]);
|
||||
|
||||
__syncthreads();
|
||||
|
||||
if (__ockl_lane_u32() == 0)
|
||||
output[idx / warpSize] = result[numWarp];
|
||||
if (__ockl_lane_u32() == 0) output[idx / warpSize] = result[numWarp];
|
||||
}
|
||||
|
||||
template <class T, template<typename> class Op>
|
||||
__global__ void reduceOpSync(T* __restrict__ output, const T* __restrict__ input, unsigned long long mask)
|
||||
{
|
||||
template <class T, template <typename> class Op>
|
||||
__global__ void reduceOpSync(T* __restrict__ output, const T* __restrict__ input,
|
||||
unsigned long long mask) {
|
||||
int idx = threadIdx.x + blockIdx.x * kBlockDim;
|
||||
T result;
|
||||
|
||||
@@ -146,18 +120,16 @@ __global__ void reduceOpSync(T* __restrict__ output, const T* __restrict__ input
|
||||
else
|
||||
static_assert(std::is_void<T>::value, "Unsupported operator");
|
||||
|
||||
if (__ockl_activelane_u32() == 0)
|
||||
output[idx / warpSize] = result;
|
||||
if (__ockl_activelane_u32() == 0) output[idx / warpSize] = result;
|
||||
}
|
||||
}
|
||||
|
||||
template <class T, template <typename> class Op>
|
||||
class AtomicBenchmark : public Benchmark<AtomicBenchmark<T, Op>> {
|
||||
public:
|
||||
void operator()(T* output, const T* input, int numItems, unsigned long long mask)
|
||||
{
|
||||
dim3 blockDim = { kBlockDim };
|
||||
dim3 gridDim = { static_cast<uint32_t>(std::ceil(numItems / static_cast<float>(blockDim.x))) };
|
||||
public:
|
||||
void operator()(T* output, const T* input, int numItems, unsigned long long mask) {
|
||||
dim3 blockDim = {kBlockDim};
|
||||
dim3 gridDim = {static_cast<uint32_t>(std::ceil(numItems / static_cast<float>(blockDim.x)))};
|
||||
|
||||
hipDeviceProp_t props;
|
||||
HIP_CHECK(hipGetDeviceProperties(&props, 0));
|
||||
@@ -187,11 +159,10 @@ public:
|
||||
|
||||
template <class T, template <typename> class Op>
|
||||
class ReduceSyncBenchmark : public Benchmark<ReduceSyncBenchmark<T, Op>> {
|
||||
public:
|
||||
void operator()(T* output, T* input, int numItems, unsigned long long mask)
|
||||
{
|
||||
dim3 blockDim = { kBlockDim };
|
||||
dim3 gridDim = { static_cast<uint32_t>(std::ceil(numItems / static_cast<float>(blockDim.x))) };
|
||||
public:
|
||||
void operator()(T* output, T* input, int numItems, unsigned long long mask) {
|
||||
dim3 blockDim = {kBlockDim};
|
||||
dim3 gridDim = {static_cast<uint32_t>(std::ceil(numItems / static_cast<float>(blockDim.x)))};
|
||||
|
||||
|
||||
TIMED_SECTION(kTimerTypeEvent) {
|
||||
@@ -202,8 +173,7 @@ public:
|
||||
};
|
||||
|
||||
template <class T, template <typename> class Op>
|
||||
void checkResults(T* d_atomicsResult, T* d_reduceResult, size_t numBytes, unsigned long long mask)
|
||||
{
|
||||
void checkResults(T* d_atomicsResult, T* d_reduceResult, size_t numBytes, unsigned long long mask) {
|
||||
using namespace Catch::Matchers;
|
||||
LinearAllocGuard<T> outputAtomic(LinearAllocs::malloc, numBytes);
|
||||
LinearAllocGuard<T> outputReduce(LinearAllocs::malloc, numBytes);
|
||||
@@ -229,47 +199,38 @@ void checkResults(T* d_atomicsResult, T* d_reduceResult, size_t numBytes, unsign
|
||||
}
|
||||
}
|
||||
|
||||
template <class T, template <typename> class Op>
|
||||
struct IsLogicalOp {
|
||||
template <class T, template <typename> class Op> struct IsLogicalOp {
|
||||
static constexpr bool value = false;
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct IsLogicalOp<T, std::logical_and> {
|
||||
template <class T> struct IsLogicalOp<T, std::logical_and> {
|
||||
static constexpr bool value = true;
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct IsLogicalOp<T, std::logical_or> {
|
||||
template <class T> struct IsLogicalOp<T, std::logical_or> {
|
||||
static constexpr bool value = true;
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct IsLogicalOp<T, XorOp> {
|
||||
template <class T> struct IsLogicalOp<T, XorOp> {
|
||||
static constexpr bool value = true;
|
||||
};
|
||||
|
||||
// Neither long long or fp16 have atomic operations. In those cases
|
||||
// we only benchmark reduce sync operations, we cannot compare with native atomics
|
||||
template <class T>
|
||||
struct HasAtomicOps {
|
||||
template <class T> struct HasAtomicOps {
|
||||
static constexpr bool value = true;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct HasAtomicOps<half> {
|
||||
template <> struct HasAtomicOps<half> {
|
||||
static constexpr bool value = false;
|
||||
};
|
||||
|
||||
template <>
|
||||
struct HasAtomicOps<long long> {
|
||||
template <> struct HasAtomicOps<long long> {
|
||||
static constexpr bool value = false;
|
||||
};
|
||||
|
||||
template <class T, template <typename> class Op>
|
||||
struct ReduceBenchmark {
|
||||
void Run()
|
||||
{
|
||||
template <class T, template <typename> class Op> struct ReduceBenchmark {
|
||||
void Run() {
|
||||
static constexpr int numMasks = 6;
|
||||
using distribution = typename DistributionType<T>::type;
|
||||
ReduceSyncBenchmark<T, Op> benchmarkReduce;
|
||||
@@ -287,27 +248,27 @@ struct ReduceBenchmark {
|
||||
distribution dist;
|
||||
int halfWaveSize = wavefrontSize / 2;
|
||||
unsigned long long halfBitsOn = (1ul << (wavefrontSize / 2)) - 1;
|
||||
unsigned long long fullMask = -1ul,
|
||||
halfHighBitsOn = halfBitsOn << halfWaveSize,
|
||||
unsigned long long fullMask = -1ul, halfHighBitsOn = halfBitsOn << halfWaveSize,
|
||||
high16BitsOn = halfBitsOn << (wavefrontSize - 16),
|
||||
high8BitsOn = halfBitsOn << (wavefrontSize - 8),
|
||||
high4BitsOn = halfBitsOn << (wavefrontSize - 4),
|
||||
allButOne = -1 & ~1;
|
||||
high4BitsOn = halfBitsOn << (wavefrontSize - 4), allButOne = -1 & ~1;
|
||||
const char* typeStr = typeToString<T>();
|
||||
const char* opStr = opToString<T, Op>();
|
||||
std::map<std::string, unsigned long long> masks;
|
||||
std::pair<std::string, unsigned long long> masksPairs[] = { { "full mask", fullMask },
|
||||
{ "high order 32 bits on", halfHighBitsOn },
|
||||
{ "high order 16 bits on", high16BitsOn },
|
||||
{ "high order 8 bits on", high8BitsOn },
|
||||
{ "high order 4 bits on", high4BitsOn },
|
||||
{ "all but one", allButOne } };
|
||||
std::pair<std::string, unsigned long long> masksPairs[] = {
|
||||
{"full mask", fullMask},
|
||||
{"high order 32 bits on", halfHighBitsOn},
|
||||
{"high order 16 bits on", high16BitsOn},
|
||||
{"high order 8 bits on", high8BitsOn},
|
||||
{"high order 4 bits on", high4BitsOn},
|
||||
{"all but one", allButOne}};
|
||||
int pos = 0, numMask = 0;
|
||||
|
||||
for (auto& mask : masksPairs) {
|
||||
// don't use 'halfHighBitsOn' on warp size 32; it's the same as high16BitsOn
|
||||
if (wavefrontSize != 32 || mask.second != halfHighBitsOn) {
|
||||
masks.emplace(std::to_string(numMask) + " - " + mask.first, wavefrontSize == 64? mask.second : mask.second & 0xFFFFFFFF);
|
||||
masks.emplace(std::to_string(numMask) + " - " + mask.first,
|
||||
wavefrontSize == 64 ? mask.second : mask.second & 0xFFFFFFFF);
|
||||
numMask++;
|
||||
}
|
||||
}
|
||||
@@ -315,8 +276,7 @@ struct ReduceBenchmark {
|
||||
// avoid generating values different than 1 or 0 for logical operators;
|
||||
// otherwise the atomic version of the kernels would produce different results as
|
||||
// atomicAnd/Or() are bitwise operations, not logical
|
||||
if constexpr (IsLogicalOp<T, Op>::value)
|
||||
dist = distribution(0, 1);
|
||||
if constexpr (IsLogicalOp<T, Op>::value) dist = distribution(0, 1);
|
||||
|
||||
for (int i = 0; i < numItems; i++) {
|
||||
input.ptr()[i] = dist(gen);
|
||||
@@ -356,7 +316,8 @@ struct ReduceBenchmark {
|
||||
printf("Checking results...\n");
|
||||
|
||||
for (const auto& mask : masks) {
|
||||
checkResults<T, Op>(d_outputsAtomic[pos].ptr(), d_outputsReduce[pos].ptr(), outputNumBytes, mask.second);
|
||||
checkResults<T, Op>(d_outputsAtomic[pos].ptr(), d_outputsReduce[pos].ptr(), outputNumBytes,
|
||||
mask.second);
|
||||
pos++;
|
||||
}
|
||||
}
|
||||
@@ -370,31 +331,36 @@ TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Add", "", int, unsigned int, unsigne
|
||||
benchmark.Run();
|
||||
}
|
||||
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Min", "", int, unsigned int, unsigned long long, long long, float, half, double) {
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Min", "", int, unsigned int, unsigned long long,
|
||||
long long, float, half, double) {
|
||||
ReduceBenchmark<TestType, MinOp> benchmark;
|
||||
|
||||
benchmark.Run();
|
||||
}
|
||||
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Max", "", int, unsigned int, unsigned long long, long long, float, half, double) {
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Max", "", int, unsigned int, unsigned long long,
|
||||
long long, float, half, double) {
|
||||
ReduceBenchmark<TestType, MaxOp> benchmark;
|
||||
|
||||
benchmark.Run();
|
||||
}
|
||||
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_And", "", int, unsigned int, unsigned long long, long long) {
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_And", "", int, unsigned int, unsigned long long,
|
||||
long long) {
|
||||
ReduceBenchmark<TestType, std::logical_and> benchmark;
|
||||
|
||||
benchmark.Run();
|
||||
}
|
||||
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Or", "", int, unsigned int, unsigned long long, long long) {
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Or", "", int, unsigned int, unsigned long long,
|
||||
long long) {
|
||||
ReduceBenchmark<TestType, std::logical_or> benchmark;
|
||||
|
||||
benchmark.Run();
|
||||
}
|
||||
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Xor", "", int, unsigned int, unsigned long long, long long) {
|
||||
TEMPLATE_TEST_CASE("Performance_Reduce_Sync_Xor", "", int, unsigned int, unsigned long long,
|
||||
long long) {
|
||||
ReduceBenchmark<TestType, XorOp> benchmark;
|
||||
|
||||
benchmark.Run();
|
||||
|
||||
@@ -21,13 +21,13 @@ THE SOFTWARE.
|
||||
#include <hip_test_defgroups.hh>
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
#define HIP_CHECK_PERF(a) \
|
||||
{ \
|
||||
auto err = a; \
|
||||
if ((err != hipSuccess) && (err != hipErrorNotReady)) { \
|
||||
printf(#a "= Error! %s\n", hipGetErrorString(err)); \
|
||||
exit(1); \
|
||||
} \
|
||||
#define HIP_CHECK_PERF(a) \
|
||||
{ \
|
||||
auto err = a; \
|
||||
if ((err != hipSuccess) && (err != hipErrorNotReady)) { \
|
||||
printf(#a "= Error! %s\n", hipGetErrorString(err)); \
|
||||
exit(1); \
|
||||
} \
|
||||
}
|
||||
/**
|
||||
* @addtogroup hipEventRecord hipEventRecord
|
||||
@@ -40,7 +40,7 @@ __global__ void null_kernel() {
|
||||
__shared__ int temp[256];
|
||||
temp[threadIdx.x] = sinf(float(threadIdx.x));
|
||||
}
|
||||
void rocm_null_gpu_job(void *stream) {
|
||||
void rocm_null_gpu_job(void* stream) {
|
||||
hipLaunchKernelGGL(null_kernel, 1, 256, 0, (hipStream_t)stream);
|
||||
}
|
||||
std::vector<std::vector<hipStream_t>> stream_pool;
|
||||
@@ -48,12 +48,12 @@ std::atomic<int> counter(0);
|
||||
bool do_kill = false;
|
||||
std::chrono::system_clock::time_point thread_reports[16];
|
||||
void thread_job(int dev, int virt) {
|
||||
HIP_CHECK_PERF(hipSetDevice(dev)); // use dev
|
||||
uint8_t *mem;
|
||||
HIP_CHECK_PERF(hipSetDevice(dev)); // use dev
|
||||
uint8_t* mem;
|
||||
HIP_CHECK_PERF(hipMalloc(&mem, 512));
|
||||
void *hmem2;
|
||||
void* hmem2;
|
||||
HIP_CHECK_PERF(hipHostAlloc(&hmem2, 512, 0));
|
||||
uint8_t *hmem = (uint8_t *)hmem2;
|
||||
uint8_t* hmem = (uint8_t*)hmem2;
|
||||
hipStream_t exec_stream = stream_pool[dev][virt];
|
||||
hipStream_t h2d_stream = stream_pool[dev][virt + 4];
|
||||
hipStream_t d2h_stream = stream_pool[dev][virt + 8];
|
||||
@@ -63,10 +63,8 @@ void thread_job(int dev, int virt) {
|
||||
uint64_t n = 0;
|
||||
while (!do_kill) {
|
||||
rocm_null_gpu_job(exec_stream);
|
||||
HIP_CHECK_PERF(
|
||||
hipMemcpyAsync(hmem, mem, 4, hipMemcpyDeviceToHost, d2h_stream));
|
||||
HIP_CHECK_PERF(hipMemcpyAsync(mem + 256, hmem + 256, 4,
|
||||
hipMemcpyHostToDevice, h2d_stream));
|
||||
HIP_CHECK_PERF(hipMemcpyAsync(hmem, mem, 4, hipMemcpyDeviceToHost, d2h_stream));
|
||||
HIP_CHECK_PERF(hipMemcpyAsync(mem + 256, hmem + 256, 4, hipMemcpyHostToDevice, h2d_stream));
|
||||
HIP_CHECK_PERF(hipEventRecord(eh2d, h2d_stream));
|
||||
HIP_CHECK_PERF(hipEventRecord(ed2h, d2h_stream));
|
||||
HIP_CHECK_PERF(hipEventQuery(eh2d));
|
||||
@@ -109,14 +107,13 @@ TEST_CASE("Unit_hipEventOverFlow_PerfTest") {
|
||||
HIP_CHECK_PERF(hipGetDeviceCount(&mgpu));
|
||||
stream_pool.resize(mgpu);
|
||||
HIP_CHECK_PERF(hipSetDeviceFlags(hipDeviceScheduleSpin));
|
||||
std::vector<uint8_t *> memory_buffers[2];
|
||||
std::vector<uint8_t*> memory_buffers[2];
|
||||
for (int i = 0; i < mgpu; i++) {
|
||||
HIP_CHECK_PERF(hipSetDevice(i));
|
||||
stream_pool[i].resize(12);
|
||||
memory_buffers[i].resize(128);
|
||||
for (int j = 0; j < 12; j++)
|
||||
HIP_CHECK_PERF(
|
||||
hipStreamCreateWithFlags(&stream_pool[i][j], hipStreamNonBlocking));
|
||||
HIP_CHECK_PERF(hipStreamCreateWithFlags(&stream_pool[i][j], hipStreamNonBlocking));
|
||||
for (int j = 0; j < 128; j++)
|
||||
HIP_CHECK_PERF(hipMalloc(&memory_buffers[i][j], 4096 * ((j & 1) + 1)));
|
||||
}
|
||||
@@ -125,8 +122,7 @@ TEST_CASE("Unit_hipEventOverFlow_PerfTest") {
|
||||
printf("RUNNING ON %d DEVICES\n", nDev);
|
||||
do_kill = false;
|
||||
std::vector<std::thread> threads;
|
||||
for (int i = 0; i < nDev * 4; i++)
|
||||
threads.push_back(std::thread(thread_job, i / 4, i % 4));
|
||||
for (int i = 0; i < nDev * 4; i++) threads.push_back(std::thread(thread_job, i / 4, i % 4));
|
||||
usleep(1000000);
|
||||
auto t1 = std::chrono::system_clock::now();
|
||||
int count = int(counter);
|
||||
@@ -135,15 +131,12 @@ TEST_CASE("Unit_hipEventOverFlow_PerfTest") {
|
||||
for (int t = 0; t < 10; t++) {
|
||||
usleep(1000000);
|
||||
auto t2 = std::chrono::system_clock::now();
|
||||
auto duration =
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(t2 - t1)
|
||||
.count();
|
||||
auto duration = std::chrono::duration_cast<std::chrono::microseconds>(t2 - t1).count();
|
||||
int count2 = int(counter);
|
||||
for (int i = 0; i < nDev * 4; i++) {
|
||||
if (std::chrono::duration_cast<std::chrono::microseconds>(
|
||||
t2 - thread_reports[i])
|
||||
.count() >= 1000000) {
|
||||
printf("Thread %d/%d is stuck\n", i/4, i%4);
|
||||
if (std::chrono::duration_cast<std::chrono::microseconds>(t2 - thread_reports[i]).count() >=
|
||||
1000000) {
|
||||
printf("Thread %d/%d is stuck\n", i / 4, i % 4);
|
||||
}
|
||||
}
|
||||
total_count += count2 - count;
|
||||
@@ -153,8 +146,7 @@ TEST_CASE("Unit_hipEventOverFlow_PerfTest") {
|
||||
}
|
||||
printf("AVERAGE: %ld / %f = %f job/s\n", total_count, total_time, total_count / total_time);
|
||||
do_kill = true;
|
||||
for (auto &t : threads)
|
||||
t.join();
|
||||
for (auto& t : threads) t.join();
|
||||
for (int i = 0; i < nDev; i++) {
|
||||
HIP_CHECK_PERF(hipSetDevice(i));
|
||||
HIP_CHECK_PERF(hipDeviceSynchronize());
|
||||
|
||||
@@ -21,13 +21,13 @@ THE SOFTWARE.
|
||||
#include <hip_test_defgroups.hh>
|
||||
#include <unistd.h>
|
||||
#include <vector>
|
||||
#define HIP_CHECK_PERF(a) \
|
||||
{ \
|
||||
auto err = a; \
|
||||
if ((err != hipSuccess) && (err != hipErrorNotReady)) { \
|
||||
printf(#a "= Error! %s\n", hipGetErrorString(err)); \
|
||||
exit(1); \
|
||||
} \
|
||||
#define HIP_CHECK_PERF(a) \
|
||||
{ \
|
||||
auto err = a; \
|
||||
if ((err != hipSuccess) && (err != hipErrorNotReady)) { \
|
||||
printf(#a "= Error! %s\n", hipGetErrorString(err)); \
|
||||
exit(1); \
|
||||
} \
|
||||
}
|
||||
/**
|
||||
* @addtogroup hipLaunchKernelGGL hipLaunchKernelGGL
|
||||
@@ -38,7 +38,7 @@ __global__ void empty_kernel() {
|
||||
__shared__ int temp[256];
|
||||
temp[threadIdx.x] = sinf(float(threadIdx.x));
|
||||
}
|
||||
void rocm_empty_gpu_job(void *stream) {
|
||||
void rocm_empty_gpu_job(void* stream) {
|
||||
hipLaunchKernelGGL(empty_kernel, 1, 256, 0, (hipStream_t)stream);
|
||||
}
|
||||
std::vector<std::vector<hipStream_t>> stream_pools;
|
||||
@@ -84,16 +84,14 @@ TEST_CASE("Unit_hipKernelLookUp_PerfTest") {
|
||||
HIP_CHECK_PERF(hipSetDevice(i));
|
||||
stream_pools[i].resize(12);
|
||||
for (int j = 0; j < 12; j++)
|
||||
HIP_CHECK_PERF(
|
||||
hipStreamCreateWithFlags(&stream_pools[i][j], hipStreamNonBlocking));
|
||||
HIP_CHECK_PERF(hipStreamCreateWithFlags(&stream_pools[i][j], hipStreamNonBlocking));
|
||||
}
|
||||
for (int nDev = 1; nDev <= mgpu; nDev++) {
|
||||
count = 0;
|
||||
INFO("RUNNING ON "<<nDev<<" DEVICES\n");
|
||||
INFO("RUNNING ON " << nDev << " DEVICES\n");
|
||||
kill = false;
|
||||
std::vector<std::thread> threads;
|
||||
for (int i = 0; i < nDev * 4; i++)
|
||||
threads.push_back(std::thread(thread_jobs, i / 4, i % 4));
|
||||
for (int i = 0; i < nDev * 4; i++) threads.push_back(std::thread(thread_jobs, i / 4, i % 4));
|
||||
usleep(1000000);
|
||||
auto t1 = std::chrono::system_clock::now();
|
||||
int counter = int(count);
|
||||
@@ -102,15 +100,12 @@ TEST_CASE("Unit_hipKernelLookUp_PerfTest") {
|
||||
for (int t = 0; t < 10; t++) {
|
||||
usleep(1000000);
|
||||
auto t2 = std::chrono::system_clock::now();
|
||||
auto duration =
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(t2 - t1)
|
||||
.count();
|
||||
auto duration = std::chrono::duration_cast<std::chrono::microseconds>(t2 - t1).count();
|
||||
int counter2 = int(count);
|
||||
for (int i = 0; i < nDev * 4; i++) {
|
||||
if (std::chrono::duration_cast<std::chrono::microseconds>(
|
||||
t2 - thread_report[i])
|
||||
.count() >= 1000000) {
|
||||
INFO("Thread "<<i/4<<"/"<<i%4<<" is stuck\n");
|
||||
if (std::chrono::duration_cast<std::chrono::microseconds>(t2 - thread_report[i]).count() >=
|
||||
1000000) {
|
||||
INFO("Thread " << i / 4 << "/" << i % 4 << " is stuck\n");
|
||||
}
|
||||
}
|
||||
total_count += counter2 - counter;
|
||||
@@ -118,10 +113,9 @@ TEST_CASE("Unit_hipKernelLookUp_PerfTest") {
|
||||
t1 = t2;
|
||||
counter = counter2;
|
||||
}
|
||||
INFO("AVERAGE: "<<total_count<<"/"<<total_time<<" = "<<total_count / total_time);
|
||||
INFO("AVERAGE: " << total_count << "/" << total_time << " = " << total_count / total_time);
|
||||
kill = true;
|
||||
for (auto &t : threads)
|
||||
t.join();
|
||||
for (auto& t : threads) t.join();
|
||||
for (int i = 0; i < nDev; i++) {
|
||||
HIP_CHECK_PERF(hipSetDevice(i));
|
||||
HIP_CHECK_PERF(hipDeviceSynchronize());
|
||||
|
||||
@@ -39,7 +39,7 @@ static constexpr int launches = 5;
|
||||
/**
|
||||
* In fillKernel, all elements of the array filled with given value
|
||||
*/
|
||||
static __global__ void fillKernel(int *arr, int size, int value) {
|
||||
static __global__ void fillKernel(int* arr, int size, int value) {
|
||||
int offset = blockDim.x * blockIdx.x + threadIdx.x;
|
||||
int stride = blockDim.x * gridDim.x;
|
||||
for (int i = offset; i < size; i += stride) {
|
||||
@@ -50,7 +50,7 @@ static __global__ void fillKernel(int *arr, int size, int value) {
|
||||
/**
|
||||
* In addOneKernel, all elements of the array are incremented by 1
|
||||
*/
|
||||
static __global__ void addOneKernel(int *arr, int size) {
|
||||
static __global__ void addOneKernel(int* arr, int size) {
|
||||
int offset = blockDim.x * blockIdx.x + threadIdx.x;
|
||||
int stride = blockDim.x * gridDim.x;
|
||||
for (int i = offset; i < size; i += stride) {
|
||||
@@ -62,7 +62,7 @@ static __global__ void addOneKernel(int *arr, int size) {
|
||||
* In addKernel, Array1 and Array2 will be added by element wise
|
||||
* and stored in Array 1
|
||||
*/
|
||||
static __global__ void addKernel(int *arr1, int *arr2, int size) {
|
||||
static __global__ void addKernel(int* arr1, int* arr2, int size) {
|
||||
int offset = blockDim.x * blockIdx.x + threadIdx.x;
|
||||
int stride = blockDim.x * gridDim.x;
|
||||
for (int i = offset; i < size; i += stride) {
|
||||
@@ -93,7 +93,7 @@ static __global__ void addKernel(int *arr1, int *arr2, int size) {
|
||||
TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
|
||||
constexpr int numberOfNodes = 1024;
|
||||
|
||||
int *devMem[numberOfNodes];
|
||||
int* devMem[numberOfNodes];
|
||||
for (int i = 0; i < numberOfNodes; i++) {
|
||||
devMem[i] = nullptr;
|
||||
}
|
||||
@@ -116,17 +116,15 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
|
||||
memAllocNodeParams.bytesize = sizeof(int);
|
||||
|
||||
if (i == 0) {
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0,
|
||||
&memAllocNodeParams));
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0, &memAllocNodeParams));
|
||||
} else {
|
||||
::std::vector<hipGraphNode_t> memAllocNodeDependencies;
|
||||
memAllocNodeDependencies.push_back(memAllocNode[i - 1]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(
|
||||
&memAllocNode[i], graph, memAllocNodeDependencies.data(),
|
||||
memAllocNodeDependencies.size(), &memAllocNodeParams));
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, memAllocNodeDependencies.data(),
|
||||
memAllocNodeDependencies.size(), &memAllocNodeParams));
|
||||
}
|
||||
devMem[i] = reinterpret_cast<int *>(memAllocNodeParams.dptr);
|
||||
devMem[i] = reinterpret_cast<int*>(memAllocNodeParams.dptr);
|
||||
REQUIRE(devMem[i] != nullptr);
|
||||
}
|
||||
|
||||
@@ -136,16 +134,16 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
|
||||
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
|
||||
memFreeNodeDependencies.push_back(memAllocNode[numberOfNodes - 1]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(
|
||||
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(), reinterpret_cast<void *>(devMem[i])));
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(),
|
||||
reinterpret_cast<void*>(devMem[i])));
|
||||
} else {
|
||||
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
|
||||
memFreeNodeDependencies.push_back(memFreeNode[i - 1]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(
|
||||
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(), reinterpret_cast<void *>(devMem[i])));
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(),
|
||||
reinterpret_cast<void*>(devMem[i])));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -165,27 +163,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
|
||||
}
|
||||
|
||||
auto launch_stop = std::chrono::high_resolution_clock::now();
|
||||
auto launch_result =
|
||||
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
|
||||
auto sync_start = std::chrono::high_resolution_clock::now();
|
||||
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
|
||||
auto sync_stop = std::chrono::high_resolution_clock::now();
|
||||
auto sync_result =
|
||||
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
|
||||
std::cout << "Time taken to Execute : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
launch_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
std::cout << "Time taken to Synchronize : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
sync_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
HIP_CHECK(hipGraphExecDestroy(graphExec));
|
||||
@@ -225,7 +217,7 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SingleBranchNoOperations") {
|
||||
TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
|
||||
constexpr int SIZE = 100;
|
||||
|
||||
char *dev[SIZE];
|
||||
char* dev[SIZE];
|
||||
for (int i = 0; i < SIZE; i++) {
|
||||
dev[i] = nullptr;
|
||||
}
|
||||
@@ -241,8 +233,8 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
|
||||
hipGraph_t graph;
|
||||
HIP_CHECK(hipGraphCreate(&graph, 0));
|
||||
|
||||
hipGraphNode_t memAllocNode[SIZE], memsetNode[SIZE], kernelNode[SIZE],
|
||||
memcpyNode[SIZE], memFreeNode[SIZE];
|
||||
hipGraphNode_t memAllocNode[SIZE], memsetNode[SIZE], kernelNode[SIZE], memcpyNode[SIZE],
|
||||
memFreeNode[SIZE];
|
||||
|
||||
// Prapare Mem alloc Nodes
|
||||
for (int i = 0; i < SIZE; i++) {
|
||||
@@ -254,24 +246,22 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
|
||||
memAllocNodeParams.bytesize = sizeof(char);
|
||||
|
||||
if (i == 0) {
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0,
|
||||
&memAllocNodeParams));
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0, &memAllocNodeParams));
|
||||
} else {
|
||||
::std::vector<hipGraphNode_t> memAllocNodeDependencies;
|
||||
memAllocNodeDependencies.push_back(memAllocNode[i - 1]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(
|
||||
&memAllocNode[i], graph, memAllocNodeDependencies.data(),
|
||||
memAllocNodeDependencies.size(), &memAllocNodeParams));
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, memAllocNodeDependencies.data(),
|
||||
memAllocNodeDependencies.size(), &memAllocNodeParams));
|
||||
}
|
||||
dev[i] = reinterpret_cast<char *>(memAllocNodeParams.dptr);
|
||||
dev[i] = reinterpret_cast<char*>(memAllocNodeParams.dptr);
|
||||
REQUIRE(dev[i] != nullptr);
|
||||
}
|
||||
|
||||
// Prapare Memset Nodes
|
||||
for (int i = 0; i < SIZE; i++) {
|
||||
hipMemsetParams pMemsetParams{};
|
||||
pMemsetParams.dst = reinterpret_cast<void *>(dev[i]);
|
||||
pMemsetParams.dst = reinterpret_cast<void*>(dev[i]);
|
||||
pMemsetParams.elementSize = 1;
|
||||
pMemsetParams.height = 1;
|
||||
pMemsetParams.pitch = 1;
|
||||
@@ -284,21 +274,19 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
|
||||
} else {
|
||||
memsetNodeDependencies.push_back(memsetNode[i - 1]);
|
||||
}
|
||||
HIP_CHECK(hipGraphAddMemsetNode(
|
||||
&memsetNode[i], graph, memsetNodeDependencies.data(),
|
||||
memsetNodeDependencies.size(), &pMemsetParams));
|
||||
HIP_CHECK(hipGraphAddMemsetNode(&memsetNode[i], graph, memsetNodeDependencies.data(),
|
||||
memsetNodeDependencies.size(), &pMemsetParams));
|
||||
}
|
||||
|
||||
// Prapare Kernel Nodes
|
||||
for (int i = 0; i < SIZE; i++) {
|
||||
hipKernelNodeParams kernelNodeParams{};
|
||||
kernelNodeParams.func = reinterpret_cast<void *>(addOneKernel);
|
||||
kernelNodeParams.func = reinterpret_cast<void*>(addOneKernel);
|
||||
kernelNodeParams.gridDim = dim3(1, 1, 1);
|
||||
kernelNodeParams.blockDim = dim3(1, 1, 1);
|
||||
kernelNodeParams.sharedMemBytes = 0;
|
||||
int size = 1;
|
||||
void *kernelArgs[2] = {reinterpret_cast<void *>(&dev[i]),
|
||||
reinterpret_cast<void *>(&size)};
|
||||
void* kernelArgs[2] = {reinterpret_cast<void*>(&dev[i]), reinterpret_cast<void*>(&size)};
|
||||
kernelNodeParams.kernelParams = kernelArgs;
|
||||
kernelNodeParams.extra = nullptr;
|
||||
|
||||
@@ -309,9 +297,8 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
|
||||
kernelNodeDependencies.push_back(kernelNode[i - 1]);
|
||||
}
|
||||
|
||||
HIP_CHECK(hipGraphAddKernelNode(
|
||||
&kernelNode[i], graph, kernelNodeDependencies.data(),
|
||||
kernelNodeDependencies.size(), &kernelNodeParams));
|
||||
HIP_CHECK(hipGraphAddKernelNode(&kernelNode[i], graph, kernelNodeDependencies.data(),
|
||||
kernelNodeDependencies.size(), &kernelNodeParams));
|
||||
}
|
||||
|
||||
// Prapare Memcpy Nodes
|
||||
@@ -330,9 +317,8 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
|
||||
} else {
|
||||
memcpyNodeDependencies.push_back(memcpyNode[i - 1]);
|
||||
}
|
||||
HIP_CHECK(hipGraphAddMemcpyNode(
|
||||
&memcpyNode[i], graph, memcpyNodeDependencies.data(),
|
||||
memcpyNodeDependencies.size(), &pMemcpyParams));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode(&memcpyNode[i], graph, memcpyNodeDependencies.data(),
|
||||
memcpyNodeDependencies.size(), &pMemcpyParams));
|
||||
}
|
||||
|
||||
// Prapare Mem free Nodes
|
||||
@@ -341,16 +327,16 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
|
||||
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
|
||||
memFreeNodeDependencies.push_back(memcpyNode[SIZE - 1]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(
|
||||
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(), reinterpret_cast<void *>(dev[i])));
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(),
|
||||
reinterpret_cast<void*>(dev[i])));
|
||||
} else {
|
||||
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
|
||||
memFreeNodeDependencies.push_back(memFreeNode[i - 1]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(
|
||||
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(), reinterpret_cast<void *>(dev[i])));
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(),
|
||||
reinterpret_cast<void*>(dev[i])));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -369,27 +355,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_SerialNodesSingleBranchWithOps") {
|
||||
HIP_CHECK(hipGraphLaunch(graphExec, stream));
|
||||
}
|
||||
auto launch_stop = std::chrono::high_resolution_clock::now();
|
||||
auto launch_result =
|
||||
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
|
||||
auto sync_start = std::chrono::high_resolution_clock::now();
|
||||
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
|
||||
auto sync_stop = std::chrono::high_resolution_clock::now();
|
||||
auto sync_result =
|
||||
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
|
||||
std::cout << "Time taken to Execute : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
launch_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
std::cout << "Time taken to Synchronize : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
sync_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
for (int i = 0; i < SIZE; i++) {
|
||||
@@ -434,14 +414,14 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
|
||||
constexpr int BRANCHES = 10;
|
||||
|
||||
int value = 100;
|
||||
int *hostMemSrc = new int[SIZE];
|
||||
int* hostMemSrc = new int[SIZE];
|
||||
REQUIRE(hostMemSrc != nullptr);
|
||||
|
||||
int *devMemSrc1 = nullptr;
|
||||
int* devMemSrc1 = nullptr;
|
||||
HIP_CHECK(hipMalloc(&devMemSrc1, NBYTES));
|
||||
REQUIRE(devMemSrc1 != nullptr);
|
||||
|
||||
int *devMemSrc2[BRANCHES];
|
||||
int* devMemSrc2[BRANCHES];
|
||||
for (int i = 0; i < BRANCHES; i++) {
|
||||
devMemSrc2[i] = nullptr;
|
||||
}
|
||||
@@ -455,14 +435,12 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
|
||||
hipGraph_t graph;
|
||||
HIP_CHECK(hipGraphCreate(&graph, 0));
|
||||
|
||||
hipGraphNode_t memcpyNodeH2D, memAllocNode[BRANCHES],
|
||||
fillKernelNode[BRANCHES], addKernelNode[BRANCHES],
|
||||
memcpyNodeD2H[BRANCHES], memFreeNode[BRANCHES], memcpyNodeH2H;
|
||||
hipGraphNode_t memcpyNodeH2D, memAllocNode[BRANCHES], fillKernelNode[BRANCHES],
|
||||
addKernelNode[BRANCHES], memcpyNodeD2H[BRANCHES], memFreeNode[BRANCHES], memcpyNodeH2H;
|
||||
|
||||
// Add H2D Node
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memcpyNodeH2D, graph, nullptr, 0,
|
||||
devMemSrc1, hostMemSrc, NBYTES,
|
||||
hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memcpyNodeH2D, graph, nullptr, 0, devMemSrc1, hostMemSrc,
|
||||
NBYTES, hipMemcpyHostToDevice));
|
||||
|
||||
for (int branch = 0; branch < BRANCHES; branch++) {
|
||||
// Add Mem alloc Nodes
|
||||
@@ -476,11 +454,10 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
|
||||
memAllocNodeParams.poolProps.location.id = 0;
|
||||
memAllocNodeParams.bytesize = NBYTES;
|
||||
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(
|
||||
&memAllocNode[branch], graph, memAllocNodeDependencies.data(),
|
||||
memAllocNodeDependencies.size(), &memAllocNodeParams));
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[branch], graph, memAllocNodeDependencies.data(),
|
||||
memAllocNodeDependencies.size(), &memAllocNodeParams));
|
||||
|
||||
devMemSrc2[branch] = reinterpret_cast<int *>(memAllocNodeParams.dptr);
|
||||
devMemSrc2[branch] = reinterpret_cast<int*>(memAllocNodeParams.dptr);
|
||||
REQUIRE(devMemSrc2[branch] != nullptr);
|
||||
|
||||
// Add Kernel Nodes (fillKernel)
|
||||
@@ -488,57 +465,52 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
|
||||
kernelNodeDependencies.push_back(memAllocNode[branch]);
|
||||
|
||||
hipKernelNodeParams kernelNodeParams{};
|
||||
kernelNodeParams.func = reinterpret_cast<void *>(fillKernel);
|
||||
kernelNodeParams.func = reinterpret_cast<void*>(fillKernel);
|
||||
kernelNodeParams.gridDim = dim3(1, 1, 1);
|
||||
kernelNodeParams.blockDim = dim3(1, 1, 1);
|
||||
kernelNodeParams.sharedMemBytes = 0;
|
||||
int size = SIZE;
|
||||
void *kernelArgs[3] = {reinterpret_cast<void *>(&devMemSrc2[branch]),
|
||||
reinterpret_cast<void *>(&size),
|
||||
reinterpret_cast<void *>(&value)};
|
||||
void* kernelArgs[3] = {reinterpret_cast<void*>(&devMemSrc2[branch]),
|
||||
reinterpret_cast<void*>(&size), reinterpret_cast<void*>(&value)};
|
||||
kernelNodeParams.kernelParams = kernelArgs;
|
||||
kernelNodeParams.extra = nullptr;
|
||||
|
||||
HIP_CHECK(hipGraphAddKernelNode(
|
||||
&fillKernelNode[branch], graph, kernelNodeDependencies.data(),
|
||||
kernelNodeDependencies.size(), &kernelNodeParams));
|
||||
HIP_CHECK(hipGraphAddKernelNode(&fillKernelNode[branch], graph, kernelNodeDependencies.data(),
|
||||
kernelNodeDependencies.size(), &kernelNodeParams));
|
||||
|
||||
// Add Kernel Nodes (addKernel)
|
||||
::std::vector<hipGraphNode_t> kernelNodeDependencies2;
|
||||
kernelNodeDependencies2.push_back(fillKernelNode[branch]);
|
||||
|
||||
hipKernelNodeParams kernelNodeParams2{};
|
||||
kernelNodeParams2.func = reinterpret_cast<void *>(addKernel);
|
||||
kernelNodeParams2.func = reinterpret_cast<void*>(addKernel);
|
||||
kernelNodeParams2.gridDim = dim3(1, 1, 1);
|
||||
kernelNodeParams2.blockDim = dim3(1, 1, 1);
|
||||
kernelNodeParams2.sharedMemBytes = 0;
|
||||
int size2 = SIZE;
|
||||
void *kernelArgs2[3] = {reinterpret_cast<void *>(&devMemSrc2[branch]),
|
||||
reinterpret_cast<void *>(&devMemSrc1),
|
||||
reinterpret_cast<void *>(&size2)};
|
||||
void* kernelArgs2[3] = {reinterpret_cast<void*>(&devMemSrc2[branch]),
|
||||
reinterpret_cast<void*>(&devMemSrc1), reinterpret_cast<void*>(&size2)};
|
||||
kernelNodeParams2.kernelParams = kernelArgs2;
|
||||
kernelNodeParams2.extra = nullptr;
|
||||
|
||||
HIP_CHECK(hipGraphAddKernelNode(
|
||||
&addKernelNode[branch], graph, kernelNodeDependencies2.data(),
|
||||
kernelNodeDependencies2.size(), &kernelNodeParams2));
|
||||
HIP_CHECK(hipGraphAddKernelNode(&addKernelNode[branch], graph, kernelNodeDependencies2.data(),
|
||||
kernelNodeDependencies2.size(), &kernelNodeParams2));
|
||||
|
||||
// Add D2H Nodes
|
||||
::std::vector<hipGraphNode_t> memcpyNodeD2HDependencies;
|
||||
memcpyNodeD2HDependencies.push_back(addKernelNode[branch]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(
|
||||
&memcpyNodeD2H[branch], graph, memcpyNodeD2HDependencies.data(),
|
||||
memcpyNodeD2HDependencies.size(), hostMemDst[branch],
|
||||
devMemSrc2[branch], NBYTES, hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memcpyNodeD2H[branch], graph,
|
||||
memcpyNodeD2HDependencies.data(),
|
||||
memcpyNodeD2HDependencies.size(), hostMemDst[branch],
|
||||
devMemSrc2[branch], NBYTES, hipMemcpyDeviceToHost));
|
||||
|
||||
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
|
||||
memFreeNodeDependencies.push_back(memcpyNodeD2H[branch]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(
|
||||
&memFreeNode[branch], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(),
|
||||
reinterpret_cast<void *>(devMemSrc2[branch])));
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[branch], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(),
|
||||
reinterpret_cast<void*>(devMemSrc2[branch])));
|
||||
}
|
||||
|
||||
// Add H2H Node
|
||||
@@ -547,10 +519,9 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
|
||||
memcpyNodeH2HDependencies.push_back(memFreeNode[i]);
|
||||
}
|
||||
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(
|
||||
&memcpyNodeH2H, graph, memcpyNodeH2HDependencies.data(),
|
||||
memcpyNodeH2HDependencies.size(), finalHostDst, hostMemDst,
|
||||
BRANCHES * SIZE * sizeof(int), hipMemcpyHostToHost));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memcpyNodeH2H, graph, memcpyNodeH2HDependencies.data(),
|
||||
memcpyNodeH2HDependencies.size(), finalHostDst, hostMemDst,
|
||||
BRANCHES * SIZE * sizeof(int), hipMemcpyHostToHost));
|
||||
|
||||
hipGraphExec_t graphExec;
|
||||
HIP_CHECK(hipGraphInstantiateWithFlags(&graphExec, graph, 0));
|
||||
@@ -570,27 +541,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
|
||||
}
|
||||
|
||||
auto launch_stop = std::chrono::high_resolution_clock::now();
|
||||
auto launch_result =
|
||||
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
|
||||
auto sync_start = std::chrono::high_resolution_clock::now();
|
||||
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
|
||||
auto sync_stop = std::chrono::high_resolution_clock::now();
|
||||
auto sync_result =
|
||||
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
|
||||
std::cout << "Time taken to Execute : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
launch_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
std::cout << "Time taken to Synchronize : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
sync_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
for (int branch = 0; branch < BRANCHES; branch++) {
|
||||
@@ -633,7 +598,7 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleBranches") {
|
||||
TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
|
||||
constexpr int BRANCHES = 10;
|
||||
|
||||
char *dev[BRANCHES];
|
||||
char* dev[BRANCHES];
|
||||
for (int i = 0; i < BRANCHES; i++) {
|
||||
dev[i] = nullptr;
|
||||
}
|
||||
@@ -649,8 +614,8 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
|
||||
hipGraph_t graph;
|
||||
HIP_CHECK(hipGraphCreate(&graph, 0));
|
||||
|
||||
hipGraphNode_t memAllocNode[BRANCHES], memsetNode[BRANCHES],
|
||||
kernelNode[BRANCHES], memcpyNode[BRANCHES], memFreeNode[BRANCHES];
|
||||
hipGraphNode_t memAllocNode[BRANCHES], memsetNode[BRANCHES], kernelNode[BRANCHES],
|
||||
memcpyNode[BRANCHES], memFreeNode[BRANCHES];
|
||||
|
||||
// Prapare Mem alloc Nodes
|
||||
for (int i = 0; i < BRANCHES; i++) {
|
||||
@@ -661,14 +626,13 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
|
||||
memAllocNodeParams.poolProps.location.id = 0;
|
||||
memAllocNodeParams.bytesize = sizeof(char);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0,
|
||||
&memAllocNodeParams));
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0, &memAllocNodeParams));
|
||||
|
||||
dev[i] = reinterpret_cast<char *>(memAllocNodeParams.dptr);
|
||||
dev[i] = reinterpret_cast<char*>(memAllocNodeParams.dptr);
|
||||
REQUIRE(dev[i] != nullptr);
|
||||
|
||||
hipMemsetParams pMemsetParams{};
|
||||
pMemsetParams.dst = reinterpret_cast<void *>(dev[i]);
|
||||
pMemsetParams.dst = reinterpret_cast<void*>(dev[i]);
|
||||
pMemsetParams.elementSize = 1;
|
||||
pMemsetParams.height = 1;
|
||||
pMemsetParams.pitch = 1;
|
||||
@@ -678,27 +642,24 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
|
||||
::std::vector<hipGraphNode_t> memsetNodeDependencies;
|
||||
memsetNodeDependencies.push_back(memAllocNode[i]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemsetNode(
|
||||
&memsetNode[i], graph, memsetNodeDependencies.data(),
|
||||
memsetNodeDependencies.size(), &pMemsetParams));
|
||||
HIP_CHECK(hipGraphAddMemsetNode(&memsetNode[i], graph, memsetNodeDependencies.data(),
|
||||
memsetNodeDependencies.size(), &pMemsetParams));
|
||||
|
||||
hipKernelNodeParams kernelNodeParams{};
|
||||
kernelNodeParams.func = reinterpret_cast<void *>(addOneKernel);
|
||||
kernelNodeParams.func = reinterpret_cast<void*>(addOneKernel);
|
||||
kernelNodeParams.gridDim = dim3(1, 1, 1);
|
||||
kernelNodeParams.blockDim = dim3(1, 1, 1);
|
||||
kernelNodeParams.sharedMemBytes = 0;
|
||||
int size = 1;
|
||||
void *kernelArgs[2] = {reinterpret_cast<void *>(&dev[i]),
|
||||
reinterpret_cast<void *>(&size)};
|
||||
void* kernelArgs[2] = {reinterpret_cast<void*>(&dev[i]), reinterpret_cast<void*>(&size)};
|
||||
kernelNodeParams.kernelParams = kernelArgs;
|
||||
kernelNodeParams.extra = nullptr;
|
||||
|
||||
::std::vector<hipGraphNode_t> kernelNodeDependencies;
|
||||
kernelNodeDependencies.push_back(memsetNode[i]);
|
||||
|
||||
HIP_CHECK(hipGraphAddKernelNode(
|
||||
&kernelNode[i], graph, kernelNodeDependencies.data(),
|
||||
kernelNodeDependencies.size(), &kernelNodeParams));
|
||||
HIP_CHECK(hipGraphAddKernelNode(&kernelNode[i], graph, kernelNodeDependencies.data(),
|
||||
kernelNodeDependencies.size(), &kernelNodeParams));
|
||||
|
||||
hipMemcpy3DParms pMemcpyParams{};
|
||||
pMemcpyParams.srcPos = make_hipPos(0, 0, 0);
|
||||
@@ -711,16 +672,15 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
|
||||
::std::vector<hipGraphNode_t> memcpyNodeDependencies;
|
||||
memcpyNodeDependencies.push_back(kernelNode[i]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemcpyNode(
|
||||
&memcpyNode[i], graph, memcpyNodeDependencies.data(),
|
||||
memcpyNodeDependencies.size(), &pMemcpyParams));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode(&memcpyNode[i], graph, memcpyNodeDependencies.data(),
|
||||
memcpyNodeDependencies.size(), &pMemcpyParams));
|
||||
|
||||
::std::vector<hipGraphNode_t> memFreeNodeDependencies;
|
||||
memFreeNodeDependencies.push_back(memcpyNode[i]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(
|
||||
&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(), reinterpret_cast<void *>(dev[i])));
|
||||
HIP_CHECK(hipGraphAddMemFreeNode(&memFreeNode[i], graph, memFreeNodeDependencies.data(),
|
||||
memFreeNodeDependencies.size(),
|
||||
reinterpret_cast<void*>(dev[i])));
|
||||
}
|
||||
|
||||
hipGraphExec_t graphExec;
|
||||
@@ -738,27 +698,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
|
||||
HIP_CHECK(hipGraphLaunch(graphExec, stream));
|
||||
}
|
||||
auto launch_stop = std::chrono::high_resolution_clock::now();
|
||||
auto launch_result =
|
||||
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
|
||||
auto sync_start = std::chrono::high_resolution_clock::now();
|
||||
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
|
||||
auto sync_stop = std::chrono::high_resolution_clock::now();
|
||||
auto sync_result =
|
||||
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
|
||||
std::cout << "Time taken to Execute : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
launch_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
std::cout << "Time taken to Synchronize : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
sync_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
for (int i = 0; i < BRANCHES; i++) {
|
||||
@@ -791,7 +745,7 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_MultipleIndependentBranches") {
|
||||
TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_OneBranchNoOps_AutoFreeOnLaunch") {
|
||||
constexpr int SIZE = 1024;
|
||||
|
||||
int *devMem[SIZE];
|
||||
int* devMem[SIZE];
|
||||
for (int i = 0; i < SIZE; i++) {
|
||||
devMem[i] = nullptr;
|
||||
}
|
||||
@@ -813,23 +767,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_OneBranchNoOps_AutoFreeOnLaunch") {
|
||||
memAllocNodeParams.bytesize = sizeof(int);
|
||||
|
||||
if (i == 0) {
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0,
|
||||
&memAllocNodeParams));
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, nullptr, 0, &memAllocNodeParams));
|
||||
} else {
|
||||
::std::vector<hipGraphNode_t> memAllocNodeDependencies;
|
||||
memAllocNodeDependencies.push_back(memAllocNode[i - 1]);
|
||||
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(
|
||||
&memAllocNode[i], graph, memAllocNodeDependencies.data(),
|
||||
memAllocNodeDependencies.size(), &memAllocNodeParams));
|
||||
HIP_CHECK(hipGraphAddMemAllocNode(&memAllocNode[i], graph, memAllocNodeDependencies.data(),
|
||||
memAllocNodeDependencies.size(), &memAllocNodeParams));
|
||||
}
|
||||
devMem[i] = reinterpret_cast<int *>(memAllocNodeParams.dptr);
|
||||
devMem[i] = reinterpret_cast<int*>(memAllocNodeParams.dptr);
|
||||
REQUIRE(devMem[i] != nullptr);
|
||||
}
|
||||
|
||||
hipGraphExec_t graphExec;
|
||||
HIP_CHECK(hipGraphInstantiateWithFlags(
|
||||
&graphExec, graph, hipGraphInstantiateFlagAutoFreeOnLaunch));
|
||||
HIP_CHECK(
|
||||
hipGraphInstantiateWithFlags(&graphExec, graph, hipGraphInstantiateFlagAutoFreeOnLaunch));
|
||||
|
||||
// Warm up call
|
||||
HIP_CHECK(hipGraphLaunch(graphExec, stream));
|
||||
@@ -844,27 +796,21 @@ TEST_CASE("Perf_GraphWithMoreAllocFreeNodes_OneBranchNoOps_AutoFreeOnLaunch") {
|
||||
}
|
||||
|
||||
auto launch_stop = std::chrono::high_resolution_clock::now();
|
||||
auto launch_result =
|
||||
std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
auto launch_result = std::chrono::duration<double, std::milli>(launch_stop - launch_start);
|
||||
|
||||
auto sync_start = std::chrono::high_resolution_clock::now();
|
||||
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
|
||||
auto sync_stop = std::chrono::high_resolution_clock::now();
|
||||
auto sync_result =
|
||||
std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
auto sync_result = std::chrono::duration<double, std::milli>(sync_stop - sync_start);
|
||||
|
||||
std::cout << "Time taken to Execute : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
launch_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(launch_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
std::cout << "Time taken to Synchronize : "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(
|
||||
sync_result)
|
||||
.count()
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(sync_result).count()
|
||||
<< " millisecs " << std::endl;
|
||||
|
||||
HIP_CHECK(hipGraphExecDestroy(graphExec));
|
||||
|
||||
@@ -26,7 +26,7 @@ THE SOFTWARE.
|
||||
static constexpr int N = 1024;
|
||||
static constexpr int Nbytes = N * sizeof(int);
|
||||
static size_t NElem{N};
|
||||
static constexpr int blocksPerCU = 6; // to hide latency
|
||||
static constexpr int blocksPerCU = 6; // to hide latency
|
||||
static constexpr int threadsPerBlock = 256;
|
||||
// Num of parallel Branches
|
||||
const unsigned int kNumNode = 5;
|
||||
@@ -39,8 +39,7 @@ const unsigned int kNumNode = 5;
|
||||
* - Launches an executable graph in the specified stream.
|
||||
*/
|
||||
unsigned blocks = HipTest::setNumBlocks(blocksPerCU, threadsPerBlock, N);
|
||||
template <typename T>
|
||||
__global__ void vectorADD(const T *A_d, const T *B_d, T *C_d, size_t NELEM) {
|
||||
template <typename T> __global__ void vectorADD(const T* A_d, const T* B_d, T* C_d, size_t NELEM) {
|
||||
size_t offset = (blockIdx.x * blockDim.x + threadIdx.x);
|
||||
size_t stride = blockDim.x * gridDim.x;
|
||||
|
||||
@@ -73,32 +72,31 @@ TEST_CASE("Unit_hipGraph_Performance_Improvement_ParallelGraph") {
|
||||
HIP_CHECK(hipStreamCreate(&stream));
|
||||
HIP_CHECK(hipGraphCreate(&graph, 0));
|
||||
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy1, graph, nullptr, 0, A_d, A_h,
|
||||
Nbytes, hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy2, graph, nullptr, 0, B_d, B_h,
|
||||
Nbytes, hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy1, graph, nullptr, 0, A_d, A_h, Nbytes,
|
||||
hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy2, graph, nullptr, 0, B_d, B_h, Nbytes,
|
||||
hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipGraphAddDependencies(graph, &memCpy1, &memCpy2, 1));
|
||||
|
||||
for (int i = 0; i < kNumNode; i++) {
|
||||
hipKernelNodeParams kernelNodeParams{};
|
||||
void *kernelArgs[] = {&A_d, &B_d, &C_d, reinterpret_cast<void *>(&NElem)};
|
||||
kernelNodeParams.func = reinterpret_cast<void *>(vectorADD<int>);
|
||||
void* kernelArgs[] = {&A_d, &B_d, &C_d, reinterpret_cast<void*>(&NElem)};
|
||||
kernelNodeParams.func = reinterpret_cast<void*>(vectorADD<int>);
|
||||
kernelNodeParams.gridDim = dim3(blocks);
|
||||
kernelNodeParams.blockDim = dim3(threadsPerBlock);
|
||||
kernelNodeParams.sharedMemBytes = 0;
|
||||
kernelNodeParams.kernelParams = reinterpret_cast<void **>(kernelArgs);
|
||||
kernelNodeParams.kernelParams = reinterpret_cast<void**>(kernelArgs);
|
||||
kernelNodeParams.extra = nullptr;
|
||||
HIP_CHECK(
|
||||
hipGraphAddKernelNode(&kNode[i], graph, nullptr, 0, &kernelNodeParams));
|
||||
HIP_CHECK(hipGraphAddKernelNode(&kNode[i], graph, nullptr, 0, &kernelNodeParams));
|
||||
HIP_CHECK(hipGraphAddDependencies(graph, &memCpy2, &kNode[i], 1));
|
||||
}
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy3, graph, nullptr, 0, C_h, C_d,
|
||||
Nbytes, hipMemcpyDeviceToHost));
|
||||
HIP_CHECK(hipGraphAddMemcpyNode1D(&memCpy3, graph, nullptr, 0, C_h, C_d, Nbytes,
|
||||
hipMemcpyDeviceToHost));
|
||||
for (int i = 0; i < kNumNode; i++) {
|
||||
HIP_CHECK(hipGraphAddDependencies(graph, &kNode[i], &memCpy3, 1));
|
||||
}
|
||||
|
||||
hipGraphNode_t *nodes{nullptr};
|
||||
hipGraphNode_t* nodes{nullptr};
|
||||
size_t numNodes = 0;
|
||||
HIP_CHECK(hipGraphGetNodes(graph, nodes, &numNodes));
|
||||
INFO("Num of nodes in the graph: " << numNodes);
|
||||
@@ -113,10 +111,9 @@ TEST_CASE("Unit_hipGraph_Performance_Improvement_ParallelGraph") {
|
||||
// Stop time
|
||||
auto stop = std::chrono::high_resolution_clock::now();
|
||||
auto duration = stop - start;
|
||||
INFO(
|
||||
"Time taken for Graph: "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
|
||||
<< " milliSeconds");
|
||||
INFO("Time taken for Graph: "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
|
||||
<< " milliSeconds");
|
||||
// Verify graph execution result
|
||||
HipTest::checkVectorADD(A_h, B_h, C_h, N);
|
||||
|
||||
@@ -150,18 +147,17 @@ TEST_CASE("Unit_hipGraph_Performance_With_Stream_Operations") {
|
||||
HIP_CHECK(hipMemcpyAsync(A_d, A_h, Nbytes, hipMemcpyDefault, stream));
|
||||
HIP_CHECK(hipMemcpyAsync(B_d, B_h, Nbytes, hipMemcpyDefault, stream));
|
||||
for (int i = 0; i < kNumNode; i++) {
|
||||
hipLaunchKernelGGL(vectorADD, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
stream, A_d, B_d, C_d, NElem);
|
||||
hipLaunchKernelGGL(vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, stream, A_d, B_d, C_d,
|
||||
NElem);
|
||||
}
|
||||
HIP_CHECK(hipMemcpyAsync(C_h, C_d, Nbytes, hipMemcpyDefault, stream));
|
||||
HIP_CHECK(hipStreamSynchronize(stream));
|
||||
}
|
||||
auto stop = std::chrono::high_resolution_clock::now();
|
||||
auto duration = stop - start;
|
||||
INFO(
|
||||
"Time taken for Stream: "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
|
||||
<< " milliSeconds");
|
||||
INFO("Time taken for Stream: "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
|
||||
<< " milliSeconds");
|
||||
// Verify graph execution result
|
||||
HipTest::checkVectorADD(A_h, B_h, C_h, N);
|
||||
|
||||
@@ -193,8 +189,8 @@ TEST_CASE("Unit_hipGraph_Performance_With_Stream_Capture") {
|
||||
HIP_CHECK(hipMemcpyAsync(A_d, A_h, Nbytes, hipMemcpyDefault, stream));
|
||||
HIP_CHECK(hipMemcpyAsync(B_d, B_h, Nbytes, hipMemcpyDefault, stream));
|
||||
for (int i = 0; i < kNumNode; i++) {
|
||||
hipLaunchKernelGGL(vectorADD, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
stream, A_d, B_d, C_d, NElem);
|
||||
hipLaunchKernelGGL(vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, stream, A_d, B_d, C_d,
|
||||
NElem);
|
||||
}
|
||||
HIP_CHECK(hipMemcpyAsync(C_h, C_d, Nbytes, hipMemcpyDefault, stream));
|
||||
HIP_CHECK(hipStreamEndCapture(stream, &graph));
|
||||
@@ -208,10 +204,9 @@ TEST_CASE("Unit_hipGraph_Performance_With_Stream_Capture") {
|
||||
HIP_CHECK(hipStreamSynchronize(streamForGraph));
|
||||
auto stop = std::chrono::high_resolution_clock::now();
|
||||
auto duration = stop - start;
|
||||
INFO(
|
||||
"Time taken for Graph via Stream Capture: "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
|
||||
<< " milliSeconds");
|
||||
INFO("Time taken for Graph via Stream Capture: "
|
||||
<< std::chrono::duration_cast<std::chrono::milliseconds>(duration).count()
|
||||
<< " milliSeconds");
|
||||
// Verify graph execution result
|
||||
HipTest::checkVectorADD(A_h, B_h, C_h, N);
|
||||
|
||||
@@ -223,4 +218,3 @@ TEST_CASE("Unit_hipGraph_Performance_With_Stream_Capture") {
|
||||
* End doxygen group GraphTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
|
||||
@@ -18,40 +18,39 @@ THE SOFTWARE.
|
||||
*/
|
||||
|
||||
/**
|
||||
* @addtogroup hipMemcpyAsync hipMemcpyAsync
|
||||
* @{
|
||||
* @ingroup perfMemoryTest
|
||||
* `hipMemcpyAsync(void* dst, const void* src, size_t count,
|
||||
* hipMemcpyKind kind, hipStream_t stream = 0)` -
|
||||
* Copies data between host and device.
|
||||
*/
|
||||
* @addtogroup hipMemcpyAsync hipMemcpyAsync
|
||||
* @{
|
||||
* @ingroup perfMemoryTest
|
||||
* `hipMemcpyAsync(void* dst, const void* src, size_t count,
|
||||
* hipMemcpyKind kind, hipStream_t stream = 0)` -
|
||||
* Copies data between host and device.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
|
||||
#define NUM_SIZES 9
|
||||
// 4KB, 8KB, 64KB, 256KB, 1 MB, 4MB, 16 MB, 16MB+10
|
||||
static const unsigned int Sizes[NUM_SIZES] =
|
||||
{4096, 8192, 65536, 262144, 524288, 1048576, 4194304, 16777216, 16777216+10};
|
||||
static const unsigned int Sizes[NUM_SIZES] = {4096, 8192, 65536, 262144, 524288,
|
||||
1048576, 4194304, 16777216, 16777216 + 10};
|
||||
|
||||
static const unsigned int Iterations[2] = {1, 1000};
|
||||
|
||||
#define BUF_TYPES 4
|
||||
// 16 ways to combine 4 different buffer types
|
||||
#define NUM_SUBTESTS (BUF_TYPES*BUF_TYPES)
|
||||
#define NUM_SUBTESTS (BUF_TYPES * BUF_TYPES)
|
||||
|
||||
static void setData(void *ptr, unsigned int size, char value) {
|
||||
char *ptr2 = reinterpret_cast<char *>(ptr);
|
||||
for (unsigned int i = 0; i < size ; i++) {
|
||||
static void setData(void* ptr, unsigned int size, char value) {
|
||||
char* ptr2 = reinterpret_cast<char*>(ptr);
|
||||
for (unsigned int i = 0; i < size; i++) {
|
||||
ptr2[i] = value;
|
||||
}
|
||||
}
|
||||
|
||||
static void checkData(void *ptr, unsigned int size, char value) {
|
||||
char *ptr2 = reinterpret_cast<char *>(ptr);
|
||||
static void checkData(void* ptr, unsigned int size, char value) {
|
||||
char* ptr2 = reinterpret_cast<char*>(ptr);
|
||||
for (unsigned int i = 0; i < size; i++) {
|
||||
if (ptr2[i] != value) {
|
||||
INFO("Validation failed at " << i << " Got " << ptr2[i] <<
|
||||
" Expected " << value);
|
||||
INFO("Validation failed at " << i << " Got " << ptr2[i] << " Expected " << value);
|
||||
REQUIRE(false);
|
||||
}
|
||||
}
|
||||
@@ -63,17 +62,17 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
|
||||
bool hostMalloc[2] = {false};
|
||||
bool hostRegister[2] = {false};
|
||||
bool unpinnedMalloc[2] = {false};
|
||||
void *memptr[2] = {NULL};
|
||||
void *alignedmemptr[2] = {NULL};
|
||||
void *srcBuffer = NULL;
|
||||
void *dstBuffer = NULL;
|
||||
void* memptr[2] = {NULL};
|
||||
void* alignedmemptr[2] = {NULL};
|
||||
void* srcBuffer = NULL;
|
||||
void* dstBuffer = NULL;
|
||||
|
||||
int numTests = (p_tests == -1) ? (NUM_SIZES*NUM_SUBTESTS*2 - 1) : p_tests;
|
||||
int numTests = (p_tests == -1) ? (NUM_SIZES * NUM_SUBTESTS * 2 - 1) : p_tests;
|
||||
int test = (p_tests == -1) ? 0 : p_tests;
|
||||
|
||||
for ( ; test <= numTests; test++ ) {
|
||||
for (; test <= numTests; test++) {
|
||||
unsigned int srcTest = (test / NUM_SIZES) % BUF_TYPES;
|
||||
unsigned int dstTest = (test / (NUM_SIZES*BUF_TYPES)) % BUF_TYPES;
|
||||
unsigned int dstTest = (test / (NUM_SIZES * BUF_TYPES)) % BUF_TYPES;
|
||||
bufSize_ = Sizes[test % NUM_SIZES];
|
||||
hostMalloc[0] = hostMalloc[1] = false;
|
||||
hostRegister[0] = hostRegister[1] = false;
|
||||
@@ -101,8 +100,7 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
|
||||
numIter = Iterations[test / (NUM_SIZES * NUM_SUBTESTS)];
|
||||
|
||||
if (hostMalloc[0]) {
|
||||
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&srcBuffer),
|
||||
bufSize_, 0));
|
||||
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&srcBuffer), bufSize_, 0));
|
||||
setData(srcBuffer, bufSize_, 0xd0);
|
||||
} else if (hostRegister[0]) {
|
||||
memptr[0] = malloc(bufSize_ + 4096);
|
||||
@@ -121,8 +119,7 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
|
||||
}
|
||||
|
||||
if (hostMalloc[1]) {
|
||||
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&dstBuffer),
|
||||
bufSize_, 0));
|
||||
HIP_CHECK(hipHostMalloc(reinterpret_cast<void**>(&dstBuffer), bufSize_, 0));
|
||||
} else if (hostRegister[1]) {
|
||||
memptr[1] = malloc(bufSize_ + 4096);
|
||||
alignedmemptr[1] = reinterpret_cast<void*>(memptr[1]);
|
||||
@@ -143,8 +140,7 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
|
||||
auto all_start = std::chrono::steady_clock::now();
|
||||
|
||||
for (unsigned int i = 0; i < numIter; i++) {
|
||||
HIP_CHECK(hipMemcpyAsync(dstBuffer, srcBuffer, bufSize_,
|
||||
hipMemcpyDefault, NULL));
|
||||
HIP_CHECK(hipMemcpyAsync(dstBuffer, srcBuffer, bufSize_, hipMemcpyDefault, NULL));
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
@@ -152,11 +148,11 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
|
||||
std::chrono::duration<double> elapsed_secs = all_end - all_start;
|
||||
|
||||
// read speed in GB/s
|
||||
double perf = (static_cast<double>(bufSize_ * numIter) *
|
||||
static_cast<double>(1e-09)) / elapsed_secs.count();
|
||||
double perf = (static_cast<double>(bufSize_ * numIter) * static_cast<double>(1e-09)) /
|
||||
elapsed_secs.count();
|
||||
|
||||
const char *strSrc = NULL;
|
||||
const char *strDst = NULL;
|
||||
const char* strSrc = NULL;
|
||||
const char* strDst = NULL;
|
||||
if (hostMalloc[0])
|
||||
strSrc = "hHM";
|
||||
else if (hostRegister[0])
|
||||
@@ -178,15 +174,15 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
|
||||
// Double results when src and dst are both on device
|
||||
if ((!hostMalloc[0] && !hostRegister[0] && !unpinnedMalloc[0]) &&
|
||||
(!hostMalloc[1] && !hostRegister[1] && !unpinnedMalloc[1]))
|
||||
perf *= 2.0;
|
||||
perf *= 2.0;
|
||||
// Double results when src and dst are both in sysmem
|
||||
if ((hostMalloc[0] || hostRegister[0] || unpinnedMalloc[0]) &&
|
||||
(hostMalloc[1] || hostRegister[1] || unpinnedMalloc[1]))
|
||||
perf *= 2.0;
|
||||
perf *= 2.0;
|
||||
|
||||
INFO("HIPPerfBufferCopySpeed[" << test << "]\t( " << bufSize_ <<
|
||||
")\ts:" << strSrc << " d:" << strDst << "\ti:" << numIter <<
|
||||
"\t(GB/s) perf\t" << (float)perf);
|
||||
INFO("HIPPerfBufferCopySpeed[" << test << "]\t( " << bufSize_ << ")\ts:" << strSrc
|
||||
<< " d:" << strDst << "\ti:" << numIter << "\t(GB/s) perf\t"
|
||||
<< (float)perf);
|
||||
|
||||
// Verification
|
||||
void* temp = malloc(bufSize_ + 4096);
|
||||
@@ -224,40 +220,42 @@ static bool hipPerfBufferCopySpeed_test(int p_tests) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Verify hipPerfBufferCopySpeed status.
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - perftests/memory/hipPerfBufferCopySpeed.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.6
|
||||
*/
|
||||
* Test Description
|
||||
* ------------------------
|
||||
* - Verify hipPerfBufferCopySpeed status.
|
||||
* Test source
|
||||
* ------------------------
|
||||
* - perftests/memory/hipPerfBufferCopySpeed.cc
|
||||
* Test requirements
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 5.6
|
||||
*/
|
||||
|
||||
TEST_CASE("Perf_hipPerfBufferCopySpeed_test") {
|
||||
int numDevices = 0;
|
||||
HIP_CHECK(hipGetDeviceCount(&numDevices));
|
||||
|
||||
if (numDevices <= 0) {
|
||||
SUCCEED("Skipped testcase hipPerfBufferCopySpeed as"
|
||||
"there is no device to test.");
|
||||
SUCCEED(
|
||||
"Skipped testcase hipPerfBufferCopySpeed as"
|
||||
"there is no device to test.");
|
||||
} else {
|
||||
int deviceId = 0;
|
||||
HIP_CHECK(hipSetDevice(deviceId));
|
||||
hipDeviceProp_t props;
|
||||
HIP_CHECK(hipGetDeviceProperties(&props, deviceId));
|
||||
|
||||
INFO("hipPerfBufferCopySpeed - info: Set device to " << deviceId
|
||||
<< " : " << props.name << "Legend: unp - unpinned(malloc),"
|
||||
" hM - hipMalloc(device)\n hHR - hipHostRegister(pinned),"
|
||||
" hHM - hipHostMalloc(prePinned)\n");
|
||||
INFO("hipPerfBufferCopySpeed - info: Set device to "
|
||||
<< deviceId << " : " << props.name
|
||||
<< "Legend: unp - unpinned(malloc),"
|
||||
" hM - hipMalloc(device)\n hHR - hipHostRegister(pinned),"
|
||||
" hHM - hipHostMalloc(prePinned)\n");
|
||||
|
||||
REQUIRE(true == hipPerfBufferCopySpeed_test(1));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group perfMemoryTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group perfMemoryTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -32,72 +32,70 @@ THE SOFTWARE.
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <iomanip>
|
||||
//#define VERIFY_DATA
|
||||
// #define VERIFY_DATA
|
||||
using namespace std;
|
||||
enum DEV_MEM_TYPE { COARSE_GRAINED, FINE_GRAINED, EXTENDED_FINE_GRAINED, UNKNOWN_MEM};
|
||||
enum DEV_MEM_TYPE { COARSE_GRAINED, FINE_GRAINED, EXTENDED_FINE_GRAINED, UNKNOWN_MEM };
|
||||
|
||||
typedef long long T; // You may change to any type
|
||||
typedef long long T; // You may change to any type
|
||||
|
||||
static constexpr int nWarmup = 1; // warmup iteration number
|
||||
static constexpr int nIters = 10; // interation number for test
|
||||
static constexpr size_t dataBytes = 1024*1024*1024;
|
||||
static constexpr int nWarmup = 1; // warmup iteration number
|
||||
static constexpr int nIters = 10; // interation number for test
|
||||
static constexpr size_t dataBytes = 1024 * 1024 * 1024;
|
||||
|
||||
template <typename T>
|
||||
static __global__ void copy_kernel(T* dst, T* src, size_t N) {
|
||||
template <typename T> static __global__ void copy_kernel(T* dst, T* src, size_t N) {
|
||||
const size_t off = blockDim.x * gridDim.x;
|
||||
for (size_t i = blockIdx.x * blockDim.x + threadIdx.x; i < N; i += off)
|
||||
dst[i] = src[i];
|
||||
for (size_t i = blockIdx.x * blockDim.x + threadIdx.x; i < N; i += off) dst[i] = src[i];
|
||||
}
|
||||
|
||||
static string getMemType(DEV_MEM_TYPE memType) {
|
||||
switch (memType) {
|
||||
case COARSE_GRAINED:
|
||||
return "coarse";
|
||||
case FINE_GRAINED:
|
||||
return "fine";
|
||||
case EXTENDED_FINE_GRAINED:
|
||||
// Extended - Scope Fine Grained Memory: read is cached, write is not
|
||||
return "extended fine";
|
||||
default:
|
||||
return "unknown mem type";
|
||||
case COARSE_GRAINED:
|
||||
return "coarse";
|
||||
case FINE_GRAINED:
|
||||
return "fine";
|
||||
case EXTENDED_FINE_GRAINED:
|
||||
// Extended - Scope Fine Grained Memory: read is cached, write is not
|
||||
return "extended fine";
|
||||
default:
|
||||
return "unknown mem type";
|
||||
}
|
||||
}
|
||||
|
||||
static void mallocDevBuf(void** pp, size_t size, DEV_MEM_TYPE memType) {
|
||||
switch (memType) {
|
||||
case COARSE_GRAINED:
|
||||
HIP_CHECK(hipMalloc(pp, size));
|
||||
break;
|
||||
case FINE_GRAINED:
|
||||
case COARSE_GRAINED:
|
||||
HIP_CHECK(hipMalloc(pp, size));
|
||||
break;
|
||||
case FINE_GRAINED:
|
||||
#if HT_AMD
|
||||
HIP_CHECK(hipExtMallocWithFlags(pp, size, hipDeviceMallocFinegrained));
|
||||
HIP_CHECK(hipExtMallocWithFlags(pp, size, hipDeviceMallocFinegrained));
|
||||
#else
|
||||
fprintf(stderr, "Unsupported memType for nvidia hardware: %d\n", memType);
|
||||
REQUIRE(false);
|
||||
fprintf(stderr, "Unsupported memType for nvidia hardware: %d\n", memType);
|
||||
REQUIRE(false);
|
||||
#endif
|
||||
break;
|
||||
case EXTENDED_FINE_GRAINED:
|
||||
// Extended - Scope Fine Grained Memory: read is cached, write is not
|
||||
// Perf gain compared with cacheable write
|
||||
break;
|
||||
case EXTENDED_FINE_GRAINED:
|
||||
// Extended - Scope Fine Grained Memory: read is cached, write is not
|
||||
// Perf gain compared with cacheable write
|
||||
#if HT_AMD
|
||||
HIP_CHECK(hipExtMallocWithFlags(pp, size, hipDeviceMallocUncached));
|
||||
HIP_CHECK(hipExtMallocWithFlags(pp, size, hipDeviceMallocUncached));
|
||||
#else
|
||||
fprintf(stderr, "Unsupported memType for nvidia hardware: %d\n", memType);
|
||||
REQUIRE(false);
|
||||
fprintf(stderr, "Unsupported memType for nvidia hardware: %d\n", memType);
|
||||
REQUIRE(false);
|
||||
#endif
|
||||
break;
|
||||
default:
|
||||
fprintf(stderr, "Unknown memType = %d\n", memType);
|
||||
REQUIRE(false);
|
||||
break;
|
||||
break;
|
||||
default:
|
||||
fprintf(stderr, "Unknown memType = %d\n", memType);
|
||||
REQUIRE(false);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
|
||||
DEV_MEM_TYPE srcType, DEV_MEM_TYPE dstType) {
|
||||
static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu, DEV_MEM_TYPE srcType,
|
||||
DEV_MEM_TYPE dstType) {
|
||||
int nGpus = 0;
|
||||
unsigned int threadsPerBlock = 1024;
|
||||
unsigned int blocks = 16; // DEBUG_CLR_LIMIT_BLIT_WG
|
||||
unsigned int blocks = 16; // DEBUG_CLR_LIMIT_BLIT_WG
|
||||
HIP_CHECK(hipGetDeviceCount(&nGpus));
|
||||
if (nGpus < 2) {
|
||||
fprintf(stderr, "Need at least 2 GPUs, skipped!\n");
|
||||
@@ -120,7 +118,7 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
|
||||
#endif
|
||||
char** srcBuf = reinterpret_cast<char**>(malloc(nGpus * nGpus * sizeof(char*)));
|
||||
char** dstBuf = reinterpret_cast<char**>(malloc(nGpus * nGpus * sizeof(char*)));
|
||||
hipStream_t *streams = (hipStream_t*)malloc(nGpus*nGpus*sizeof(hipStream_t));
|
||||
hipStream_t* streams = (hipStream_t*)malloc(nGpus * nGpus * sizeof(hipStream_t));
|
||||
for (int local = 0; local < nGpus; local++) {
|
||||
HIP_CHECK(hipSetDevice(local));
|
||||
for (int remote = 0; remote < nGpus; remote++) {
|
||||
@@ -130,48 +128,49 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
|
||||
HIP_CHECK(hipStreamCreateWithFlags(&streams[local * nGpus + remote], hipStreamNonBlocking));
|
||||
HIP_CHECK(hipDeviceEnablePeerAccess(remote, 0));
|
||||
#ifdef VERIFY_DATA
|
||||
HIP_CHECK(hipMemcpy(srcBuf[local * nGpus + remote], hostMem0.data(), dataBytes, hipMemcpyHostToDevice));
|
||||
HIP_CHECK(hipMemcpy(srcBuf[local * nGpus + remote], hostMem0.data(), dataBytes,
|
||||
hipMemcpyHostToDevice));
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
unsigned N = dataBytes / sizeof(T); // Number of T in buffer of dataBytes bytes.
|
||||
unsigned N = dataBytes / sizeof(T); // Number of T in buffer of dataBytes bytes.
|
||||
REQUIRE(N * sizeof(T) == dataBytes);
|
||||
|
||||
auto test = [&](int iters) {
|
||||
for (int it = 0; it < iters; it++) {
|
||||
for (int local = 0; local < nGpus; local++) {
|
||||
HIP_CHECK(hipSetDevice(local));
|
||||
for (int i = 0; i < nGpus-1; i++) {
|
||||
for (int i = 0; i < nGpus - 1; i++) {
|
||||
int remote = (local + i + 1) % nGpus;
|
||||
if (toRemote) {
|
||||
// local to remotes
|
||||
if (kernelCopy) {
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
streams[local * nGpus + remote],
|
||||
reinterpret_cast<T*>(dstBuf[remote * nGpus + local]),
|
||||
reinterpret_cast<T*>(srcBuf[local * nGpus + remote]),
|
||||
static_cast<size_t>(N));
|
||||
HIP_CHECK(hipGetLastError());
|
||||
} else {
|
||||
HIP_CHECK(hipMemcpyPeerAsync(dstBuf[remote * nGpus + local], remote,
|
||||
srcBuf[local * nGpus + remote], local,
|
||||
dataBytes, streams[local * nGpus + remote]));
|
||||
}
|
||||
// local to remotes
|
||||
if (kernelCopy) {
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
streams[local * nGpus + remote],
|
||||
reinterpret_cast<T*>(dstBuf[remote * nGpus + local]),
|
||||
reinterpret_cast<T*>(srcBuf[local * nGpus + remote]),
|
||||
static_cast<size_t>(N));
|
||||
HIP_CHECK(hipGetLastError());
|
||||
} else {
|
||||
HIP_CHECK(hipMemcpyPeerAsync(dstBuf[remote * nGpus + local], remote,
|
||||
srcBuf[local * nGpus + remote], local, dataBytes,
|
||||
streams[local * nGpus + remote]));
|
||||
}
|
||||
} else {
|
||||
// remotes to local
|
||||
if (kernelCopy) {
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
streams[remote * nGpus + local],
|
||||
reinterpret_cast<T*>(dstBuf[local * nGpus + remote]),
|
||||
reinterpret_cast<T*>(srcBuf[remote * nGpus + local]),
|
||||
static_cast<size_t>(N));
|
||||
HIP_CHECK(hipGetLastError());
|
||||
} else {
|
||||
HIPCHECK(hipMemcpyPeerAsync(dstBuf[local * nGpus + remote], local,
|
||||
srcBuf[remote* nGpus + local], remote,
|
||||
dataBytes, streams[remote * nGpus + local]));
|
||||
}
|
||||
// remotes to local
|
||||
if (kernelCopy) {
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
streams[remote * nGpus + local],
|
||||
reinterpret_cast<T*>(dstBuf[local * nGpus + remote]),
|
||||
reinterpret_cast<T*>(srcBuf[remote * nGpus + local]),
|
||||
static_cast<size_t>(N));
|
||||
HIP_CHECK(hipGetLastError());
|
||||
} else {
|
||||
HIPCHECK(hipMemcpyPeerAsync(dstBuf[local * nGpus + remote], local,
|
||||
srcBuf[remote * nGpus + local], remote, dataBytes,
|
||||
streams[remote * nGpus + local]));
|
||||
}
|
||||
}
|
||||
}
|
||||
if (onOneGpu) break;
|
||||
@@ -197,10 +196,9 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
|
||||
test(nWarmup);
|
||||
auto cpuStart = std::chrono::steady_clock::now();
|
||||
test(nIters);
|
||||
std::chrono::duration<double, std::milli> cpuMS =
|
||||
std::chrono::steady_clock::now() - cpuStart;
|
||||
std::chrono::duration<double, std::milli> cpuMS = std::chrono::steady_clock::now() - cpuStart;
|
||||
fprintf(stderr, "%s: Time: %f ms/iter, AvgCopyBW: %f GB/s per GPU\n", title.c_str(),
|
||||
cpuMS.count()/nIters, (nGpus-1)*dataBytes/cpuMS.count()*nIters/1e6);
|
||||
cpuMS.count() / nIters, (nGpus - 1) * dataBytes / cpuMS.count() * nIters / 1e6);
|
||||
|
||||
// exit
|
||||
for (int local = 0; local < nGpus; local++) {
|
||||
@@ -214,18 +212,17 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu,
|
||||
memset(hostMem1.data(), 0, dataBytes);
|
||||
if (toRemote) {
|
||||
HIP_CHECK(hipMemcpy(hostMem1.data(), dstBuf[remote * nGpus + local], dataBytes,
|
||||
hipMemcpyDeviceToHost));
|
||||
}
|
||||
else {
|
||||
hipMemcpyDeviceToHost));
|
||||
} else {
|
||||
HIP_CHECK(hipMemcpy(hostMem1.data(), dstBuf[local * nGpus + remote], dataBytes,
|
||||
hipMemcpyDeviceToHost));
|
||||
hipMemcpyDeviceToHost));
|
||||
}
|
||||
REQUIRE(hostMem1 == hostMem0);
|
||||
} else if (!onOneGpu) {
|
||||
// All dstBuf will be enumed regardless of toRemote
|
||||
memset(hostMem1.data(), 0, dataBytes);
|
||||
HIP_CHECK(hipMemcpy(hostMem1.data(), dstBuf[local * nGpus + remote], dataBytes,
|
||||
hipMemcpyDeviceToHost));
|
||||
hipMemcpyDeviceToHost));
|
||||
REQUIRE(hostMem1 == hostMem0);
|
||||
}
|
||||
#endif
|
||||
@@ -245,8 +242,8 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu) {
|
||||
#if HT_AMD
|
||||
for (int srcType = COARSE_GRAINED; srcType < UNKNOWN_MEM; srcType++) {
|
||||
for (int dstType = COARSE_GRAINED; dstType < UNKNOWN_MEM; dstType++) {
|
||||
testCopyPerf(toRemote, kernelCopy, onOneGpu,
|
||||
static_cast<DEV_MEM_TYPE>(srcType), static_cast<DEV_MEM_TYPE>(dstType));
|
||||
testCopyPerf(toRemote, kernelCopy, onOneGpu, static_cast<DEV_MEM_TYPE>(srcType),
|
||||
static_cast<DEV_MEM_TYPE>(dstType));
|
||||
}
|
||||
}
|
||||
#else
|
||||
@@ -271,7 +268,7 @@ static void testCopyPerf(bool toRemote, bool kernelCopy, bool onOneGpu) {
|
||||
* - HIP_VERSION >= 6.0
|
||||
*/
|
||||
TEST_CASE("Perf_PerfBufferCopySpeedAll2All_test - hipMemcpyPeerAsync - remotes to local") {
|
||||
testCopyPerf(false, false, false);
|
||||
testCopyPerf(false, false, false);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -414,6 +411,6 @@ TEST_CASE("Perf_PerfBufferCopySpeedOne2All_test - kernel copy - local to remotes
|
||||
}
|
||||
|
||||
/**
|
||||
* End doxygen group perfMemoryTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group perfMemoryTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -18,13 +18,13 @@ THE SOFTWARE.
|
||||
*/
|
||||
|
||||
/**
|
||||
* @addtogroup hipMemcpyAsync
|
||||
* @{
|
||||
* @ingroup perfMemoryTest
|
||||
* `hipMemcpyAsync(void* dst, const void* src, size_t count,
|
||||
* hipMemcpyKind kind, hipStream_t stream = 0)` -
|
||||
* Copies data between devices.
|
||||
*/
|
||||
* @addtogroup hipMemcpyAsync
|
||||
* @{
|
||||
* @ingroup perfMemoryTest
|
||||
* `hipMemcpyAsync(void* dst, const void* src, size_t count,
|
||||
* hipMemcpyKind kind, hipStream_t stream = 0)` -
|
||||
* Copies data between devices.
|
||||
*/
|
||||
|
||||
#include <hip_test_common.hh>
|
||||
#include <hip_array_common.hh>
|
||||
@@ -33,27 +33,27 @@ THE SOFTWARE.
|
||||
#include <sstream>
|
||||
#include <iomanip>
|
||||
using namespace std;
|
||||
typedef long long T; // You may change to any type
|
||||
typedef long long T; // You may change to any type
|
||||
|
||||
//#define VERIFY_DATA
|
||||
// #define VERIFY_DATA
|
||||
enum TIMING_MODE { TIMING_MODE_CPU, TIMING_MODE_GPU };
|
||||
|
||||
// -sizes are in bytes, +sizes are in kb, last size must be largest
|
||||
#if 1
|
||||
static constexpr int sizes[] = {-64, -256, -512, 1, 2, 4, 8, 16, 32, 64, 128, 256, 512, 1024,
|
||||
2048, 4096, 8192, 16384, 32768, 65536, 131072, 262144};
|
||||
static constexpr int sizes[] = {-64, -256, -512, 1, 2, 4, 8, 16,
|
||||
32, 64, 128, 256, 512, 1024, 2048, 4096,
|
||||
8192, 16384, 32768, 65536, 131072, 262144};
|
||||
#else
|
||||
static constexpr int sizes[] = { 262144 };
|
||||
static constexpr int sizes[] = {262144};
|
||||
#endif
|
||||
static constexpr int nSizes = sizeof(sizes) / sizeof(sizes[0]);
|
||||
static constexpr double megaSize = 1000000.;
|
||||
static constexpr int defaultIterations = 200;
|
||||
static constexpr unsigned threadsPerBlock = 1024;
|
||||
|
||||
template <typename T>
|
||||
static __global__ void copy_kernel(T* dst, T* src, size_t N) {
|
||||
template <typename T> static __global__ void copy_kernel(T* dst, T* src, size_t N) {
|
||||
size_t idx = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
if(idx < N) dst[idx] = src[idx]; // We make sure idx < N
|
||||
if (idx < N) dst[idx] = src[idx]; // We make sure idx < N
|
||||
}
|
||||
|
||||
static size_t sizeToBytes(int size) { return (size < 0) ? -size : size * 1024; }
|
||||
@@ -68,7 +68,7 @@ static string sizeToString(int size) {
|
||||
return ss.str();
|
||||
}
|
||||
|
||||
static void checkP2PSupport(){
|
||||
static void checkP2PSupport() {
|
||||
int deviceCnt = 0;
|
||||
HIP_CHECK(hipGetDeviceCount(&deviceCnt));
|
||||
cout << "Total no. of available gpu #" << deviceCnt << "\n" << endl;
|
||||
@@ -87,8 +87,7 @@ static void checkP2PSupport(){
|
||||
++PeerCnt;
|
||||
}
|
||||
}
|
||||
if (PeerCnt == 0)
|
||||
cout << "NONE" << " ";
|
||||
if (PeerCnt == 0) cout << "NONE" << " ";
|
||||
|
||||
cout << std::endl;
|
||||
cout << " peer2peer not supported : ";
|
||||
@@ -101,21 +100,20 @@ static void checkP2PSupport(){
|
||||
++nonPeerCnt;
|
||||
}
|
||||
}
|
||||
if (nonPeerCnt == 0)
|
||||
cout << "NONE" << " ";
|
||||
if (nonPeerCnt == 0) cout << "NONE" << " ";
|
||||
|
||||
cout << "\n" << endl;
|
||||
}
|
||||
|
||||
cout << "\nNote: For non-supported peer2peer devices, memcopy will use/follow the normal "
|
||||
"behaviour (GPU1-->host then host-->GPU2)\n\n"
|
||||
<< endl;
|
||||
"behaviour (GPU1-->host then host-->GPU2)\n\n"
|
||||
<< endl;
|
||||
}
|
||||
|
||||
static void outputMatrix(const string &title, const int numGPUs, const vector<double> &data,
|
||||
const TIMING_MODE mode, const size_t dataSize, const int iterations) {
|
||||
fprintf(stderr, "%s, Timing %s, Data %zu KB, Iterations %d\n ",
|
||||
title.c_str(), mode == TIMING_MODE_GPU ? "GPU" : "CPU", dataSize, iterations);
|
||||
static void outputMatrix(const string& title, const int numGPUs, const vector<double>& data,
|
||||
const TIMING_MODE mode, const size_t dataSize, const int iterations) {
|
||||
fprintf(stderr, "%s, Timing %s, Data %zu KB, Iterations %d\n ", title.c_str(),
|
||||
mode == TIMING_MODE_GPU ? "GPU" : "CPU", dataSize, iterations);
|
||||
for (int j = 0; j < numGPUs; j++) {
|
||||
fprintf(stderr, "%9d ", j);
|
||||
}
|
||||
@@ -130,7 +128,7 @@ static void outputMatrix(const string &title, const int numGPUs, const vector<do
|
||||
}
|
||||
|
||||
static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingMode,
|
||||
const bool useHipMemcpyAsync) {
|
||||
const bool useHipMemcpyAsync) {
|
||||
const char* method = useHipMemcpyAsync ? "hipMemcpyAsync()" : "copy kernel";
|
||||
int gpuCount = 0;
|
||||
HIP_CHECK(hipGetDeviceCount(&gpuCount));
|
||||
@@ -150,8 +148,8 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
|
||||
for (int peerGpu = 0; peerGpu < gpuCount; peerGpu++) {
|
||||
HIP_CHECK(hipSetDevice(currentGpu));
|
||||
|
||||
fprintf(stderr, "Uni: Gpu%d -> Gpu%d by %s, Timing %s, Iterations %d\n",
|
||||
currentGpu, peerGpu, method, timingMode == TIMING_MODE_GPU ? "GPU" : "CPU", iterations);
|
||||
fprintf(stderr, "Uni: Gpu%d -> Gpu%d by %s, Timing %s, Iterations %d\n", currentGpu, peerGpu,
|
||||
method, timingMode == TIMING_MODE_GPU ? "GPU" : "CPU", iterations);
|
||||
|
||||
if (currentGpu != peerGpu) {
|
||||
int canAccessPeer = 0;
|
||||
@@ -173,18 +171,16 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
|
||||
#ifdef VERIFY_DATA
|
||||
HIP_CHECK(hipMemcpy(currentGpuMem, hostMem0.data(), numMax, hipMemcpyHostToDevice));
|
||||
#endif
|
||||
unsigned N = numMax / sizeof(T); // Number of T in buffer of numMax bytes.
|
||||
REQUIRE(N * sizeof(T) == numMax); // To prevent verification failure
|
||||
unsigned N = numMax / sizeof(T); // Number of T in buffer of numMax bytes.
|
||||
REQUIRE(N * sizeof(T) == numMax); // To prevent verification failure
|
||||
unsigned blocks = (N + threadsPerBlock - 1) / threadsPerBlock;
|
||||
// Warmup
|
||||
if (useHipMemcpyAsync) {
|
||||
HIP_CHECK(hipMemcpyAsync(peerGpuMem, currentGpuMem, numMax,
|
||||
hipMemcpyDeviceToDevice, 0));
|
||||
}
|
||||
else {
|
||||
HIP_CHECK(hipMemcpyAsync(peerGpuMem, currentGpuMem, numMax, hipMemcpyDeviceToDevice, 0));
|
||||
} else {
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, 0,
|
||||
reinterpret_cast<T*>(peerGpuMem), reinterpret_cast<T*>(currentGpuMem),
|
||||
static_cast<size_t>(N));
|
||||
reinterpret_cast<T*>(peerGpuMem), reinterpret_cast<T*>(currentGpuMem),
|
||||
static_cast<size_t>(N));
|
||||
HIP_CHECK(hipGetLastError());
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
@@ -193,7 +189,7 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
|
||||
HIP_CHECK(hipMemcpy(hostMem1.data(), peerGpuMem, numMax, hipMemcpyDeviceToHost));
|
||||
REQUIRE(hostMem1 == hostMem0);
|
||||
#endif
|
||||
float t = 0; // in ms
|
||||
float t = 0; // in ms
|
||||
auto cpuStart = std::chrono::steady_clock::now();
|
||||
hipEvent_t eventStart, eventStop;
|
||||
hipStream_t stream;
|
||||
@@ -216,17 +212,17 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
|
||||
HIP_CHECK(hipEventRecord(eventStart, stream));
|
||||
}
|
||||
|
||||
for (size_t offsetEnd = numMax - nbytes, offset = 0, j = 0;
|
||||
j < iterations; j++, offset += nbytes) {
|
||||
for (size_t offsetEnd = numMax - nbytes, offset = 0, j = 0; j < iterations;
|
||||
j++, offset += nbytes) {
|
||||
if (offset > offsetEnd) offset = 0;
|
||||
if (useHipMemcpyAsync) {
|
||||
HIP_CHECK(hipMemcpyAsync(peerGpuMem + offset,
|
||||
currentGpuMem + offset, nbytes, hipMemcpyDeviceToDevice, stream));
|
||||
}
|
||||
else {
|
||||
HIP_CHECK(hipMemcpyAsync(peerGpuMem + offset, currentGpuMem + offset, nbytes,
|
||||
hipMemcpyDeviceToDevice, stream));
|
||||
} else {
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, stream,
|
||||
reinterpret_cast<T*>(peerGpuMem + offset),
|
||||
reinterpret_cast<T*>(currentGpuMem + offset), static_cast<size_t>(N));
|
||||
reinterpret_cast<T*>(peerGpuMem + offset),
|
||||
reinterpret_cast<T*>(currentGpuMem + offset),
|
||||
static_cast<size_t>(N));
|
||||
HIP_CHECK(hipGetLastError());
|
||||
}
|
||||
}
|
||||
@@ -237,14 +233,14 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
|
||||
} else if (timingMode == TIMING_MODE_CPU) {
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
std::chrono::duration<double, std::milli> cpuMs =
|
||||
std::chrono::steady_clock::now() - cpuStart;
|
||||
std::chrono::steady_clock::now() - cpuStart;
|
||||
t = cpuMs.count();
|
||||
}
|
||||
t /= iterations;
|
||||
double bandwidth = nbytes / megaSize / t; // GByte/s
|
||||
double bandwidth = nbytes / megaSize / t; // GByte/s
|
||||
|
||||
fprintf(stderr, "%8s, %-9.06lf, %-4.08lf\n",
|
||||
sizeToString(thisSize).c_str(), t, bandwidth);
|
||||
fprintf(stderr, "%8s, %-9.06lf, %-4.08lf\n", sizeToString(thisSize).c_str(),
|
||||
t, bandwidth);
|
||||
if (i == (nSizes - 1)) {
|
||||
timeMs[currentGpu * gpuCount + peerGpu] = t;
|
||||
bandWidth[currentGpu * gpuCount + peerGpu] = bandwidth;
|
||||
@@ -269,10 +265,10 @@ static void testP2PUniDirMemPerf(const int iterations, const TIMING_MODE timingM
|
||||
HIP_CHECK(hipStreamDestroy(stream));
|
||||
}
|
||||
}
|
||||
outputMatrix(string("Unidirectional ") + method + " Time Table(ms)", gpuCount, timeMs,
|
||||
timingMode, sizes[nSizes - 1], iterations);
|
||||
outputMatrix(string("Unidirectional ") + method + " Bandwith Table(GB/s)", gpuCount,
|
||||
bandWidth, timingMode, sizes[nSizes - 1], iterations);
|
||||
outputMatrix(string("Unidirectional ") + method + " Time Table(ms)", gpuCount, timeMs, timingMode,
|
||||
sizes[nSizes - 1], iterations);
|
||||
outputMatrix(string("Unidirectional ") + method + " Bandwith Table(GB/s)", gpuCount, bandWidth,
|
||||
timingMode, sizes[nSizes - 1], iterations);
|
||||
}
|
||||
|
||||
static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsync) {
|
||||
@@ -289,8 +285,8 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
|
||||
for (int currentGpu = 0; currentGpu < gpuCount; currentGpu++) {
|
||||
for (int peerGpu = 0; peerGpu < gpuCount; peerGpu++) {
|
||||
HIP_CHECK(hipSetDevice(currentGpu));
|
||||
fprintf(stderr, "Bi: Gpu%d <-> Gpu%d by %s, Timing GPU, Iterations %d\n",
|
||||
currentGpu, peerGpu, method, iterations);
|
||||
fprintf(stderr, "Bi: Gpu%d <-> Gpu%d by %s, Timing GPU, Iterations %d\n", currentGpu, peerGpu,
|
||||
method, iterations);
|
||||
|
||||
if (currentGpu != peerGpu) {
|
||||
int canAccessPeer = 0;
|
||||
@@ -310,10 +306,13 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
|
||||
HIP_CHECK(hipSetDevice(currentGpu));
|
||||
HIP_CHECK(hipDeviceEnablePeerAccess(peerGpu, 0));
|
||||
}
|
||||
fprintf(stderr, "Gpu%d -> Gpu%d *"
|
||||
"* Gpu%d -> Gpu%d\n", currentGpu, peerGpu, peerGpu, currentGpu);
|
||||
fprintf(stderr, "Size(KB) Time(ms) Bandwidth(GB/s) "
|
||||
" Time(ms) Bandwidth(GB/s)\n");
|
||||
fprintf(stderr,
|
||||
"Gpu%d -> Gpu%d *"
|
||||
"* Gpu%d -> Gpu%d\n",
|
||||
currentGpu, peerGpu, peerGpu, currentGpu);
|
||||
fprintf(stderr,
|
||||
"Size(KB) Time(ms) Bandwidth(GB/s) "
|
||||
" Time(ms) Bandwidth(GB/s)\n");
|
||||
|
||||
unsigned char *currentGpuMem[2], *peerGpuMem[2];
|
||||
HIP_CHECK(hipMalloc((void**)¤tGpuMem[0], numMax));
|
||||
@@ -326,30 +325,31 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
|
||||
|
||||
HIP_CHECK(hipSetDevice(currentGpu));
|
||||
|
||||
unsigned N = numMax / sizeof(T); // Number of T in buffer of numMax bytes.
|
||||
unsigned N = numMax / sizeof(T); // Number of T in buffer of numMax bytes.
|
||||
REQUIRE(N * sizeof(T) == numMax);
|
||||
unsigned blocks = (N + threadsPerBlock - 1) / threadsPerBlock;
|
||||
|
||||
// Warmup. currentGpu is the current device
|
||||
if (useHipMemcpyAsync) {
|
||||
HIP_CHECK(hipMemcpyAsync(peerGpuMem[0], currentGpuMem[0], numMax, hipMemcpyDeviceToDevice, 0));
|
||||
HIP_CHECK(
|
||||
hipMemcpyAsync(peerGpuMem[0], currentGpuMem[0], numMax, hipMemcpyDeviceToDevice, 0));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
HIP_CHECK(hipSetDevice(peerGpu));
|
||||
HIP_CHECK(hipMemcpyAsync(currentGpuMem[1], peerGpuMem[1], numMax, hipMemcpyDeviceToDevice, 0));
|
||||
HIP_CHECK(
|
||||
hipMemcpyAsync(currentGpuMem[1], peerGpuMem[1], numMax, hipMemcpyDeviceToDevice, 0));
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
// Warmup
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
0, reinterpret_cast<T*>(peerGpuMem[0]),
|
||||
reinterpret_cast<T*>(currentGpuMem[0]), static_cast<size_t>(N));
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, 0,
|
||||
reinterpret_cast<T*>(peerGpuMem[0]),
|
||||
reinterpret_cast<T*>(currentGpuMem[0]), static_cast<size_t>(N));
|
||||
HIP_CHECK(hipGetLastError());
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
|
||||
HIP_CHECK(hipSetDevice(peerGpu));
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
0, reinterpret_cast<T*>(currentGpuMem[1]),
|
||||
reinterpret_cast<T*>(peerGpuMem[1]), static_cast<size_t>(N));
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, 0,
|
||||
reinterpret_cast<T*>(currentGpuMem[1]),
|
||||
reinterpret_cast<T*>(peerGpuMem[1]), static_cast<size_t>(N));
|
||||
HIP_CHECK(hipGetLastError());
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
}
|
||||
@@ -377,21 +377,22 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
|
||||
HIP_CHECK(hipEventRecord(eventStart[1], stream[1]));
|
||||
|
||||
for (size_t offsetEnd = numMax - nbytes, offset = 0, j = 0; j < iterations;
|
||||
j++, offset += nbytes) {
|
||||
j++, offset += nbytes) {
|
||||
if (offset > offsetEnd) offset = 0;
|
||||
if (useHipMemcpyAsync) {
|
||||
HIP_CHECK(hipMemcpyAsync(peerGpuMem[0] + offset, currentGpuMem[0] + offset, nbytes,
|
||||
hipMemcpyDeviceToDevice, stream[0]));
|
||||
hipMemcpyDeviceToDevice, stream[0]));
|
||||
HIP_CHECK(hipMemcpyAsync(currentGpuMem[1] + offset, peerGpuMem[1] + offset, nbytes,
|
||||
hipMemcpyDeviceToDevice, stream[1]));
|
||||
}
|
||||
else {
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
stream[0], reinterpret_cast<T*>(peerGpuMem[0] + offset),
|
||||
reinterpret_cast<T*>(currentGpuMem[0] + offset), static_cast<size_t>(N));
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0,
|
||||
stream[1], reinterpret_cast<T*>(currentGpuMem[1] + offset),
|
||||
reinterpret_cast<T*>(peerGpuMem[1] + offset), static_cast<size_t>(N));
|
||||
hipMemcpyDeviceToDevice, stream[1]));
|
||||
} else {
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, stream[0],
|
||||
reinterpret_cast<T*>(peerGpuMem[0] + offset),
|
||||
reinterpret_cast<T*>(currentGpuMem[0] + offset),
|
||||
static_cast<size_t>(N));
|
||||
hipLaunchKernelGGL(copy_kernel<T>, dim3(blocks), dim3(threadsPerBlock), 0, stream[1],
|
||||
reinterpret_cast<T*>(currentGpuMem[1] + offset),
|
||||
reinterpret_cast<T*>(peerGpuMem[1] + offset),
|
||||
static_cast<size_t>(N));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -406,12 +407,13 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
|
||||
for (int n = 0; n < 2; n++) {
|
||||
HIP_CHECK(hipEventElapsedTime(&t[n], eventStart[n], eventStop[n]));
|
||||
t[n] /= iterations;
|
||||
bandwidth[n] = nbytes / megaSize / t[n]; // GByte/s
|
||||
bandwidth[n] = nbytes / megaSize / t[n]; // GByte/s
|
||||
}
|
||||
|
||||
fprintf(stderr, "%8s, %-9.06lf, %-4.08lf, "
|
||||
"%-9.06lf, %-4.08lf\n", sizeToString(thisSize).c_str(), t[0],
|
||||
bandwidth[0], t[1], bandwidth[1]);
|
||||
fprintf(stderr,
|
||||
"%8s, %-9.06lf, %-4.08lf, "
|
||||
"%-9.06lf, %-4.08lf\n",
|
||||
sizeToString(thisSize).c_str(), t[0], bandwidth[0], t[1], bandwidth[1]);
|
||||
|
||||
if (i == (nSizes - 1)) {
|
||||
timeMs[currentGpu * gpuCount + peerGpu] = (t[0] + t[1]) / 2;
|
||||
@@ -434,9 +436,9 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
|
||||
}
|
||||
}
|
||||
outputMatrix(string("Bidirectional ") + method + " Time Table(ms)", gpuCount, timeMs,
|
||||
TIMING_MODE_GPU, sizes[nSizes - 1], iterations);
|
||||
outputMatrix(string("Bidirectional ") + method + " Bandwith Table(GB/s)", gpuCount,
|
||||
bandWidth, TIMING_MODE_GPU, sizes[nSizes - 1], iterations);
|
||||
TIMING_MODE_GPU, sizes[nSizes - 1], iterations);
|
||||
outputMatrix(string("Bidirectional ") + method + " Bandwith Table(GB/s)", gpuCount, bandWidth,
|
||||
TIMING_MODE_GPU, sizes[nSizes - 1], iterations);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -455,14 +457,14 @@ static void testP2PBiDirMemPerf(const int iterations, const bool useHipMemcpyAsy
|
||||
* - HIP_VERSION >= 6.0
|
||||
*/
|
||||
TEST_CASE("Perf_hipTestP2PUniDirMemcpyAsync_test - Timing CPU") {
|
||||
const int iterations = cmd_options.iterations == 1000 ?
|
||||
defaultIterations : cmd_options.iterations;
|
||||
const int iterations =
|
||||
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
|
||||
testP2PUniDirMemPerf(iterations, TIMING_MODE_CPU, true);
|
||||
}
|
||||
|
||||
TEST_CASE("Perf_hipTestP2PUniDirMemcpyAsync_test - Timing GPU") {
|
||||
const int iterations = cmd_options.iterations == 1000 ?
|
||||
defaultIterations : cmd_options.iterations;
|
||||
const int iterations =
|
||||
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
|
||||
testP2PUniDirMemPerf(iterations, TIMING_MODE_GPU, true);
|
||||
}
|
||||
|
||||
@@ -480,14 +482,14 @@ TEST_CASE("Perf_hipTestP2PUniDirMemcpyAsync_test - Timing GPU") {
|
||||
* - HIP_VERSION >= 6.0
|
||||
*/
|
||||
TEST_CASE("Perf_hipTestP2PUniDirKernelCopy_test - Timing CPU") {
|
||||
const int iterations = cmd_options.iterations == 1000 ?
|
||||
defaultIterations : cmd_options.iterations;
|
||||
const int iterations =
|
||||
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
|
||||
testP2PUniDirMemPerf(iterations, TIMING_MODE_CPU, false);
|
||||
}
|
||||
|
||||
TEST_CASE("Perf_hipTestP2PUniDirKernelCopy_test - Timing GPU") {
|
||||
const int iterations = cmd_options.iterations == 1000 ?
|
||||
defaultIterations : cmd_options.iterations;
|
||||
const int iterations =
|
||||
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
|
||||
testP2PUniDirMemPerf(iterations, TIMING_MODE_GPU, false);
|
||||
}
|
||||
|
||||
@@ -507,8 +509,8 @@ TEST_CASE("Perf_hipTestP2PUniDirKernelCopy_test - Timing GPU") {
|
||||
* - HIP_VERSION >= 6.0
|
||||
*/
|
||||
TEST_CASE("Perf_hipTestP2PBiDirMemcpyAsync_test") {
|
||||
const int iterations = cmd_options.iterations == 1000 ?
|
||||
defaultIterations : cmd_options.iterations;
|
||||
const int iterations =
|
||||
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
|
||||
testP2PBiDirMemPerf(iterations, true);
|
||||
}
|
||||
|
||||
@@ -526,9 +528,9 @@ TEST_CASE("Perf_hipTestP2PBiDirMemcpyAsync_test") {
|
||||
* - HIP_VERSION >= 6.0
|
||||
*/
|
||||
TEST_CASE("Perf_hipTestP2PBiDirKernelCopy_test") {
|
||||
const int iterations = cmd_options.iterations == 1000 ?
|
||||
defaultIterations : cmd_options.iterations;
|
||||
testP2PBiDirMemPerf(iterations, false);
|
||||
const int iterations =
|
||||
cmd_options.iterations == 1000 ? defaultIterations : cmd_options.iterations;
|
||||
testP2PBiDirMemPerf(iterations, false);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -542,11 +544,9 @@ TEST_CASE("Perf_hipTestP2PBiDirKernelCopy_test") {
|
||||
* ------------------------
|
||||
* - HIP_VERSION >= 6.0
|
||||
*/
|
||||
TEST_CASE("Perf_hipCheckP2PSupport") {
|
||||
checkP2PSupport();
|
||||
}
|
||||
TEST_CASE("Perf_hipCheckP2PSupport") { checkP2PSupport(); }
|
||||
|
||||
/**
|
||||
* End doxygen group perfMemoryTest.
|
||||
* @}
|
||||
*/
|
||||
* End doxygen group perfMemoryTest.
|
||||
* @}
|
||||
*/
|
||||
|
||||
@@ -25,12 +25,11 @@ __global__ void mallocTest() {
|
||||
memset(ptr, 0, size);
|
||||
free(ptr);
|
||||
}
|
||||
__global__ void mallocTest_1()
|
||||
{
|
||||
size_t size = 1024;
|
||||
int* ptr = (int*)malloc(size);
|
||||
memset(ptr, 0, size);
|
||||
free(ptr);
|
||||
__global__ void mallocTest_1() {
|
||||
size_t size = 1024;
|
||||
int* ptr = (int*)malloc(size);
|
||||
memset(ptr, 0, size);
|
||||
free(ptr);
|
||||
}
|
||||
/**
|
||||
* The tests in this file are added to see the performance improvement with the
|
||||
@@ -64,7 +63,7 @@ __global__ void mallocTest_1()
|
||||
* - HIP_VERSION >= 6.5
|
||||
*/
|
||||
TEST_CASE("Unit_Perf_Device_Heap_Memory_Allocation") {
|
||||
HIP_CHECK(hipDeviceSetLimit(hipLimitMallocHeapSize, 128*1024*1024));
|
||||
HIP_CHECK(hipDeviceSetLimit(hipLimitMallocHeapSize, 128 * 1024 * 1024));
|
||||
hipEvent_t event;
|
||||
HIP_CHECK(hipEventCreate(&event));
|
||||
REQUIRE(event != nullptr);
|
||||
@@ -87,8 +86,8 @@ TEST_CASE("Unit_Perf_Device_Heap_Memory_Allocation") {
|
||||
REQUIRE(time > time_1);
|
||||
HIP_CHECK(hipEventDestroy(event));
|
||||
HIP_CHECK(hipStreamDestroy(stream));
|
||||
std::cout<<"First Kernel Latency: "<<time<<" micro seconds"<<std::endl;
|
||||
std::cout<<"Second Kernel Latency: "<<time_1<<" micro seconds"<<std::endl;
|
||||
std::cout << "First Kernel Latency: " << time << " micro seconds" << std::endl;
|
||||
std::cout << "Second Kernel Latency: " << time_1 << " micro seconds" << std::endl;
|
||||
}
|
||||
/**
|
||||
* End doxygen group PerformanceTest.
|
||||
|
||||
@@ -29,14 +29,12 @@ THE SOFTWARE.
|
||||
* Helper function to get and print the Total device and free device memory,
|
||||
* and reserved current, used current memory from pool.
|
||||
*/
|
||||
void getAndPrintMemoryDetails(const hipMemPool_t &pool) {
|
||||
void getAndPrintMemoryDetails(const hipMemPool_t& pool) {
|
||||
size_t freeVRAM = 0, totalVRAM = 0, reservedCurrent = 0, usedCurrent = 0;
|
||||
|
||||
HIP_CHECK(hipMemGetInfo(&freeVRAM, &totalVRAM));
|
||||
HIP_CHECK(hipMemPoolGetAttribute(pool, hipMemPoolAttrReservedMemCurrent,
|
||||
&reservedCurrent));
|
||||
HIP_CHECK(hipMemPoolGetAttribute(pool, hipMemPoolAttrUsedMemCurrent,
|
||||
&usedCurrent));
|
||||
HIP_CHECK(hipMemPoolGetAttribute(pool, hipMemPoolAttrReservedMemCurrent, &reservedCurrent));
|
||||
HIP_CHECK(hipMemPoolGetAttribute(pool, hipMemPoolAttrUsedMemCurrent, &usedCurrent));
|
||||
|
||||
std::cout << "\n Total device memory (GB) : " << totalVRAM / 1_GB;
|
||||
std::cout << "\n Free device memory (GB) : " << freeVRAM / 1_GB;
|
||||
@@ -80,8 +78,7 @@ TEST_CASE("Perf_MempoolManager_hipMallocAsync_hipFreeAsync") {
|
||||
HIP_CHECK(hipDeviceGetDefaultMemPool(&pool, device));
|
||||
|
||||
uint64_t threshold = 30_GB;
|
||||
HIP_CHECK(hipMemPoolSetAttribute(pool, hipMemPoolAttrReleaseThreshold,
|
||||
&threshold));
|
||||
HIP_CHECK(hipMemPoolSetAttribute(pool, hipMemPoolAttrReleaseThreshold, &threshold));
|
||||
|
||||
std::cout << "\n Memory details at start : ";
|
||||
getAndPrintMemoryDetails(pool);
|
||||
@@ -90,7 +87,7 @@ TEST_CASE("Perf_MempoolManager_hipMallocAsync_hipFreeAsync") {
|
||||
HIP_CHECK(hipStreamCreate(&stream));
|
||||
|
||||
constexpr int ptrs = 20;
|
||||
void *dPtr[ptrs];
|
||||
void* dPtr[ptrs];
|
||||
for (int i = 0; i < ptrs; i++) {
|
||||
dPtr[i] = nullptr;
|
||||
}
|
||||
|
||||
@@ -23,9 +23,7 @@
|
||||
|
||||
using namespace std;
|
||||
|
||||
__global__
|
||||
static void _noop_kernel() {
|
||||
}
|
||||
__global__ static void _noop_kernel() {}
|
||||
|
||||
|
||||
TEST_CASE("Perf_KernelLaunchLatency_IncreasingNumberOfStreams") {
|
||||
@@ -67,12 +65,11 @@ TEST_CASE("Perf_KernelLaunchLatency_IncreasingNumberOfStreams") {
|
||||
for (auto& numHipStreams : streamsNumber) {
|
||||
vector<hipStream_t> streams(numHipStreams);
|
||||
|
||||
if(isBlocking) {
|
||||
if (isBlocking) {
|
||||
for (int i = 0; i < numHipStreams; ++i) {
|
||||
HIP_CHECK(hipStreamCreate(&streams[i]));
|
||||
}
|
||||
}
|
||||
else {
|
||||
} else {
|
||||
for (int i = 0; i < numHipStreams; ++i) {
|
||||
HIP_CHECK(hipStreamCreateWithFlags(&streams[i], hipStreamNonBlocking));
|
||||
}
|
||||
|
||||
@@ -48,7 +48,7 @@ class hipPerfStreamCreateCopyDestroy {
|
||||
numStreams_(0),
|
||||
totalStreams_{1, 2, 4, 8},
|
||||
totalBuffers_{1, 100, 1000, 5000} {};
|
||||
~hipPerfStreamCreateCopyDestroy(){};
|
||||
~hipPerfStreamCreateCopyDestroy() {};
|
||||
bool open(int deviceID);
|
||||
bool run(unsigned int testNumber);
|
||||
};
|
||||
|
||||
@@ -89,20 +89,19 @@ bool ValidateUsingCopy(int deviceId, void* dev_ptr, size_t data_size,
|
||||
HIP_CHECK(hipMemcpy(dev_ptr, A_h.data(), data_size, hipMemcpyHostToDevice));
|
||||
auto end = std::chrono::high_resolution_clock::now();
|
||||
h2d_elapsed = std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
|
||||
|
||||
start = std::chrono::high_resolution_clock::now();
|
||||
HIP_CHECK(hipMemcpy(B_h.data(), dev_ptr, data_size, hipMemcpyDeviceToHost));
|
||||
end = std::chrono::high_resolution_clock::now();
|
||||
d2h_elapsed = std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
|
||||
|
||||
if (debug_failure) {
|
||||
REQUIRE(true == std::equal(B_h.begin(), B_h.end(), A_h.data()));
|
||||
} else {
|
||||
assert(A_h.size() == B_h.size());
|
||||
for (size_t idx = 0; idx < A_h.size(); ++idx) {
|
||||
if (A_h[idx] != B_h[idx]) {
|
||||
std::cout << "Failed at first index: " << idx
|
||||
<< " Expected: " << A_h[idx]
|
||||
std::cout << "Failed at first index: " << idx << " Expected: " << A_h[idx]
|
||||
<< " Value: " << B_h[idx] << std::endl;
|
||||
break;
|
||||
}
|
||||
@@ -135,8 +134,8 @@ bool TestOnDevice(int deviceId) {
|
||||
auto start = std::chrono::high_resolution_clock::now();
|
||||
HIP_CHECK(hipMemAddressReserve(&dev_ptr, size_idx, granularity, nullptr, 0));
|
||||
auto end = std::chrono::high_resolution_clock::now();
|
||||
std::chrono::microseconds reserve_elapsed
|
||||
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
std::chrono::microseconds reserve_elapsed =
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
std::vector<hipMemGenericAllocationHandle_t> physmem_handles;
|
||||
std::chrono::microseconds alloc_elapsed;
|
||||
std::chrono::microseconds map_elapsed;
|
||||
@@ -233,34 +232,34 @@ bool TestOnDevice(int deviceId) {
|
||||
++chunk_idx;
|
||||
}
|
||||
end = std::chrono::high_resolution_clock::now();
|
||||
std::chrono::microseconds unmap_elapsed
|
||||
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
std::chrono::microseconds unmap_elapsed =
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
|
||||
start = std::chrono::high_resolution_clock::now();
|
||||
for (auto& physmem_handle : physmem_handles) {
|
||||
HIP_CHECK(hipMemRelease(physmem_handle));
|
||||
}
|
||||
end = std::chrono::high_resolution_clock::now();
|
||||
std::chrono::microseconds release_elapsed
|
||||
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
std::chrono::microseconds release_elapsed =
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
|
||||
start = std::chrono::high_resolution_clock::now();
|
||||
HIP_CHECK(hipMemAddressFree(dev_ptr, size_idx));
|
||||
end = std::chrono::high_resolution_clock::now();
|
||||
std::chrono::microseconds free_elapsed
|
||||
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
std::chrono::microseconds free_elapsed =
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
|
||||
// Print the results
|
||||
std::cout << "-------------Size: " << (size_idx / kGB) << " GB----------------" << std::endl;
|
||||
std::cout << "Time taken to reserve : " << reserve_elapsed.count()
|
||||
<< " micro seconds and free: " << free_elapsed.count()
|
||||
<< " micro seconds" << std::endl;
|
||||
std::cout <<"Time taken to alloc : " << alloc_elapsed.count()
|
||||
<< " micro seconds and release: "<< release_elapsed.count()
|
||||
<< " micro seconds" << std::endl;
|
||||
<< " micro seconds and free: " << free_elapsed.count() << " micro seconds"
|
||||
<< std::endl;
|
||||
std::cout << "Time taken to alloc : " << alloc_elapsed.count()
|
||||
<< " micro seconds and release: " << release_elapsed.count() << " micro seconds"
|
||||
<< std::endl;
|
||||
std::cout << "Time taken to map : " << map_elapsed.count()
|
||||
<< " micro seconds and unmap: " << unmap_elapsed.count()
|
||||
<< " micro seconds" << std::endl;
|
||||
<< " micro seconds and unmap: " << unmap_elapsed.count() << " micro seconds"
|
||||
<< std::endl;
|
||||
std::cout << "Time taken to H2D : " << h2d_elapsed.count()
|
||||
<< " micro seconds and D2H: " << d2h_elapsed.count() << " micro seconds" << std::endl;
|
||||
std::cout << "-------------------------/hipMallocPerf------------------------" << std::endl;
|
||||
@@ -269,20 +268,20 @@ bool TestOnDevice(int deviceId) {
|
||||
start = std::chrono::high_resolution_clock::now();
|
||||
HIP_CHECK(hipMalloc(&dev_ptr_legacy, size_idx));
|
||||
end = std::chrono::high_resolution_clock::now();
|
||||
std::chrono::microseconds hm_elapsed
|
||||
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
std::chrono::microseconds hm_elapsed =
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
start = std::chrono::high_resolution_clock::now();
|
||||
HIP_CHECK(hipFree(dev_ptr_legacy));
|
||||
end = std::chrono::high_resolution_clock::now();
|
||||
std::chrono::microseconds hf_elapsed
|
||||
= std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
std::chrono::microseconds hf_elapsed =
|
||||
std::chrono::duration_cast<std::chrono::microseconds>(end - start);
|
||||
std::cout << "Time taken for hipMalloc : " << hm_elapsed.count()
|
||||
<< " micro seconds and hipFree: " << hf_elapsed.count()
|
||||
<< " micro seconds" << std::endl;
|
||||
<< " micro seconds and hipFree: " << hf_elapsed.count() << " micro seconds"
|
||||
<< std::endl;
|
||||
std::cout << "---------------------------------------------------------------" << std::endl;
|
||||
std::cout << std::endl;
|
||||
}
|
||||
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
|
||||
@@ -21,9 +21,9 @@ THE SOFTWARE.
|
||||
#include <hip_test_checkers.hh>
|
||||
#include <unistd.h>
|
||||
// Size Macros
|
||||
#define MEMORY_CHUNK_SIZE (1024*1024)
|
||||
#define MEMORY_CHUNK_SIZE_ODD (1025*1025)
|
||||
#define MAXIMUM_CHUNKS (256*1024)
|
||||
#define MEMORY_CHUNK_SIZE (1024 * 1024)
|
||||
#define MEMORY_CHUNK_SIZE_ODD (1025 * 1025)
|
||||
#define MAXIMUM_CHUNKS (256 * 1024)
|
||||
// Subtest Macros
|
||||
#define NO_ALLOCATION_ONHOST 0
|
||||
#define ALLOCATE_ONHOST_HIPMALLOCMANAGED 1
|
||||
@@ -57,14 +57,12 @@ __device__ static int* dev_common_ptr;
|
||||
* This kernel checks kernel allocation of size more than available
|
||||
* memory.
|
||||
*/
|
||||
static __global__ void kerTestDynamicAllocNeg(int test_type,
|
||||
size_t perThreadSize,
|
||||
int *ret) {
|
||||
static __global__ void kerTestDynamicAllocNeg(int test_type, size_t perThreadSize, int* ret) {
|
||||
// Allocate
|
||||
char* ptr = nullptr;
|
||||
printf("Memory to allocate in GPU = %zu \n", perThreadSize);
|
||||
if (test_type == TEST_MALLOC_FREE) {
|
||||
ptr = reinterpret_cast<char*> (malloc(perThreadSize));
|
||||
ptr = reinterpret_cast<char*>(malloc(perThreadSize));
|
||||
} else {
|
||||
ptr = new char[perThreadSize];
|
||||
}
|
||||
@@ -87,9 +85,8 @@ static __global__ void kerTestDynamicAllocNeg(int test_type,
|
||||
/**
|
||||
* This kernel allocates memory till nullptr is returned.
|
||||
*/
|
||||
static __global__ void kerAllocTillExhaust(int test_type,
|
||||
size_t *total_allocated_mem,
|
||||
size_t mem_chunk_size) {
|
||||
static __global__ void kerAllocTillExhaust(int test_type, size_t* total_allocated_mem,
|
||||
size_t mem_chunk_size) {
|
||||
int myId = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
// Allocate memory in thread 0 of block 0
|
||||
if (0 == myId) {
|
||||
@@ -99,16 +96,14 @@ static __global__ void kerAllocTillExhaust(int test_type,
|
||||
int idx = 0;
|
||||
if (test_type == TEST_MALLOC_FREE) {
|
||||
do {
|
||||
dev_mem_glob[idx] =
|
||||
reinterpret_cast<char*> (malloc(mem_chunk_size));
|
||||
dev_mem_glob[idx] = reinterpret_cast<char*>(malloc(mem_chunk_size));
|
||||
if (idx >= MAXIMUM_CHUNKS) {
|
||||
break;
|
||||
}
|
||||
} while (dev_mem_glob[idx++] != nullptr);
|
||||
} else {
|
||||
do {
|
||||
dev_mem_glob[idx] =
|
||||
reinterpret_cast<char*> (new char[mem_chunk_size]);
|
||||
dev_mem_glob[idx] = reinterpret_cast<char*>(new char[mem_chunk_size]);
|
||||
if (idx >= MAXIMUM_CHUNKS) {
|
||||
break;
|
||||
}
|
||||
@@ -116,8 +111,7 @@ static __global__ void kerAllocTillExhaust(int test_type,
|
||||
}
|
||||
idx = 0;
|
||||
*total_allocated_mem = 0;
|
||||
while ((dev_mem_glob[idx] != nullptr) &&
|
||||
(idx < MAXIMUM_CHUNKS)) {
|
||||
while ((dev_mem_glob[idx] != nullptr) && (idx < MAXIMUM_CHUNKS)) {
|
||||
*total_allocated_mem = *total_allocated_mem + mem_chunk_size;
|
||||
idx++;
|
||||
}
|
||||
@@ -155,18 +149,15 @@ static __global__ void kerFreeAll(int test_type) {
|
||||
* access this memory in all threads of the block. The memory is
|
||||
* finally deleted in last thread of each block.
|
||||
*/
|
||||
static __global__ void kerBlockLevelMemoryAllocation(int *outputBuf,
|
||||
int test_type) {
|
||||
static __global__ void kerBlockLevelMemoryAllocation(int* outputBuf, int test_type) {
|
||||
int myThreadId = threadIdx.x, lastThreadId = (blockDim.x - 1);
|
||||
int myId = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
// Allocate memory in thread 0
|
||||
if (0 == myThreadId) {
|
||||
if (test_type == TEST_MALLOC_FREE) {
|
||||
dev_mem[blockIdx.x] =
|
||||
reinterpret_cast<int*> (malloc(blockDim.x*sizeof(int)));
|
||||
dev_mem[blockIdx.x] = reinterpret_cast<int*>(malloc(blockDim.x * sizeof(int)));
|
||||
} else {
|
||||
dev_mem[blockIdx.x] =
|
||||
reinterpret_cast<int*> (new int[blockDim.x]);
|
||||
dev_mem[blockIdx.x] = reinterpret_cast<int*>(new int[blockDim.x]);
|
||||
}
|
||||
}
|
||||
// All threads wait at this barrier
|
||||
@@ -176,7 +167,7 @@ static __global__ void kerBlockLevelMemoryAllocation(int *outputBuf,
|
||||
printf("Device Allocation Failed in thread = %d \n", myId);
|
||||
return;
|
||||
}
|
||||
int *ptr = reinterpret_cast<int*> (dev_mem[blockIdx.x]);
|
||||
int* ptr = reinterpret_cast<int*>(dev_mem[blockIdx.x]);
|
||||
// Copy to buffer
|
||||
ptr[myThreadId] = myId;
|
||||
// All threads wait
|
||||
@@ -202,11 +193,9 @@ static __global__ void kerAlloc(int test_type) {
|
||||
// Allocate memory in thread 0 of block 0
|
||||
if (0 == myId) {
|
||||
if (test_type == TEST_MALLOC_FREE) {
|
||||
dev_common_ptr =
|
||||
reinterpret_cast<int*> (malloc(blockDim.x*gridDim.x*sizeof(int)));
|
||||
dev_common_ptr = reinterpret_cast<int*>(malloc(blockDim.x * gridDim.x * sizeof(int)));
|
||||
} else {
|
||||
dev_common_ptr =
|
||||
reinterpret_cast<int*> (new int[blockDim.x*gridDim.x]);
|
||||
dev_common_ptr = reinterpret_cast<int*>(new int[blockDim.x * gridDim.x]);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -229,7 +218,7 @@ static __global__ void kerWrite() {
|
||||
* This kernel copies the contents of memory allocated in <kerAlloc>
|
||||
* to host and deletes the memory from thread 0.
|
||||
*/
|
||||
static __global__ void kerFree(int *outputBuf, int test_type) {
|
||||
static __global__ void kerFree(int* outputBuf, int test_type) {
|
||||
int myId = threadIdx.x + blockDim.x * blockIdx.x;
|
||||
// Check allocated memory in all threads in block before access
|
||||
if (dev_common_ptr == nullptr) {
|
||||
@@ -237,7 +226,7 @@ static __global__ void kerFree(int *outputBuf, int test_type) {
|
||||
return;
|
||||
}
|
||||
if (0 == myId) {
|
||||
for (size_t idx = 0; idx < (blockDim.x*gridDim.x); idx++) {
|
||||
for (size_t idx = 0; idx < (blockDim.x * gridDim.x); idx++) {
|
||||
outputBuf[idx] = dev_common_ptr[idx];
|
||||
}
|
||||
if (test_type == TEST_MALLOC_FREE) {
|
||||
@@ -254,8 +243,7 @@ static __global__ void kerFree(int *outputBuf, int test_type) {
|
||||
* kerFreeAll<<<>>> to test memory allocation till all device
|
||||
* memory is exhausted.
|
||||
*/
|
||||
static bool TestAllocationOfAllAvailableMemory(int test_type,
|
||||
int category, size_t mem_chunk_size) {
|
||||
static bool TestAllocationOfAllAvailableMemory(int test_type, int category, size_t mem_chunk_size) {
|
||||
size_t avail1 = 0, avail2 = 0, tot = 0;
|
||||
constexpr size_t host_alloc = 2147483648; // 2 GB
|
||||
HIP_CHECK(hipMemGetInfo(&avail1, &tot));
|
||||
@@ -263,12 +251,11 @@ static bool TestAllocationOfAllAvailableMemory(int test_type,
|
||||
HIP_CHECK(hipDeviceSetLimit(hipLimitMallocHeapSize, avail1));
|
||||
#endif
|
||||
size_t *tot_alloc_mem_d = nullptr, *tot_alloc_mem_h = nullptr;
|
||||
tot_alloc_mem_h =
|
||||
reinterpret_cast<size_t*> (malloc(sizeof(size_t)));
|
||||
tot_alloc_mem_h = reinterpret_cast<size_t*>(malloc(sizeof(size_t)));
|
||||
REQUIRE(nullptr != tot_alloc_mem_h);
|
||||
HIP_CHECK(hipMalloc(&tot_alloc_mem_d, sizeof(size_t)));
|
||||
REQUIRE(nullptr != tot_alloc_mem_d);
|
||||
char *devptrHost = nullptr;
|
||||
char* devptrHost = nullptr;
|
||||
if (category == ALLOCATE_ONHOST_HIPMALLOCMANAGED) {
|
||||
HIP_CHECK(hipMallocManaged(&devptrHost, host_alloc));
|
||||
} else if (category == ALLOCATE_ONHOST_HIPMALLOC) {
|
||||
@@ -278,12 +265,10 @@ static bool TestAllocationOfAllAvailableMemory(int test_type,
|
||||
INFO("Total available memory " << tot);
|
||||
INFO("Available memory before allocation " << avail1);
|
||||
// Launch Test Kernel
|
||||
kerAllocTillExhaust<<<1, 1>>>(test_type, tot_alloc_mem_d,
|
||||
mem_chunk_size);
|
||||
kerAllocTillExhaust<<<1, 1>>>(test_type, tot_alloc_mem_d, mem_chunk_size);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
// Copy to host buffer
|
||||
HIP_CHECK(hipMemcpy(tot_alloc_mem_h, tot_alloc_mem_d,
|
||||
sizeof(size_t), hipMemcpyDefault));
|
||||
HIP_CHECK(hipMemcpy(tot_alloc_mem_h, tot_alloc_mem_d, sizeof(size_t), hipMemcpyDefault));
|
||||
HIP_CHECK(hipMemGetInfo(&avail2, &tot));
|
||||
kerFreeAll<<<1, 1>>>(test_type);
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
@@ -312,11 +297,10 @@ static bool TestAllocationOfAllAvailableMemory(int test_type,
|
||||
* Local function: Launch kerBlockLevelMemoryAllocation<<<>>>
|
||||
* in a loop to stress test allocation and deallocation.
|
||||
*/
|
||||
static bool TestMemoryAllocationInLoop(int test_type,
|
||||
bool isMultikernel = false) {
|
||||
static bool TestMemoryAllocationInLoop(int test_type, bool isMultikernel = false) {
|
||||
int *outputVec_d{nullptr}, *outputVec_h{nullptr};
|
||||
int arraysize = (BLOCKSIZE * GRIDSIZE);
|
||||
outputVec_h = reinterpret_cast<int*> (malloc(sizeof(int) * arraysize));
|
||||
outputVec_h = reinterpret_cast<int*>(malloc(sizeof(int) * arraysize));
|
||||
REQUIRE(outputVec_h != nullptr);
|
||||
HIP_CHECK(hipMalloc(&outputVec_d, (sizeof(int) * arraysize)));
|
||||
bool bPassed = true;
|
||||
@@ -333,13 +317,11 @@ static bool TestMemoryAllocationInLoop(int test_type,
|
||||
kerWrite<<<GRIDSIZE, BLOCKSIZE>>>();
|
||||
kerFree<<<GRIDSIZE, BLOCKSIZE>>>(outputVec_d, test_type);
|
||||
} else {
|
||||
kerBlockLevelMemoryAllocation<<<GRIDSIZE, BLOCKSIZE>>>(outputVec_d,
|
||||
test_type);
|
||||
kerBlockLevelMemoryAllocation<<<GRIDSIZE, BLOCKSIZE>>>(outputVec_d, test_type);
|
||||
}
|
||||
HIP_CHECK(hipDeviceSynchronize());
|
||||
// Copy to host buffer
|
||||
HIP_CHECK(hipMemcpy(outputVec_h, outputVec_d, sizeof(int) * arraysize,
|
||||
hipMemcpyDefault));
|
||||
HIP_CHECK(hipMemcpy(outputVec_h, outputVec_d, sizeof(int) * arraysize, hipMemcpyDefault));
|
||||
bPassed = true;
|
||||
for (int idx = 0; idx < arraysize; idx++) {
|
||||
if (outputVec_h[idx] != idx) {
|
||||
@@ -359,32 +341,36 @@ static bool TestMemoryAllocationInLoop(int test_type,
|
||||
* Scenario: Test malloc till nullptr is returned using even chunksize.
|
||||
*/
|
||||
TEST_CASE("Stress_deviceAllocation_malloc_Even") {
|
||||
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE,
|
||||
NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE));
|
||||
REQUIRE(true ==
|
||||
TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE, NO_ALLOCATION_ONHOST,
|
||||
MEMORY_CHUNK_SIZE));
|
||||
}
|
||||
|
||||
/**
|
||||
* Scenario: Test malloc till nullptr is returned using odd chunksize.
|
||||
*/
|
||||
TEST_CASE("Stress_deviceAllocation_malloc_Odd") {
|
||||
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE,
|
||||
NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE_ODD));
|
||||
REQUIRE(true ==
|
||||
TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE, NO_ALLOCATION_ONHOST,
|
||||
MEMORY_CHUNK_SIZE_ODD));
|
||||
}
|
||||
|
||||
/**
|
||||
* Scenario: Test new till nullptr is returned using even chunksize.
|
||||
*/
|
||||
TEST_CASE("Stress_deviceAllocation_new_Even") {
|
||||
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE,
|
||||
NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE));
|
||||
REQUIRE(
|
||||
true ==
|
||||
TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE, NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE));
|
||||
}
|
||||
|
||||
/**
|
||||
* Scenario: Test new till nullptr is returned using odd chunksize.
|
||||
*/
|
||||
TEST_CASE("Stress_deviceAllocation_new_Odd") {
|
||||
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE,
|
||||
NO_ALLOCATION_ONHOST, MEMORY_CHUNK_SIZE_ODD));
|
||||
REQUIRE(true ==
|
||||
TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE, NO_ALLOCATION_ONHOST,
|
||||
MEMORY_CHUNK_SIZE_ODD));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -393,8 +379,9 @@ TEST_CASE("Stress_deviceAllocation_new_Odd") {
|
||||
* from host.
|
||||
*/
|
||||
TEST_CASE("Stress_deviceAllocation_malloc_hipmallocmanaged") {
|
||||
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE,
|
||||
ALLOCATE_ONHOST_HIPMALLOCMANAGED, MEMORY_CHUNK_SIZE));
|
||||
REQUIRE(true ==
|
||||
TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE, ALLOCATE_ONHOST_HIPMALLOCMANAGED,
|
||||
MEMORY_CHUNK_SIZE));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -403,8 +390,9 @@ TEST_CASE("Stress_deviceAllocation_malloc_hipmallocmanaged") {
|
||||
* from host.
|
||||
*/
|
||||
TEST_CASE("Stress_deviceAllocation_new_hipmallocmanaged") {
|
||||
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE,
|
||||
ALLOCATE_ONHOST_HIPMALLOCMANAGED, MEMORY_CHUNK_SIZE));
|
||||
REQUIRE(true ==
|
||||
TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE, ALLOCATE_ONHOST_HIPMALLOCMANAGED,
|
||||
MEMORY_CHUNK_SIZE));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -412,8 +400,9 @@ TEST_CASE("Stress_deviceAllocation_new_hipmallocmanaged") {
|
||||
* is returned. Device memory is also allocated using hipmalloc from host.
|
||||
*/
|
||||
TEST_CASE("Stress_deviceAllocation_malloc_hipmalloc") {
|
||||
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE,
|
||||
ALLOCATE_ONHOST_HIPMALLOC, MEMORY_CHUNK_SIZE));
|
||||
REQUIRE(true ==
|
||||
TestAllocationOfAllAvailableMemory(TEST_MALLOC_FREE, ALLOCATE_ONHOST_HIPMALLOC,
|
||||
MEMORY_CHUNK_SIZE));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -421,8 +410,9 @@ TEST_CASE("Stress_deviceAllocation_malloc_hipmalloc") {
|
||||
* is returned. Device memory is also allocated using hipmalloc from host.
|
||||
*/
|
||||
TEST_CASE("Stress_deviceAllocation_new_hipmalloc") {
|
||||
REQUIRE(true == TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE,
|
||||
ALLOCATE_ONHOST_HIPMALLOC, MEMORY_CHUNK_SIZE));
|
||||
REQUIRE(true ==
|
||||
TestAllocationOfAllAvailableMemory(TEST_NEW_DELETE, ALLOCATE_ONHOST_HIPMALLOC,
|
||||
MEMORY_CHUNK_SIZE));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -434,7 +424,7 @@ TEST_CASE("Stress_deviceAllocation_Negative") {
|
||||
size_t avail = 0, tot = 0;
|
||||
HIP_CHECK(hipMemGetInfo(&avail, &tot));
|
||||
printf("Available Memory in GPU = %zu \n", avail);
|
||||
ret_h = reinterpret_cast<int*> (malloc(sizeof(int)));
|
||||
ret_h = reinterpret_cast<int*>(malloc(sizeof(int)));
|
||||
REQUIRE(ret_h != nullptr);
|
||||
HIP_CHECK(hipMalloc(&ret_d, (sizeof(int))));
|
||||
SECTION("Test allocation with malloc") {
|
||||
|
||||
@@ -41,8 +41,7 @@ TEST_CASE("Stress_hipHostMalloc_MaxAllocation") {
|
||||
INFO("Max Allocation of " << memFree << " bytes!");
|
||||
while (hipHostMalloc(&d_ptr, memFree) != hipSuccess && memFree > 1) {
|
||||
counter++;
|
||||
INFO("Attempt to allocate " << memFree << \
|
||||
" bytes out of " << devMemFree << "bytes Failed!");
|
||||
INFO("Attempt to allocate " << memFree << " bytes out of " << devMemFree << "bytes Failed!");
|
||||
memFree >>= 1; // reduce the memory to be allocated by half
|
||||
REQUIRE(counter <= 2); // Make sure that we are atleast able to allocate
|
||||
// 1/4th of max memory
|
||||
@@ -50,8 +49,7 @@ TEST_CASE("Stress_hipHostMalloc_MaxAllocation") {
|
||||
|
||||
HIP_CHECK(hipMemset(d_ptr, 1, memFree));
|
||||
HIP_CHECK(hipDeviceSynchronize()); // Flush caches
|
||||
REQUIRE(std::all_of(d_ptr, d_ptr + memFree,
|
||||
[](unsigned char n) { return n == 1; }));
|
||||
REQUIRE(std::all_of(d_ptr, d_ptr + memFree, [](unsigned char n) { return n == 1; }));
|
||||
HIP_CHECK(hipHostFree(d_ptr));
|
||||
}
|
||||
|
||||
@@ -67,8 +65,7 @@ TEST_CASE("Stress_hipHostMalloc_MaxAllocation_AllGpu") {
|
||||
// Get available GPU memory and total GPU memory
|
||||
HIP_CHECK(hipSetDevice(dev));
|
||||
HIP_CHECK(hipMemGetInfo(&availableMem, &maxGpuMem));
|
||||
size_t allocsize = maxGpuMem +
|
||||
((maxGpuMem*ADDITIONAL_MEMORY_PERCENT)/100);
|
||||
size_t allocsize = maxGpuMem + ((maxGpuMem * ADDITIONAL_MEMORY_PERCENT) / 100);
|
||||
// Get free host In bytes
|
||||
size_t hostMemFree = HipTest::getMemoryAmount() * 1024 * 1024;
|
||||
if (allocsize < hostMemFree) {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user