Apply .clangformat to all repo source files

Change-Id: I7e79c6058f0303f9a98911e3b7dd2e8596079344
Этот коммит содержится в:
Maneesh Gupta
2018-03-12 11:29:03 +05:30
родитель 18e70b1e6b
Коммит 1ba06f63c4
293 изменённых файлов: 43980 добавлений и 45830 удалений
+181 -202
Просмотреть файл
@@ -35,23 +35,22 @@ THE SOFTWARE.
#include "test_common.h"
void printSep()
{
printf ("======================================================================================\n");
void printSep() {
printf(
"======================================================================================\n");
}
//-------
template<typename T>
class DeviceMemory
{
public:
template <typename T>
class DeviceMemory {
public:
DeviceMemory(size_t numElements);
~DeviceMemory();
T *A_d() const { return _A_d + _offset; };
T *B_d() const { return _B_d + _offset; };
T *C_d() const { return _C_d + _offset; };
T *C_dd() const { return _C_dd + _offset; };
T* A_d() const { return _A_d + _offset; };
T* B_d() const { return _B_d + _offset; };
T* C_d() const { return _C_d + _offset; };
T* C_dd() const { return _C_dd + _offset; };
size_t maxNumElements() const { return _maxNumElements; };
@@ -59,92 +58,83 @@ public:
void offset(int offset) { _offset = offset; };
int offset() const { return _offset; };
private:
T * _A_d;
T* _B_d;
T* _C_d;
T* _C_dd;
private:
T* _A_d;
T* _B_d;
T* _C_d;
T* _C_dd;
size_t _maxNumElements;
int _offset;
};
template<typename T>
DeviceMemory<T>::DeviceMemory(size_t numElements)
: _maxNumElements(numElements),
_offset(0)
{
T ** np = nullptr;
HipTest::initArrays (&_A_d, &_B_d, &_C_d, np, np, np, numElements, 0);
template <typename T>
DeviceMemory<T>::DeviceMemory(size_t numElements) : _maxNumElements(numElements), _offset(0) {
T** np = nullptr;
HipTest::initArrays(&_A_d, &_B_d, &_C_d, np, np, np, numElements, 0);
size_t sizeElements = numElements * sizeof(T);
HIPCHECK ( hipMalloc(&_C_dd, sizeElements) );
HIPCHECK(hipMalloc(&_C_dd, sizeElements));
}
template<typename T>
DeviceMemory<T>::~DeviceMemory ()
{
T * np = nullptr;
HipTest::freeArrays (_A_d, _B_d, _C_d, np, np, np, 0);
template <typename T>
DeviceMemory<T>::~DeviceMemory() {
T* np = nullptr;
HipTest::freeArrays(_A_d, _B_d, _C_d, np, np, np, 0);
HIPCHECK (hipFree(_C_dd));
HIPCHECK(hipFree(_C_dd));
_C_dd = NULL;
};
//-------
template<typename T>
class HostMemory
{
public:
template <typename T>
class HostMemory {
public:
HostMemory(size_t numElements, bool usePinnedHost);
void reset(size_t numElements, bool full=false) ;
void reset(size_t numElements, bool full = false);
~HostMemory();
T *A_h() const { return _A_h + _offset; };
T *B_h() const { return _B_h + _offset; };
T *C_h() const { return _C_h + _offset; };
T* A_h() const { return _A_h + _offset; };
T* B_h() const { return _B_h + _offset; };
T* C_h() const { return _C_h + _offset; };
size_t maxNumElements() const { return _maxNumElements; };
void offset(int offset) { _offset = offset; };
int offset() const { return _offset; };
public:
public:
// Host arrays, secondary copy
T * A_hh;
T* B_hh;
T* A_hh;
T* B_hh;
bool _usePinnedHost;
private:
bool _usePinnedHost;
private:
size_t _maxNumElements;
int _offset;
// Host arrays
T * _A_h;
T* _B_h;
T* _C_h;
T* _A_h;
T* _B_h;
T* _C_h;
};
template<typename T>
template <typename T>
HostMemory<T>::HostMemory(size_t numElements, bool usePinnedHost)
: _maxNumElements(numElements),
_usePinnedHost(usePinnedHost),
_offset(0)
{
T ** np = nullptr;
HipTest::initArrays (np, np, np, &_A_h, &_B_h, &_C_h, numElements, usePinnedHost);
: _maxNumElements(numElements), _usePinnedHost(usePinnedHost), _offset(0) {
T** np = nullptr;
HipTest::initArrays(np, np, np, &_A_h, &_B_h, &_C_h, numElements, usePinnedHost);
A_hh = NULL;
B_hh = NULL;
@@ -153,142 +143,137 @@ HostMemory<T>::HostMemory(size_t numElements, bool usePinnedHost)
size_t sizeElements = numElements * sizeof(T);
if (usePinnedHost) {
HIPCHECK ( hipHostMalloc((void**)&A_hh, sizeElements, hipHostMallocDefault) );
HIPCHECK ( hipHostMalloc((void**)&B_hh, sizeElements, hipHostMallocDefault) );
HIPCHECK(hipHostMalloc((void**)&A_hh, sizeElements, hipHostMallocDefault));
HIPCHECK(hipHostMalloc((void**)&B_hh, sizeElements, hipHostMallocDefault));
} else {
A_hh = (T*)malloc(sizeElements);
B_hh = (T*)malloc(sizeElements);
}
}
template<typename T>
void
HostMemory<T>::reset(size_t numElements, bool full)
{
template <typename T>
void HostMemory<T>::reset(size_t numElements, bool full) {
// Initialize the host data:
for (size_t i=0; i<numElements; i++) {
for (size_t i = 0; i < numElements; i++) {
(A_hh)[i] = 1097.0 + i;
(B_hh)[i] = 1492.0 + i; // Phi
(B_hh)[i] = 1492.0 + i; // Phi
if (full) {
(_A_h)[i] = 3.146f + i; // Pi
(_B_h)[i] = 1.618f + i; // Phi
(_A_h)[i] = 3.146f + i; // Pi
(_B_h)[i] = 1.618f + i; // Phi
}
}
}
template<typename T>
HostMemory<T>::~HostMemory ()
{
HipTest::freeArraysForHost (_A_h, _B_h, _C_h, _usePinnedHost);
template <typename T>
HostMemory<T>::~HostMemory() {
HipTest::freeArraysForHost(_A_h, _B_h, _C_h, _usePinnedHost);
if (_usePinnedHost) {
HIPCHECK (hipHostFree(A_hh));
HIPCHECK (hipHostFree(B_hh));
HIPCHECK(hipHostFree(A_hh));
HIPCHECK(hipHostFree(B_hh));
} else {
free(A_hh);
free(B_hh);
}
};
//---
// Test many different kinds of memory copies.
// The subroutine allocates memory , copies to device, runs a vector add kernel, copies back, and checks the result.
// The subroutine allocates memory , copies to device, runs a vector add kernel, copies back, and
// checks the result.
//
// IN: numElements controls the number of elements used for allocations.
// IN: usePinnedHost : If true, allocate host with hipHostMalloc and is pinned ; else allocate host memory with malloc.
// IN: useHostToHost : If true, add an extra host-to-host copy.
// IN: useDeviceToDevice : If true, add an extra deviceto-device copy after result is produced.
// IN: useMemkindDefault : If true, use memkinddefault (runtime figures out direction). if false, use explicit memcpy direction.
// IN: usePinnedHost : If true, allocate host with hipHostMalloc and is pinned ; else allocate host
// memory with malloc. IN: useHostToHost : If true, add an extra host-to-host copy. IN:
// useDeviceToDevice : If true, add an extra deviceto-device copy after result is produced. IN:
// useMemkindDefault : If true, use memkinddefault (runtime figures out direction). if false, use
// explicit memcpy direction.
//
template <typename T>
void memcpytest2(DeviceMemory<T> *dmem, HostMemory<T> *hmem, size_t numElements, bool useHostToHost, bool useDeviceToDevice, bool useMemkindDefault)
{
void memcpytest2(DeviceMemory<T>* dmem, HostMemory<T>* hmem, size_t numElements, bool useHostToHost,
bool useDeviceToDevice, bool useMemkindDefault) {
size_t sizeElements = numElements * sizeof(T);
printf ("test: %s<%s> size=%lu (%6.2fMB) usePinnedHost:%d, useHostToHost:%d, useDeviceToDevice:%d, useMemkindDefault:%d, offsets:dev:%+d host:+%d\n",
__func__,
TYPENAME(T),
sizeElements, sizeElements/1024.0/1024.0,
hmem->_usePinnedHost, useHostToHost, useDeviceToDevice, useMemkindDefault,
dmem->offset(), hmem->offset()
);
printf(
"test: %s<%s> size=%lu (%6.2fMB) usePinnedHost:%d, useHostToHost:%d, useDeviceToDevice:%d, "
"useMemkindDefault:%d, offsets:dev:%+d host:+%d\n",
__func__, TYPENAME(T), sizeElements, sizeElements / 1024.0 / 1024.0, hmem->_usePinnedHost,
useHostToHost, useDeviceToDevice, useMemkindDefault, dmem->offset(), hmem->offset());
hmem->reset(numElements);
unsigned blocks = HipTest::setNumBlocks(blocksPerCU, threadsPerBlock, numElements);
assert (numElements <= dmem->maxNumElements());
assert (numElements <= hmem->maxNumElements());
assert(numElements <= dmem->maxNumElements());
assert(numElements <= hmem->maxNumElements());
if (useHostToHost) {
// Do some extra host-to-host copies here to mix things up:
HIPCHECK ( hipMemcpy(hmem->A_hh, hmem->A_h(), sizeElements, useMemkindDefault? hipMemcpyDefault : hipMemcpyHostToHost));
HIPCHECK ( hipMemcpy(hmem->B_hh, hmem->B_h(), sizeElements, useMemkindDefault? hipMemcpyDefault : hipMemcpyHostToHost));
HIPCHECK(hipMemcpy(hmem->A_hh, hmem->A_h(), sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToHost));
HIPCHECK(hipMemcpy(hmem->B_hh, hmem->B_h(), sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToHost));
HIPCHECK ( hipMemcpy(dmem->A_d(), hmem->A_hh, sizeElements, useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToDevice));
HIPCHECK ( hipMemcpy(dmem->B_d(), hmem->B_hh, sizeElements, useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToDevice));
HIPCHECK(hipMemcpy(dmem->A_d(), hmem->A_hh, sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToDevice));
HIPCHECK(hipMemcpy(dmem->B_d(), hmem->B_hh, sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToDevice));
} else {
HIPCHECK ( hipMemcpy(dmem->A_d(), hmem->A_h(), sizeElements, useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToDevice));
HIPCHECK ( hipMemcpy(dmem->B_d(), hmem->B_h(), sizeElements, useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToDevice));
HIPCHECK(hipMemcpy(dmem->A_d(), hmem->A_h(), sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToDevice));
HIPCHECK(hipMemcpy(dmem->B_d(), hmem->B_h(), sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyHostToDevice));
}
hipLaunchKernel(
HipTest::vectorADD,
dim3(blocks),
dim3(threadsPerBlock),
0,
0,
static_cast<const T*>(dmem->A_d()),
static_cast<const T*>(dmem->B_d()),
dmem->C_d(),
numElements);
hipLaunchKernel(HipTest::vectorADD, dim3(blocks), dim3(threadsPerBlock), 0, 0,
static_cast<const T*>(dmem->A_d()), static_cast<const T*>(dmem->B_d()),
dmem->C_d(), numElements);
if (useDeviceToDevice) {
// Do an extra device-to-device copy here to mix things up:
HIPCHECK ( hipMemcpy(dmem->C_dd(), dmem->C_d(), sizeElements, useMemkindDefault? hipMemcpyDefault : hipMemcpyDeviceToDevice));
HIPCHECK(hipMemcpy(dmem->C_dd(), dmem->C_d(), sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyDeviceToDevice));
//Destroy the original dmem->C_d():
HIPCHECK ( hipMemset(dmem->C_d(), 0x5A, sizeElements));
// Destroy the original dmem->C_d():
HIPCHECK(hipMemset(dmem->C_d(), 0x5A, sizeElements));
HIPCHECK ( hipMemcpy(hmem->C_h(), dmem->C_dd(), sizeElements, useMemkindDefault? hipMemcpyDefault:hipMemcpyDeviceToHost));
HIPCHECK(hipMemcpy(hmem->C_h(), dmem->C_dd(), sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyDeviceToHost));
} else {
HIPCHECK ( hipMemcpy(hmem->C_h(), dmem->C_d(), sizeElements, useMemkindDefault? hipMemcpyDefault:hipMemcpyDeviceToHost));
HIPCHECK(hipMemcpy(hmem->C_h(), dmem->C_d(), sizeElements,
useMemkindDefault ? hipMemcpyDefault : hipMemcpyDeviceToHost));
}
HIPCHECK ( hipDeviceSynchronize() );
HIPCHECK(hipDeviceSynchronize());
HipTest::checkVectorADD(hmem->A_h(), hmem->B_h(), hmem->C_h(), numElements);
printf (" %s success\n", __func__);
printf(" %s success\n", __func__);
}
//---
//Try all the 16 possible combinations to memcpytest2 - usePinnedHost, useHostToHost, useDeviceToDevice, useMemkindDefault
template<typename T>
void memcpytest2_for_type(size_t numElements)
{
// Try all the 16 possible combinations to memcpytest2 - usePinnedHost, useHostToHost,
// useDeviceToDevice, useMemkindDefault
template <typename T>
void memcpytest2_for_type(size_t numElements) {
printSep();
DeviceMemory<T> memD(numElements);
HostMemory<T> memU(numElements, 0/*usePinnedHost*/);
HostMemory<T> memP(numElements, 1/*usePinnedHost*/);
HostMemory<T> memU(numElements, 0 /*usePinnedHost*/);
HostMemory<T> memP(numElements, 1 /*usePinnedHost*/);
for (int usePinnedHost =0; usePinnedHost<=1; usePinnedHost++) {
for (int useHostToHost =0; useHostToHost<=1; useHostToHost++) { // TODO
for (int useDeviceToDevice =0; useDeviceToDevice<=1; useDeviceToDevice++) {
for (int useMemkindDefault =0; useMemkindDefault<=1; useMemkindDefault++) {
memcpytest2<T>(&memD, usePinnedHost ? &memP : &memU, numElements, useHostToHost, useDeviceToDevice, useMemkindDefault);
for (int usePinnedHost = 0; usePinnedHost <= 1; usePinnedHost++) {
for (int useHostToHost = 0; useHostToHost <= 1; useHostToHost++) { // TODO
for (int useDeviceToDevice = 0; useDeviceToDevice <= 1; useDeviceToDevice++) {
for (int useMemkindDefault = 0; useMemkindDefault <= 1; useMemkindDefault++) {
memcpytest2<T>(&memD, usePinnedHost ? &memP : &memU, numElements, useHostToHost,
useDeviceToDevice, useMemkindDefault);
}
}
}
@@ -297,12 +282,11 @@ void memcpytest2_for_type(size_t numElements)
//---
//Try many different sizes to memory copy.
template<typename T>
void memcpytest2_sizes(size_t maxElem=0)
{
// Try many different sizes to memory copy.
template <typename T>
void memcpytest2_sizes(size_t maxElem = 0) {
printSep();
printf ("test: %s<%s>\n", __func__, TYPENAME(T));
printf("test: %s<%s>\n", __func__, TYPENAME(T));
int deviceId;
HIPCHECK(hipGetDevice(&deviceId));
@@ -311,17 +295,19 @@ void memcpytest2_sizes(size_t maxElem=0)
HIPCHECK(hipMemGetInfo(&free, &total));
if (maxElem == 0) {
maxElem = free/sizeof(T)/8;
maxElem = free / sizeof(T) / 8;
}
printf (" device#%d: hipMemGetInfo: free=%zu (%4.2fMB) total=%zu (%4.2fMB) maxSize=%6.1fMB\n",
deviceId, free, (float)(free/1024.0/1024.0), total, (float)(total/1024.0/1024.0), maxElem*sizeof(T)/1024.0/1024.0);
HIPCHECK ( hipDeviceReset() );
printf(
" device#%d: hipMemGetInfo: free=%zu (%4.2fMB) total=%zu (%4.2fMB) maxSize=%6.1fMB\n",
deviceId, free, (float)(free / 1024.0 / 1024.0), total, (float)(total / 1024.0 / 1024.0),
maxElem * sizeof(T) / 1024.0 / 1024.0);
HIPCHECK(hipDeviceReset());
DeviceMemory<T> memD(maxElem);
HostMemory<T> memU(maxElem, 0/*usePinnedHost*/);
HostMemory<T> memP(maxElem, 1/*usePinnedHost*/);
HostMemory<T> memU(maxElem, 0 /*usePinnedHost*/);
HostMemory<T> memP(maxElem, 1 /*usePinnedHost*/);
for (size_t elem=1; elem<=maxElem; elem*=2) {
for (size_t elem = 1; elem <= maxElem; elem *= 2) {
memcpytest2<T>(&memD, &memU, elem, 1, 1, 0); // unpinned host
memcpytest2<T>(&memD, &memP, elem, 1, 1, 0); // pinned host
}
@@ -329,12 +315,11 @@ void memcpytest2_sizes(size_t maxElem=0)
//---
//Try many different sizes to memory copy.
template<typename T>
void memcpytest2_offsets(size_t maxElem, bool devOffsets, bool hostOffsets)
{
// Try many different sizes to memory copy.
template <typename T>
void memcpytest2_offsets(size_t maxElem, bool devOffsets, bool hostOffsets) {
printSep();
printf ("test: %s<%s>\n", __func__, TYPENAME(T));
printf("test: %s<%s>\n", __func__, TYPENAME(T));
int deviceId;
HIPCHECK(hipGetDevice(&deviceId));
@@ -343,17 +328,19 @@ void memcpytest2_offsets(size_t maxElem, bool devOffsets, bool hostOffsets)
HIPCHECK(hipMemGetInfo(&free, &total));
printf (" device#%d: hipMemGetInfo: free=%zu (%4.2fMB) total=%zu (%4.2fMB) maxSize=%6.1fMB\n",
deviceId, free, (float)(free/1024.0/1024.0), total, (float)(total/1024.0/1024.0), maxElem*sizeof(T)/1024.0/1024.0);
HIPCHECK ( hipDeviceReset() );
printf(
" device#%d: hipMemGetInfo: free=%zu (%4.2fMB) total=%zu (%4.2fMB) maxSize=%6.1fMB\n",
deviceId, free, (float)(free / 1024.0 / 1024.0), total, (float)(total / 1024.0 / 1024.0),
maxElem * sizeof(T) / 1024.0 / 1024.0);
HIPCHECK(hipDeviceReset());
DeviceMemory<T> memD(maxElem);
HostMemory<T> memU(maxElem, 0/*usePinnedHost*/);
HostMemory<T> memP(maxElem, 1/*usePinnedHost*/);
HostMemory<T> memU(maxElem, 0 /*usePinnedHost*/);
HostMemory<T> memP(maxElem, 1 /*usePinnedHost*/);
size_t elem = maxElem / 2;
for (int offset=0; offset < 512; offset++) {
assert (elem + offset < maxElem);
for (int offset = 0; offset < 512; offset++) {
assert(elem + offset < maxElem);
if (devOffsets) {
memD.offset(offset);
}
@@ -365,8 +352,8 @@ void memcpytest2_offsets(size_t maxElem, bool devOffsets, bool hostOffsets)
memcpytest2<T>(&memD, &memP, elem, 1, 1, 0); // pinned host
}
for (int offset=512; offset < elem; offset*=2) {
assert (elem + offset < maxElem);
for (int offset = 512; offset < elem; offset *= 2) {
assert(elem + offset < maxElem);
if (devOffsets) {
memD.offset(offset);
}
@@ -381,23 +368,24 @@ void memcpytest2_offsets(size_t maxElem, bool devOffsets, bool hostOffsets)
//---
//Create multiple threads to stress multi-thread locking behavior in the allocation/deallocation/tracking logic:
template<typename T>
void multiThread_1(bool serialize, bool usePinnedHost)
{
// Create multiple threads to stress multi-thread locking behavior in the
// allocation/deallocation/tracking logic:
template <typename T>
void multiThread_1(bool serialize, bool usePinnedHost) {
printSep();
printf ("test: %s<%s> serialize=%d usePinnedHost=%d\n", __func__, TYPENAME(T), serialize, usePinnedHost);
printf("test: %s<%s> serialize=%d usePinnedHost=%d\n", __func__, TYPENAME(T), serialize,
usePinnedHost);
DeviceMemory<T> memD(N);
HostMemory<T> mem1(N, usePinnedHost);
HostMemory<T> mem2(N, usePinnedHost);
std::thread t1 (memcpytest2<T>, &memD, &mem1, N, 0,0,0);
std::thread t1(memcpytest2<T>, &memD, &mem1, N, 0, 0, 0);
if (serialize) {
t1.join();
}
std::thread t2 (memcpytest2<T>,&memD, &mem2, N, 0,0,0);
std::thread t2(memcpytest2<T>, &memD, &mem2, N, 0, 0, 0);
if (serialize) {
t2.join();
}
@@ -409,64 +397,57 @@ void multiThread_1(bool serialize, bool usePinnedHost)
}
int main(int argc, char *argv[])
{
int main(int argc, char* argv[]) {
HipTest::parseStandardArguments(argc, argv, true);
printf ("info: set device to %d\n", p_gpuDevice);
printf("info: set device to %d\n", p_gpuDevice);
HIPCHECK(hipSetDevice(p_gpuDevice));
if (p_tests & 0x1) {
printf ("\n\n=== tests&1 (types and different memcpy kinds (H2D, D2H, H2H, D2D)\n");
HIPCHECK ( hipDeviceReset() );
printf("\n\n=== tests&1 (types and different memcpy kinds (H2D, D2H, H2H, D2D)\n");
HIPCHECK(hipDeviceReset());
memcpytest2_for_type<float>(N);
memcpytest2_for_type<double>(N);
memcpytest2_for_type<char>(N);
memcpytest2_for_type<int>(N);
printf ("===\n\n\n");
printf("===\n\n\n");
}
if (p_tests & 0x2) {
// Some tests around the 64KB boundary which have historically shown issues:
printf ("\n\n=== tests&0x2 (64KB boundary)\n");
size_t maxElem = 32*1024*1024;
printf("\n\n=== tests&0x2 (64KB boundary)\n");
size_t maxElem = 32 * 1024 * 1024;
DeviceMemory<float> memD(maxElem);
HostMemory<float> memU(maxElem, 0/*usePinnedHost*/);
HostMemory<float> memP(maxElem, 0/*usePinnedHost*/);
HostMemory<float> memU(maxElem, 0 /*usePinnedHost*/);
HostMemory<float> memP(maxElem, 0 /*usePinnedHost*/);
// These all pass:
memcpytest2<float>(&memD, &memP, 15*1024*1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 16*1024*1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 16*1024*1024+16*1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 15 * 1024 * 1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 16 * 1024 * 1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 16 * 1024 * 1024 + 16 * 1024, 0, 0, 0);
// Just over 64MB:
memcpytest2<float>(&memD, &memP, 16*1024*1024+512*1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 17*1024*1024+1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 32*1024*1024, 0, 0, 0);
memcpytest2<float>(&memD, &memU, 32*1024*1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 32*1024*1024, 1, 1, 0);
memcpytest2<float>(&memD, &memP, 32*1024*1024, 1, 1, 0);
memcpytest2<float>(&memD, &memP, 16 * 1024 * 1024 + 512 * 1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 17 * 1024 * 1024 + 1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 32 * 1024 * 1024, 0, 0, 0);
memcpytest2<float>(&memD, &memU, 32 * 1024 * 1024, 0, 0, 0);
memcpytest2<float>(&memD, &memP, 32 * 1024 * 1024, 1, 1, 0);
memcpytest2<float>(&memD, &memP, 32 * 1024 * 1024, 1, 1, 0);
}
if (p_tests & 0x4) {
printf ("\n\n=== tests&4 (test sizes)\n");
HIPCHECK ( hipDeviceReset() );
printf("\n\n=== tests&4 (test sizes)\n");
HIPCHECK(hipDeviceReset());
memcpytest2_sizes<float>(0);
printSep();
}
if (p_tests & 0x8) {
printf ("\n\n=== tests&8\n");
HIPCHECK ( hipDeviceReset() );
printf("\n\n=== tests&8\n");
HIPCHECK(hipDeviceReset());
printSep();
// Simplest cases: serialize the threads, and also used pinned memory:
@@ -480,32 +461,30 @@ int main(int argc, char *argv[])
multiThread_1<float>(false, true);
// Remove serialization, and use unpinned.
multiThread_1<float>(false, false); // TODO
printf ("===\n\n\n");
multiThread_1<float>(false, false); // TODO
printf("===\n\n\n");
}
if (p_tests & 0x10) {
printf ("\n\n=== tests&0x10 (test device offsets)\n");
HIPCHECK ( hipDeviceReset() );
size_t maxSize = 256*1024;
memcpytest2_offsets<char> (maxSize, true, false);
memcpytest2_offsets<float> (maxSize, true, false);
printf("\n\n=== tests&0x10 (test device offsets)\n");
HIPCHECK(hipDeviceReset());
size_t maxSize = 256 * 1024;
memcpytest2_offsets<char>(maxSize, true, false);
memcpytest2_offsets<float>(maxSize, true, false);
memcpytest2_offsets<double>(maxSize, true, false);
}
if (p_tests & 0x20) {
printf ("\n\n=== tests&0x10 (test device offsets)\n");
HIPCHECK ( hipDeviceReset() );
size_t maxSize = 256*1024;
memcpytest2_offsets<char> (maxSize, false, true);
memcpytest2_offsets<float> (maxSize, false, true);
printf("\n\n=== tests&0x10 (test device offsets)\n");
HIPCHECK(hipDeviceReset());
size_t maxSize = 256 * 1024;
memcpytest2_offsets<char>(maxSize, false, true);
memcpytest2_offsets<float>(maxSize, false, true);
memcpytest2_offsets<double>(maxSize, false, true);
}
passed();
}