Checkpoint initial peer2peer implementation.

This commit is contained in:
Ben Sander
2016-04-06 11:28:23 -05:00
parent e180fa5b72
commit db91890f53
4 changed files with 111 additions and 17 deletions
+24 -3
View File
@@ -494,11 +494,23 @@ struct ihipEvent_t {
// will lock the mutex on construction and unlock on destruction.
//
// MUTEX_TYPE is template argument so can easily convert to FakeMutex for performance or stress testing.
template <typename MUTEX_TYPE>
template <class MUTEX_TYPE>
class ihipDeviceCriticalBase_t : LockedBase<MUTEX_TYPE>
{
public:
ihipDeviceCriticalBase_t() : _stream_id(0) {};
ihipDeviceCriticalBase_t() : _stream_id(0), _peerAgents(nullptr) {};
void init(unsigned deviceCnt) {
assert(_peerAgents == nullptr);
_peerAgents = new hsa_agent_t[deviceCnt];
};
~ihipDeviceCriticalBase_t() {
if (_peerAgents != nullptr) {
delete _peerAgents;
_peerAgents = nullptr;
}
}
friend class LockedAccessor<ihipDeviceCriticalBase_t>;
std::list<ihipStream_t*> &streams() { return _streams; };
@@ -507,10 +519,19 @@ public:
// "Allocate" a stream ID:
ihipStream_t::SeqNum_t incStreamId() { return _stream_id++; };
void recomputePeerAgents();
void addPeer(ihipDevice_t *peer);
void removePeer(ihipDevice_t *peer);
private:
std::list<ihipStream_t*> _streams; // streams associated with this device.
ihipStream_t::SeqNum_t _stream_id;
// These reflect the currently Enabled set of peers for this GPU:
std::list<ihipDevice_t*> _peers; // list of enabled peer devices.
uint32_t _peerCnt; // number of enabled peers
hsa_agent_t *_peerAgents; // efficient packed array of enabled agents (to use for allocations.)
};
// Note Mutex selected based on DeviceMutex
@@ -530,7 +551,7 @@ class ihipDevice_t
{
public: // Functions:
ihipDevice_t() {}; // note: calls constructor for _criticalData
void init(unsigned device_index, hc::accelerator &acc, unsigned flags);
void init(unsigned device_index, unsigned deviceCnt, hc::accelerator &acc, unsigned flags);
~ihipDevice_t();
void locked_addStream(ihipStream_t *s);
+5 -5
View File
@@ -908,12 +908,12 @@ hipError_t hipMemGetInfo (size_t * free, size_t * total) ;
* Returns "1" in @p canAccessPeer if the specified @p device is capable
* of directly accessing memory physically located on peerDevice , or "0" if not.
*/
hipError_t hipDeviceCanAccessPeer ( int* canAccessPeer, int device, int peerDevice );
hipError_t hipDeviceCanAccessPeer (int* canAccessPeer, int deviceId, int peerDeviceId);
/**
* @brief Disables registering memory on peerDevice for direct access from the current device.
* @brief Disable registering memory on peerDevice for direct access from the current device.
*
* If there are any allocations on peerDevice which were registered in the current device using hipPeerRegister() then these allocations will be automatically unregistered.
* Returns hipErrorPeerAccessNotEnabled if direct access to memory on peerDevice has not yet been enabled from the current device.
@@ -922,10 +922,10 @@ hipError_t hipDeviceCanAccessPeer ( int* canAccessPeer, int device, int peerDe
* TODO:cudaErrorPeerAccessNotEnabled and cudaErrorInvalidDevice error not supported in HIP, return hipErrorUnknown
* Returns #hipSuccess, #hipErrorUnknown
*/
hipError_t hipDeviceDisablePeerAccess ( int peerDevice );
hipError_t hipDeviceDisablePeerAccess (int peerDeviceId);
/**
* @brief Enables registering memory on peerDevice for direct access from the current device.
* @brief Enable registering memory on peerDevice for direct access from the current device.
*
* @param [in] peerDevice
* @param [in] flags
@@ -933,7 +933,7 @@ hipError_t hipDeviceDisablePeerAccess ( int peerDevice );
* TODO:cudaErrorInvalidDevice error not supported in HIP, return hipErrorUnknown
* Returns #hipSuccess, #hipErrorInvalidDevice, #hipErrorInvalidValue, #hipErrorUnknown
*/
hipError_t hipDeviceEnablePeerAccess ( int peerDevice, unsigned int flags );
hipError_t hipDeviceEnablePeerAccess (int peerDeviceId, unsigned int flags);
/**
* @brief Copies memory from one device to memory on another device.