Files
rocm-systems/src/group.cc
T

439 строки
18 KiB
C++
Исходник Обычный вид История

2018-09-24 16:06:59 -07:00
/*************************************************************************
2022-01-07 06:39:55 -08:00
* Copyright (c) 2015-2022, NVIDIA CORPORATION. All rights reserved.
* Modifications Copyright (c) 2019-2022 Advanced Micro Devices, Inc. All rights reserved.
2018-09-24 16:06:59 -07:00
*
* See LICENSE.txt for license information
************************************************************************/
#include "group.h"
#include "debug.h"
#include "enqueue.h"
2020-05-12 14:40:18 -07:00
#include "transport.h"
2022-05-03 01:30:26 -07:00
#include "channel.h"
2018-09-24 16:06:59 -07:00
#define MAX_ASYNC_OPS 128
thread_local pthread_t ncclGroupThreads[MAX_ASYNC_OPS];
thread_local int ncclGroupIndex = 0;
thread_local int ncclGroupMode = 0;
thread_local ncclResult_t ncclGroupError = ncclSuccess;
2021-03-06 20:32:30 -08:00
extern struct allocationTracker allocTracker[];
2018-09-24 16:06:59 -07:00
bool ncclAsyncMode() {
return ncclGroupMode > 0;
}
ncclResult_t ncclAsyncErrCheck(ncclResult_t ret) {
if (ncclGroupError == ncclSuccess || ret != ncclSuccess) ncclGroupError = ret;
return ret;
}
struct ncclInitArgs {
ncclInitFunc_t func;
int cudaDev;
ncclComm_t* newcomm;
int ndev;
ncclUniqueId commId;
int myrank;
};
struct ncclCollArgs {
ncclComm_t comm;
uint16_t connIndex;
2018-09-24 16:06:59 -07:00
};
enum ncclAsyncFuncType {
ASYNC_FUNC_INVALID = 0,
ASYNC_FUNC_INIT = 1,
ASYNC_FUNC_COLL = 2,
};
struct ncclAsyncArgs {
ncclResult_t ret;
enum ncclAsyncFuncType funcType;
union {
ncclCollArgs coll;
ncclInitArgs init;
};
};
thread_local struct ncclAsyncArgs ncclGroupArgs[MAX_ASYNC_OPS];
void* ncclAsyncThreadMain(void* args_) {
struct ncclAsyncArgs* args = (struct ncclAsyncArgs*)args_;
2020-05-12 14:40:18 -07:00
NCCLCHECKTHREAD(args->init.func(args->init.newcomm, args->init.ndev, args->init.commId, args->init.myrank, args->init.cudaDev));
2018-09-24 16:06:59 -07:00
return args;
}
2019-11-19 14:57:39 -08:00
ncclResult_t ncclAsyncInit(ncclInitFunc_t func, ncclComm_t* newcomm, int ndev, ncclUniqueId commId, int myrank, int cudaDev) {
2018-09-24 16:06:59 -07:00
if (ncclGroupIndex >= MAX_ASYNC_OPS) {
WARN("Too many async operations in progress, max is %d", MAX_ASYNC_OPS);
2019-11-19 14:57:39 -08:00
return ncclAsyncErrCheck(ncclInvalidUsage);
2018-09-24 16:06:59 -07:00
}
int index = ncclGroupIndex++;
struct ncclAsyncArgs* args = ncclGroupArgs+index;
args->funcType = ASYNC_FUNC_INIT;
args->init.func = func;
args->init.cudaDev = cudaDev;
args->init.newcomm = newcomm;
args->init.ndev = ndev;
memcpy(&args->init.commId, &commId, sizeof(commId));
args->init.myrank = myrank;
return ncclSuccess;
}
ncclResult_t ncclAsyncColl(ncclComm_t comm) {
struct ncclAsyncArgs* args = ncclGroupArgs;
for (int i=0; i<ncclGroupIndex; i++) {
if (args->coll.comm == comm) return ncclSuccess;
args++;
}
if (ncclGroupIndex >= MAX_ASYNC_OPS) {
WARN("Too many async operations in progress, max is %d", MAX_ASYNC_OPS);
2019-11-19 14:57:39 -08:00
return ncclAsyncErrCheck(ncclInvalidUsage);
2018-09-24 16:06:59 -07:00
}
ncclGroupIndex++;
args->funcType = ASYNC_FUNC_COLL;
args->coll.comm = comm;
return ncclSuccess;
}
NCCL_API(ncclResult_t, ncclGroupStart);
ncclResult_t ncclGroupStart() {
2020-09-04 14:35:05 -07:00
NVTX3_FUNC_RANGE_IN(nccl_domain);
2020-05-12 14:40:18 -07:00
if (ncclGroupMode == 0) {
memset(ncclGroupArgs, 0, sizeof(struct ncclAsyncArgs)*MAX_ASYNC_OPS);
}
2018-09-24 16:06:59 -07:00
ncclGroupMode++;
return ncclSuccess;
}
static ncclResult_t scheduleSend(struct ncclComm* comm, int peer, int chunk, size_t count, void* buff, uint64_t opCount, uint16_t connIndex) {
2022-01-07 06:39:55 -08:00
struct ncclInfo info = { ncclFuncSend, "Send",
NULL, buff, count, ncclInt8, ncclSum, peer, comm, comm->userStream, /* Args */
1, 1 };
2022-05-03 01:30:26 -07:00
int channelId;
NCCLCHECK(ncclChannelCompute(comm, peer, chunk%comm->p2pnChannelsPerPeer, ncclFuncSend, &channelId));
2022-01-07 06:39:55 -08:00
info.channelId = channelId;
info.opCount = opCount;
info.connIndex = connIndex;
2022-01-07 06:39:55 -08:00
NCCLCHECK(ncclSetupP2pKernel(&info));
return ncclSuccess;
}
static ncclResult_t scheduleRecv(struct ncclComm* comm, int peer, int chunk, size_t count, void* buff, uint64_t opCount, uint16_t connIndex) {
2022-01-07 06:39:55 -08:00
struct ncclInfo info = { ncclFuncRecv, "Recv",
NULL, buff, count, ncclInt8, ncclSum, peer, comm, comm->userStream, /* Args */
2020-05-12 14:40:18 -07:00
1, 1 };
2022-05-03 01:30:26 -07:00
int channelId;
NCCLCHECK(ncclChannelCompute(comm, peer, chunk%comm->p2pnChannelsPerPeer, ncclFuncRecv, &channelId));
2020-05-12 14:40:18 -07:00
info.channelId = channelId;
info.opCount = opCount;
info.connIndex = connIndex;
2021-04-12 16:00:11 -07:00
NCCLCHECK(ncclSetupP2pKernel(&info));
2020-05-12 14:40:18 -07:00
return ncclSuccess;
}
void* ncclAsyncThreadPreconnect(void* args_) {
struct ncclAsyncArgs* args = (struct ncclAsyncArgs*)args_;
2020-09-04 14:35:05 -07:00
struct ncclComm* comm = args->coll.comm;
CUDACHECKTHREAD(hipSetDevice(comm->cudaDev));
2021-07-08 14:12:04 -07:00
if (CPU_COUNT(&comm->cpuAffinity)) sched_setaffinity(0, sizeof(cpu_set_t), &comm->cpuAffinity);
NCCLCHECKTHREAD(ncclTransportP2pSetup(comm, NULL, args->coll.connIndex));
2020-05-12 14:40:18 -07:00
return args;
}
2020-09-04 14:35:05 -07:00
static size_t getP2pChunkSize(size_t totalSize, int minChannels, int maxChannels, size_t minSize, size_t maxSize) {
size_t size = std::max(minSize, DIVUP(totalSize, minChannels));
int nChannels = minChannels;
while (size > maxSize && nChannels <= maxChannels/2) {
nChannels *= 2;
size = DIVUP(totalSize, nChannels);
}
ALIGN_SIZE(size, minSize);
return size;
}
2022-02-12 10:30:16 -08:00
RCCL_PARAM(P2pNetThreshold, "P2P_NET_THRESHOLD", 131072);
2018-09-24 16:06:59 -07:00
NCCL_API(ncclResult_t, ncclGroupEnd);
ncclResult_t ncclGroupEnd() {
2020-09-04 14:35:05 -07:00
NVTX3_FUNC_RANGE_IN(nccl_domain);
2020-06-22 09:36:20 -07:00
if (ncclGroupMode == 0) {
WARN("ncclGroupEnd: not in a group call.");
return ncclInvalidUsage;
}
2018-09-24 16:06:59 -07:00
ncclGroupMode--;
if (ncclGroupMode > 0) return ncclSuccess;
int savedDev;
2019-07-05 15:43:00 -07:00
CUDACHECK(hipGetDevice(&savedDev));
2020-05-12 14:40:18 -07:00
int activeThreads = 0;
2019-03-14 19:39:20 -07:00
int doneArray[MAX_ASYNC_OPS];
2020-05-12 14:40:18 -07:00
for (int i=0; i<ncclGroupIndex; i++) doneArray[i] = 1;
2018-09-24 16:06:59 -07:00
ncclResult_t ret = ncclGroupError;
2021-04-12 16:00:11 -07:00
int usingCudaGraphAll = -1;
2022-01-10 08:26:01 -08:00
hipGraph_t* graphs = NULL;
2018-09-24 16:06:59 -07:00
if (ret != ncclSuccess) goto group_cleanup;
2019-11-19 14:57:39 -08:00
/* Launch async ncclCommInitRank */
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_INIT) {
pthread_create(ncclGroupThreads+i, NULL, ncclAsyncThreadMain, args);
2020-05-12 14:40:18 -07:00
activeThreads++;
doneArray[i] = 0;
}
}
/* For init, since we use threads, we just wait for threads to complete */
while (activeThreads) {
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_INIT && doneArray[i] == 0) {
int err = pthread_tryjoin_np(ncclGroupThreads[i], NULL);
if (err == EBUSY) continue;
if (err != 0) ret = ncclSystemError;
if (args->ret != ncclSuccess) ret = args->ret;
doneArray[i] = 1;
activeThreads--;
}
}
}
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL && args->coll.comm->connect[1]) {
args->coll.connIndex = 1;
2020-09-04 14:35:05 -07:00
pthread_create(ncclGroupThreads+i, NULL, ncclAsyncThreadPreconnect, args);
2020-05-12 14:40:18 -07:00
}
}
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL && args->coll.comm->connect[1]) {
2020-05-12 14:40:18 -07:00
int err = pthread_join(ncclGroupThreads[i], NULL);
if (err != 0) {
2021-02-09 15:34:08 -08:00
WARN("Error waiting for pthread_join : %s", strerror(errno));
2020-05-12 14:40:18 -07:00
return ncclSystemError;
}
2021-03-06 20:32:30 -08:00
INFO(NCCL_INIT, "comm %p rank %d total %ld bytes - P2P preconnect COMPLETE", args->coll.comm, args->coll.comm->rank, allocTracker[args->coll.comm->cudaDev].totalAllocSize);
2020-05-12 14:40:18 -07:00
NCCLCHECKGOTO(args->ret, ret, end);
args->coll.comm->connect[1] = 0;
}
}
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL && args->coll.comm->connect[NCCL_CONN_IDX_P2P_NET]) {
args->coll.connIndex = NCCL_CONN_IDX_P2P_NET;
pthread_create(ncclGroupThreads+i, NULL, ncclAsyncThreadPreconnect, args);
}
}
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL && args->coll.comm->connect[NCCL_CONN_IDX_P2P_NET]) {
int err = pthread_join(ncclGroupThreads[i], NULL);
if (err != 0) {
WARN("Error waiting for pthread_join : %s", strerror(errno));
return ncclSystemError;
}
INFO(NCCL_INIT, "comm %p rank %d total %ld bytes - P2P NET preconnect COMPLETE", args->coll.comm, args->coll.comm->rank, allocTracker[args->coll.comm->cudaDev].totalAllocSize);
NCCLCHECKGOTO(args->ret, ret, end);
args->coll.comm->connect[NCCL_CONN_IDX_P2P_NET] = 0;
2020-05-12 14:40:18 -07:00
}
}
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL) {
struct ncclComm* comm = args->coll.comm;
int rank = comm->rank;
int nRanks = comm->nRanks;
2020-09-04 14:35:05 -07:00
// Compute how much to split operations
// Natural step size matching buffer steps.
ssize_t stepSize = comm->buffSizes[NCCL_PROTO_SIMPLE] / NCCL_STEPS;
// Try to use all channels
int nChannelsMax = comm->p2pnChannelsPerPeer;
int nChannelsMin = nChannelsMax;
// Try to use all channels, but one channel per operation.
2020-12-04 18:52:32 -05:00
while (nChannelsMin*comm->nRanks > std::max(comm->nChannels, comm->p2pnChannels) && nChannelsMin > 1) nChannelsMin /= 2;
2020-09-04 14:35:05 -07:00
// Avoid overloading channels with 8+ operations as we loose the sync warp, hence a bit of bandwidth.
2020-12-04 18:52:32 -05:00
while (nChannelsMax*comm->nRanks > std::max(comm->nChannels, comm->p2pnChannels)*4 && nChannelsMax > 1) nChannelsMax /= 2;
2020-09-04 14:35:05 -07:00
while (comm->p2pSendCount > 0 || comm->p2pRecvCount > 0) {
// schedule delta 0, +1, -1, +2, -2, ...
// also make sure we don't do 0 twice, nor +n/2 and -n/2 if n is even.
for (int d=0; d<=nRanks/4; d++) {
2021-02-09 15:34:08 -08:00
int deltas[4] = { d, (nRanks-d)%nRanks, nRanks/2-d, (nRanks-(nRanks/2-d))%nRanks };
2020-09-04 14:35:05 -07:00
int index = 0;
int delta = deltas[index];
sched_delta:
uint32_t recvPeer = (rank+nRanks-delta)%nRanks;
uint32_t sendPeer = (rank+delta)%nRanks;
struct ncclP2Pinfo* recv = comm->p2pRecvs[recvPeer] ? comm->p2pRecvs[recvPeer]->getNext() : NULL;
struct ncclP2Pinfo* send = comm->p2pSends[sendPeer] ? comm->p2pSends[sendPeer]->getNext() : NULL;
2020-09-04 14:35:05 -07:00
if (recv != NULL || send != NULL) {
ssize_t totRecvBytes = -1, totSendBytes = -1;
if (recv != NULL) totRecvBytes = recv->nbytes;
if (send != NULL) totSendBytes = send->nbytes;
if (recv) comm->p2pRecvCount--;
if (send) comm->p2pSendCount--;
if (recvPeer == comm->rank) { // Check self send/recv
if (sendPeer != comm->rank) { WARN("Sendrecv schedule not aligned for self"); ret = ncclInternalError; goto group_cleanup; }
if (send && recv == NULL) { WARN("Trying to send to self without a matching recv"); ret = ncclInvalidUsage; goto group_cleanup; }
if (send == NULL && recv) { WARN("Trying to recv to self without a matching send"); ret = ncclInvalidUsage; goto group_cleanup; }
}
void* recvBuff = recv ? recv->buff : NULL;
void* sendBuff = send ? send->buff : NULL;
// After we recycle p2pSend/Recv, we're no longer allowed to dereference send or recv, only use them as boolean NULL/not NULL.
if (recv && comm->p2pRecvs[recvPeer]->peakNext() == NULL) comm->p2pRecvs[recvPeer]->recycle();
if (send && comm->p2pSends[sendPeer]->peakNext() == NULL) comm->p2pSends[sendPeer]->recycle();
2020-09-04 14:35:05 -07:00
ssize_t recvChunkSize = getP2pChunkSize(totRecvBytes, nChannelsMin, nChannelsMax, stepSize, SENDRECV_SLICEFACTOR*stepSize);
ssize_t sendChunkSize = getP2pChunkSize(totSendBytes, nChannelsMin, nChannelsMax, stepSize, SENDRECV_SLICEFACTOR*stepSize);
2020-05-12 14:40:18 -07:00
uint16_t sendIdx = 1, recvIdx = 1;
if(comm->p2pNet && totSendBytes > rcclParamP2pNetThreshold())
sendIdx = NCCL_CONN_IDX_P2P_NET;
if(comm->p2pNet && totRecvBytes > rcclParamP2pNetThreshold())
recvIdx = NCCL_CONN_IDX_P2P_NET;
2020-09-04 14:35:05 -07:00
ssize_t sendOffset = 0;
ssize_t recvOffset = 0;
int sendRemaining = 1, recvRemaining = 1;
int chunk = 0;
do {
ssize_t recvbytes = totRecvBytes-recvOffset;
ssize_t sendbytes = totSendBytes-sendOffset;
if (recvbytes > recvChunkSize) { recvbytes = recvChunkSize; } else { recvRemaining = 0; }
if (sendbytes > sendChunkSize) { sendbytes = sendChunkSize; } else { sendRemaining = 0; }
2021-02-09 15:34:08 -08:00
// 0-bytes send/recv are considered as syncs. Make sure we only add syncs when requested
// (total size == 0), otherwise set size to -1.
2022-03-30 02:25:49 -07:00
if (sendbytes < 0 || (sendbytes == 0 && totSendBytes != 0)) send = NULL;
if (recvbytes < 0 || (recvbytes == 0 && totRecvBytes != 0)) recv = NULL;
if (recv) {
NCCLCHECKGOTO(scheduleRecv(comm, recvPeer, chunk, recvbytes, ((char*)(recv->buff))+recvOffset, recv->opCount, recvIdx), ret, group_cleanup);
}
if (send) {
NCCLCHECKGOTO(scheduleSend(comm, sendPeer, chunk, sendbytes, ((char*)(send->buff))+sendOffset, send->opCount, sendIdx), ret, group_cleanup);
2020-09-04 14:35:05 -07:00
}
recvOffset += recvChunkSize;
sendOffset += sendChunkSize;
chunk++;
} while (sendRemaining || recvRemaining);
}
index++;
if (index == 1 && deltas[1] == deltas[0]) index++;
if (index == 2 && deltas[2] == deltas[0]) index++;
if (index == 3 && deltas[3] == deltas[2]) index++;
if (index == 3 && deltas[3] == deltas[1]) index++;
if (index < 4) {
delta = deltas[index];
goto sched_delta;
2020-05-12 14:40:18 -07:00
}
}
}
2019-11-19 14:57:39 -08:00
}
}
2018-09-24 16:06:59 -07:00
/* Collectives are done in three steps :
2020-09-04 14:35:05 -07:00
* 0. Save kernels previously enqueued. Compute channel, algo, proto, etc.
2018-09-24 16:06:59 -07:00
* 1. Barrier Check In. Only the last call may call cudaLaunchKernel[cooperative]
* 2. Barrier Wait. No CUDA call is permitted
* 3. Enqueue Events. CUDA event wait/enqueue.
* This is needed because step 2 cannot call any CUDA primitive, otherwise if
* cudaFree happens between 1 and 3, it could block that CUDA call and
* prevent some ranks from launching their network threads, which would
* prevent the NCCL call from completing, blocking the cudaFree call.
*/
2021-04-12 16:00:11 -07:00
// Check whether we are in cuda graph mode
NCCLCHECK(ncclCalloc(&graphs, ncclGroupIndex));
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL) {
ncclComm_t comm = args->coll.comm;
NCCLCHECKGOTO(ncclGetCudaGraph(comm, graphs+i), ret, group_cleanup);
if (usingCudaGraphAll == -1) {
usingCudaGraphAll = comm->usingCudaGraph;
} else if (usingCudaGraphAll != comm->usingCudaGraph) {
WARN("Illegal to have some communicators in graph mode while others not");
ret = ncclInvalidUsage;
goto group_cleanup;
}
}
}
2020-09-04 14:35:05 -07:00
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL) {
ncclComm_t comm = args->coll.comm;
2021-04-12 16:00:11 -07:00
NCCLCHECKGOTO(ncclSetupAsyncKernels(comm), ret, group_cleanup);
2020-09-04 14:35:05 -07:00
}
}
2018-09-24 16:06:59 -07:00
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL) {
if (args->coll.comm->userStream == hipStreamDefault/* ||
args->coll.comm->userStream == hipStreamPerThread ||
args->coll.comm->userStream == hipStreamLegacy*/)
2019-07-05 15:43:00 -07:00
CUDACHECKGOTO(hipSetDevice(args->coll.comm->cudaDev), ret, end);
2021-04-12 16:00:11 -07:00
if (usingCudaGraphAll == 1) {
NCCLCHECKGOTO(ncclCudaGraphHostSetup(args->coll.comm, graphs[i]), ret, end);
} else {
ncclEnqueueHostSetup<0>(args->coll.comm->enqueueInfo);
}
NCCLCHECKGOTO(ncclLaunchBarrier(args->coll.comm), ret, end);
2018-09-24 16:06:59 -07:00
}
}
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL) {
2019-07-05 15:43:00 -07:00
CUDACHECKGOTO(hipSetDevice(args->coll.comm->cudaDev), ret, end);
2021-04-12 16:00:11 -07:00
NCCLCHECKGOTO(ncclLaunchKernel(args->coll.comm), ret, end);
2018-09-24 16:06:59 -07:00
}
}
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
if (args->funcType == ASYNC_FUNC_COLL) {
if (args->coll.comm->userStream == hipStreamDefault/* ||
args->coll.comm->userStream == hipStreamPerThread ||
args->coll.comm->userStream == hipStreamLegacy*/)
2019-07-05 15:43:00 -07:00
CUDACHECKGOTO(hipSetDevice(args->coll.comm->cudaDev), ret, end);
2021-04-12 16:00:11 -07:00
NCCLCHECKGOTO(ncclRecordEvents(args->coll.comm), ret, end);
NCCLCHECKGOTO(ncclLaunchReset(args->coll.comm), ret, end);
2018-09-24 16:06:59 -07:00
}
}
goto end;
group_cleanup:
2019-11-19 14:57:39 -08:00
if (ret != ncclSuccess) {
// At least one call in the group failed. Since we want to make that group
// an atomic operation, we need to cancel all operations.
for (int i=0; i<ncclGroupIndex; i++) {
struct ncclAsyncArgs* args = ncclGroupArgs+i;
2020-05-12 14:40:18 -07:00
if (args->funcType == ASYNC_FUNC_INIT) {
if (args->init.newcomm) ncclCommDestroy(*args->init.newcomm);
2019-11-19 14:57:39 -08:00
*args->init.newcomm = NULL;
} else {
struct ncclComm* comm = args->coll.comm;
2020-09-04 14:35:05 -07:00
// Reset aggregation counters
comm->asyncOpCount = 0;
comm->asyncTotalSize = 0;
// Dequeue p2p lists
if (comm->p2pSendCount > 0 || comm->p2pRecvCount > 0) {
for (int peer=0; peer<comm->nRanks; peer++) {
2021-07-08 14:12:04 -07:00
if (comm->p2pSends[peer]) comm->p2pSends[peer]->recycle();
if (comm->p2pRecvs[peer]) comm->p2pRecvs[peer]->recycle();
2019-11-19 14:57:39 -08:00
}
2020-09-04 14:35:05 -07:00
comm->p2pSendCount = comm->p2pRecvCount = 0;
2019-11-19 14:57:39 -08:00
}
2021-04-12 16:00:11 -07:00
ncclLaunchReset(comm);
2018-09-24 16:06:59 -07:00
}
}
}
end:
ncclGroupError = ncclSuccess;
ncclGroupIndex = 0;
2019-07-05 15:43:00 -07:00
CUDACHECK(hipSetDevice(savedDev)); // do other clean-ups first before calling hipSetDevice, because this call can fail too
2021-04-26 14:24:50 -07:00
if (graphs) free(graphs);
2018-09-24 16:06:59 -07:00
return ret;
}