2.17.1-1
Add new NVLS algorithm for allreduce using NVLink SHARP (intra-node only).
Add new config options: cgaClusterSize, minCTAs, maxCTAs, netName.
Enable LL128 when we use PXN to close rings.
NVTX3 includes update.
Fix crash when one CollNet (SHARP) rail fails to initialize.
[ROCm/rccl commit: 5d3ab08b69]
This commit is contained in:
+177
-58
@@ -35,13 +35,13 @@
|
||||
#endif
|
||||
|
||||
const char* ncclFuncStr[NCCL_NUM_FUNCTIONS] = { "Broadcast", "Reduce", "AllGather", "ReduceScatter", "AllReduce" };
|
||||
const char* ncclAlgoStr[NCCL_NUM_ALGORITHMS] = { "Tree", "Ring", "CollNetDirect", "CollNetChain" };
|
||||
const char* ncclAlgoStr[NCCL_NUM_ALGORITHMS] = { "Tree", "Ring", "CollNetDirect", "CollNetChain", "NVLS" };
|
||||
const char* ncclProtoStr[NCCL_NUM_PROTOCOLS] = { "LL", "LL128", "Simple" };
|
||||
|
||||
NCCL_PARAM(GroupCudaStream, "GROUP_CUDA_STREAM", NCCL_GROUP_CUDA_STREAM);
|
||||
|
||||
NCCL_PARAM(CheckPointers, "CHECK_POINTERS", 0);
|
||||
NCCL_PARAM(CommBlocking, "COMM_BLOCKING", 0);
|
||||
NCCL_PARAM(CommBlocking, "COMM_BLOCKING", NCCL_CONFIG_UNDEF_INT);
|
||||
|
||||
static uint64_t hashUniqueId(ncclUniqueId const &id) {
|
||||
char const *bytes = (char const*)&id;
|
||||
@@ -67,12 +67,8 @@ ncclResult_t initGdrCopy() {
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
|
||||
NCCL_PARAM(L1SharedMemoryCarveout, "L1_SHARED_MEMORY_CARVEOUT", 0);
|
||||
|
||||
pthread_mutex_t initLock = PTHREAD_MUTEX_INITIALIZER;
|
||||
static bool initialized = false;
|
||||
static size_t maxLocalSizeBytes = 0;
|
||||
|
||||
static ncclResult_t ncclInit() {
|
||||
if (__atomic_load_n(&initialized, __ATOMIC_ACQUIRE)) return ncclSuccess;
|
||||
@@ -80,9 +76,6 @@ static ncclResult_t ncclInit() {
|
||||
if (!initialized) {
|
||||
initEnv();
|
||||
initGdrCopy();
|
||||
maxLocalSizeBytes = ncclKernMaxLocalSize();
|
||||
int carveout = ncclParamL1SharedMemoryCarveout();
|
||||
if (carveout) ncclKernSetSharedMemoryCarveout(carveout);
|
||||
// Always initialize bootstrap network
|
||||
NCCLCHECK(bootstrapNetInit());
|
||||
NCCLCHECK(ncclNetPluginInit());
|
||||
@@ -210,6 +203,8 @@ static ncclResult_t commFree(ncclComm_t comm) {
|
||||
NCCLCHECK(ncclStrongStreamDestruct(&comm->deviceStream));
|
||||
}
|
||||
|
||||
if (comm->nvlsSupport) NCCLCHECK(ncclNvlsFree(comm));
|
||||
|
||||
struct ncclDestructor* dtor = comm->destructorHead;
|
||||
while (dtor != nullptr) {
|
||||
NCCLCHECK(dtor->fn(dtor));
|
||||
@@ -220,6 +215,7 @@ static ncclResult_t commFree(ncclComm_t comm) {
|
||||
ncclMemoryStackDestruct(&comm->memPermanent);
|
||||
|
||||
ncclCudaHostFree((void *)comm->abortFlag);
|
||||
free(comm->netName);
|
||||
|
||||
commPoison(comm); // poison comm before free to avoid comm reuse.
|
||||
free(comm);
|
||||
@@ -243,8 +239,8 @@ static ncclResult_t dmaBufSupported(struct ncclComm* comm) {
|
||||
int flag = 0;
|
||||
CUdevice dev;
|
||||
int cudaDriverVersion;
|
||||
CUCHECK(cuDriverGetVersion(&cudaDriverVersion));
|
||||
if (cudaDriverVersion < 11070) return ncclInternalError;
|
||||
CUDACHECK(cudaDriverGetVersion(&cudaDriverVersion));
|
||||
if (CUPFN(cuDeviceGet) == NULL || cudaDriverVersion < 11070) return ncclInternalError;
|
||||
CUCHECK(cuDeviceGet(&dev, comm->cudaDev));
|
||||
// Query device to see if DMA-BUF support is available
|
||||
(void) CUPFN(cuDeviceGetAttribute(&flag, CU_DEVICE_ATTRIBUTE_DMA_BUF_SUPPORTED, dev));
|
||||
@@ -265,7 +261,7 @@ ncclResult_t ncclCommEnsureReady(ncclComm_t comm) {
|
||||
NCCLCHECK(ncclCommGetAsyncError(comm, &ret));
|
||||
if (ret != ncclSuccess) {
|
||||
/* if ret is not ncclInProgress, we just keep it. */
|
||||
WARN("Attempt to use communicator before the previous operation returned ncclSuccess\n");
|
||||
WARN("Attempt to use communicator before the previous operation returned ncclSuccess");
|
||||
if (ret == ncclInProgress) ret = ncclInvalidArgument;
|
||||
goto exit;
|
||||
}
|
||||
@@ -395,6 +391,7 @@ static ncclResult_t devCommSetup(ncclComm_t comm) {
|
||||
tmpCommAndChans.channels[c].tree = comm->channels[c].tree;
|
||||
tmpCommAndChans.channels[c].collnetChain = comm->channels[c].collnetChain;
|
||||
tmpCommAndChans.channels[c].collnetDirect = comm->channels[c].collnetDirect;
|
||||
tmpCommAndChans.channels[c].nvls = comm->channels[c].nvls;
|
||||
tmpCommAndChans.channels[c].workFifoDone = &comm->workFifoDone[c];
|
||||
|
||||
if (comm->channels[c].ring.userRanks != nullptr) {
|
||||
@@ -521,8 +518,8 @@ static ncclResult_t collNetTrySetup(ncclComm_t comm, struct ncclTopoGraph* collN
|
||||
struct ncclChannel* channel = comm->channels + c;
|
||||
for (int h = 0; h < nHeads; h++) {
|
||||
const int head = heads[h];
|
||||
collNetSetupFail = ncclTransportCollNetSetup(comm, collNetGraph, channel, head, head, h, collNetRecv);
|
||||
if (!collNetSetupFail) collNetSetupFail = ncclTransportCollNetSetup(comm, collNetGraph, channel, head, head, h, collNetSend);
|
||||
collNetSetupFail |= ncclTransportCollNetSetup(comm, collNetGraph, channel, head, head, h, collNetRecv);
|
||||
if (!collNetSetupFail) collNetSetupFail |= ncclTransportCollNetSetup(comm, collNetGraph, channel, head, head, h, collNetSend);
|
||||
}
|
||||
// Verify CollNet setup across ranks after trying the first channel
|
||||
if (c == 0) {
|
||||
@@ -922,6 +919,8 @@ static ncclResult_t initTransportsRank(struct ncclComm* comm, ncclUniqueId* comm
|
||||
// Check if we can setup CollNet
|
||||
if (comm->collNetSupport > 0) collNetTrySetup(comm, &collNetGraph);
|
||||
|
||||
NCCLCHECKGOTO(ncclNvlsSetup(comm), ret, fail);
|
||||
|
||||
TRACE(NCCL_INIT, "rank %d nranks %d - CONNECTED %d RINGS AND TREES", rank, nranks, comm->nChannels);
|
||||
|
||||
// Compute time models for algorithm and protocol combinations
|
||||
@@ -929,7 +928,7 @@ static ncclResult_t initTransportsRank(struct ncclComm* comm, ncclUniqueId* comm
|
||||
int myCompCap = comm->peerInfo[rank].cudaCompCap;
|
||||
int minCompCap = myCompCap, maxCompCap = myCompCap;
|
||||
for (int i = 0; i < nranks; i++) {
|
||||
minCompCap = std::min(comm->peerInfo[i].cudaCompCap, minCompCap);
|
||||
comm->minCompCap = minCompCap = std::min(comm->peerInfo[i].cudaCompCap, minCompCap);
|
||||
maxCompCap = std::max(comm->peerInfo[i].cudaCompCap, maxCompCap);
|
||||
}
|
||||
NCCLCHECKGOTO(ncclTopoTuneModel(comm, minCompCap, maxCompCap, &treeGraph, &ringGraph, &collNetGraph), ret, fail);
|
||||
@@ -938,6 +937,8 @@ static ncclResult_t initTransportsRank(struct ncclComm* comm, ncclUniqueId* comm
|
||||
// Compute nChannels per peer for p2p
|
||||
NCCLCHECKGOTO(ncclTopoComputeP2pChannels(comm), ret, fail);
|
||||
|
||||
INFO(NCCL_INIT, "%d coll channels, %d nvls channels, %d p2p channels, %d p2p channels per peer", comm->nChannels, comm->nvlsChannels, comm->p2pnChannels, comm->p2pnChannelsPerPeer);
|
||||
|
||||
do { // Setup p2p structures in comm->tasks
|
||||
struct ncclTasks* tasks = &comm->tasks;
|
||||
int nRanks = comm->nRanks;
|
||||
@@ -1004,12 +1005,13 @@ static ncclResult_t initTransportsRank(struct ncclComm* comm, ncclUniqueId* comm
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
NCCLCHECKGOTO(ncclTransportP2pSetup(comm, NULL, 1), ret, fail);
|
||||
}
|
||||
|
||||
// Connect to local net proxy
|
||||
NCCLCHECKGOTO(ncclProxyConnect(comm, TRANSPORT_NET, 1, comm->rank, &proxyConn), ret, fail);
|
||||
NCCLCHECKGOTO(ncclProxyCall(&proxyConn, ncclProxyMsgSharedInit, &comm->p2pnChannels, sizeof(int), NULL, 0), ret, fail);
|
||||
NCCLCHECKGOTO(ncclProxyCallBlocking(&proxyConn, ncclProxyMsgSharedInit, &comm->p2pnChannels, sizeof(int), NULL, 0), ret, fail);
|
||||
|
||||
// Then to remote ones when using PXN
|
||||
if (ncclPxnDisable(comm) == 0) {
|
||||
@@ -1017,7 +1019,7 @@ static ncclResult_t initTransportsRank(struct ncclComm* comm, ncclUniqueId* comm
|
||||
NCCLCHECKGOTO(ncclTopoGetPxnRanks(comm, &pxnPeers, &nranks), ret, fail);
|
||||
for (int r=0; r<nranks; r++) {
|
||||
NCCLCHECKGOTO(ncclProxyConnect(comm, TRANSPORT_NET, 1, pxnPeers[r], &proxyConn), ret, fail);
|
||||
NCCLCHECKGOTO(ncclProxyCall(&proxyConn, ncclProxyMsgSharedInit, &comm->p2pnChannels, sizeof(int), NULL, 0), ret, fail);
|
||||
NCCLCHECKGOTO(ncclProxyCallBlocking(&proxyConn, ncclProxyMsgSharedInit, &comm->p2pnChannels, sizeof(int), NULL, 0), ret, fail);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1065,6 +1067,11 @@ fail:
|
||||
}
|
||||
|
||||
NCCL_PARAM(SetStackSize, "SET_STACK_SIZE", 0);
|
||||
NCCL_PARAM(CGAClusterSize, "CGA_CLUSTER_SIZE", NCCL_CONFIG_UNDEF_INT);
|
||||
// Match config max/minCTAs
|
||||
NCCL_PARAM(MaxCTAs, "MAX_CTAS", NCCL_CONFIG_UNDEF_INT);
|
||||
NCCL_PARAM(MinCTAs, "MIN_CTAS", NCCL_CONFIG_UNDEF_INT);
|
||||
#define NCCL_MAX_CGA_CLUSTER_SIZE 8
|
||||
|
||||
struct ncclCommInitRankAsyncJob {
|
||||
struct ncclAsyncJob base;
|
||||
@@ -1087,9 +1094,16 @@ static ncclResult_t ncclCommInitRankFunc(struct ncclAsyncJob* job_) {
|
||||
ncclUniqueId commId = job->commId; // C++ struct assignment
|
||||
int myrank = job->myrank;
|
||||
int cudaDev = job->cudaDev;
|
||||
int archMajor, archMinor;
|
||||
size_t maxLocalSizeBytes = 0;
|
||||
ncclResult_t res = ncclSuccess;
|
||||
|
||||
CUDACHECKGOTO(cudaSetDevice(cudaDev), res, fail);
|
||||
CUDACHECK(cudaDeviceGetAttribute(&archMajor, cudaDevAttrComputeCapabilityMajor, cudaDev));
|
||||
CUDACHECK(cudaDeviceGetAttribute(&archMinor, cudaDevAttrComputeCapabilityMinor, cudaDev));
|
||||
comm->cudaArch = 100*archMajor + 10*archMinor;
|
||||
|
||||
NCCLCHECK(ncclInitKernelsForDevice(comm->cudaArch, &maxLocalSizeBytes));
|
||||
// Set the maximum kernel stack size of all kernels to avoid
|
||||
// a CUDA memory reconfig on load (c.f. NVSHMEM issue)
|
||||
if (maxLocalSizeBytes > 0 && ncclParamSetStackSize() == 1) {
|
||||
@@ -1114,18 +1128,143 @@ fail:
|
||||
goto exit;
|
||||
}
|
||||
|
||||
static ncclResult_t parseCommConfig(ncclComm_t comm, ncclConfig_t *config) {
|
||||
ncclResult_t ret = ncclSuccess;
|
||||
|
||||
/* first set configuration */
|
||||
if (config) {
|
||||
comm->blocking = config->blocking;
|
||||
} else {
|
||||
/* default setting of communicator */
|
||||
comm->blocking = 1;
|
||||
#define NCCL_CONFIG_DEFAULT(config, field, undef, defvalue, fieldStr, format) \
|
||||
if (config->field == undef) { \
|
||||
config->field = defvalue; \
|
||||
} else { \
|
||||
INFO(NCCL_ENV, "Comm config " fieldStr " set to " format, config->field); \
|
||||
}
|
||||
|
||||
static ncclResult_t parseCommConfig(ncclComm_t comm, ncclConfig_t *config) {
|
||||
ncclResult_t ret = ncclSuccess;
|
||||
/* config must not be NULL in this function */
|
||||
int blockingEnv;
|
||||
int cgaClusterSizeEnv;
|
||||
int minCTAsEnv;
|
||||
int maxCTAsEnv;
|
||||
const char *envNetName, *tmpNetName;
|
||||
ncclConfig_t defaultConfig = NCCL_CONFIG_INITIALIZER;
|
||||
ncclConfig_t internalConfig = NCCL_CONFIG_INITIALIZER;
|
||||
ncclConfig_t *internalConfigPtr;
|
||||
size_t realSize;
|
||||
|
||||
internalConfigPtr = &internalConfig;
|
||||
if (config) {
|
||||
memcpy((void*)&realSize, (void*)config, sizeof(size_t));
|
||||
realSize = realSize > sizeof(ncclConfig_t) ? sizeof(ncclConfig_t) : realSize;
|
||||
memcpy((void*)internalConfigPtr, (void*)config, realSize);
|
||||
if (internalConfigPtr->magic != 0xcafebeef) {
|
||||
WARN("ncclConfig_t argument not initialized via NCCL_CONFIG_INITIALIZER");
|
||||
ret = ncclInvalidArgument;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
/* check version. */
|
||||
if (internalConfigPtr->version < NCCL_VERSION(2, 14, 0)) {
|
||||
internalConfigPtr->blocking = defaultConfig.blocking;
|
||||
}
|
||||
|
||||
if (internalConfigPtr->version < NCCL_VERSION(2, 17, 0)) {
|
||||
internalConfigPtr->cgaClusterSize = defaultConfig.cgaClusterSize;
|
||||
internalConfigPtr->minCTAs = defaultConfig.minCTAs;
|
||||
internalConfigPtr->maxCTAs = defaultConfig.maxCTAs;
|
||||
internalConfigPtr->netName = defaultConfig.netName;
|
||||
}
|
||||
}
|
||||
|
||||
/* check input config attributes, -1 means user-undefined and we should use default value from NCCL. */
|
||||
if (internalConfigPtr->blocking != NCCL_CONFIG_UNDEF_INT && internalConfigPtr->blocking != 0 && internalConfigPtr->blocking != 1) {
|
||||
WARN("Invalid config blocking attribute value %d", internalConfigPtr->blocking);
|
||||
ret = ncclInvalidArgument;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
if (internalConfigPtr->cgaClusterSize != NCCL_CONFIG_UNDEF_INT && internalConfigPtr->cgaClusterSize < 0) {
|
||||
WARN("Invalid config cgaClusterSize attribute value %d", internalConfigPtr->cgaClusterSize);
|
||||
ret = ncclInvalidArgument;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
if ((internalConfigPtr->minCTAs != NCCL_CONFIG_UNDEF_INT &&
|
||||
internalConfigPtr->minCTAs <= 0) ||
|
||||
(internalConfigPtr->maxCTAs != NCCL_CONFIG_UNDEF_INT &&
|
||||
internalConfigPtr->maxCTAs <= 0) ||
|
||||
(internalConfigPtr->minCTAs > internalConfigPtr->maxCTAs)) {
|
||||
WARN("Invalid config min/max channels attribute value %d/%d", internalConfigPtr->minCTAs, internalConfigPtr->maxCTAs);
|
||||
ret = ncclInvalidArgument;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
/* default config value can be tuned on different platform. */
|
||||
NCCL_CONFIG_DEFAULT(internalConfigPtr, blocking, NCCL_CONFIG_UNDEF_INT, 1, "Blocking", "%d");
|
||||
NCCL_CONFIG_DEFAULT(internalConfigPtr, cgaClusterSize, NCCL_CONFIG_UNDEF_INT, 4, "CGA cluster size", "%d");
|
||||
NCCL_CONFIG_DEFAULT(internalConfigPtr, minCTAs, NCCL_CONFIG_UNDEF_INT, 1, "Min CTAs", "%d");
|
||||
NCCL_CONFIG_DEFAULT(internalConfigPtr, maxCTAs, NCCL_CONFIG_UNDEF_INT, MAXCHANNELS, "Max CTAs", "%d");
|
||||
NCCL_CONFIG_DEFAULT(internalConfigPtr, netName, NCCL_CONFIG_UNDEF_PTR, NULL, "Net name", "%s");
|
||||
|
||||
tmpNetName = internalConfigPtr->netName;
|
||||
|
||||
/* assign config to communicator */
|
||||
comm->blocking = internalConfigPtr->blocking;
|
||||
comm->cgaClusterSize = internalConfigPtr->cgaClusterSize;
|
||||
comm->minCTAs = internalConfigPtr->minCTAs;
|
||||
comm->maxCTAs = internalConfigPtr->maxCTAs;
|
||||
|
||||
/* override configuration from env variable. */
|
||||
blockingEnv = ncclParamCommBlocking();
|
||||
if (blockingEnv == 0 || blockingEnv == 1)
|
||||
comm->blocking = blockingEnv;
|
||||
|
||||
cgaClusterSizeEnv = ncclParamCGAClusterSize();
|
||||
if (0 <= cgaClusterSizeEnv && cgaClusterSizeEnv <= NCCL_MAX_CGA_CLUSTER_SIZE) {
|
||||
comm->cgaClusterSize = cgaClusterSizeEnv;
|
||||
} else if (cgaClusterSizeEnv > NCCL_MAX_CGA_CLUSTER_SIZE) {
|
||||
WARN("NCCL_CGA_CLUSTER_SIZE value %d is too big. Limiting value to %d.", cgaClusterSizeEnv, NCCL_MAX_CGA_CLUSTER_SIZE);
|
||||
comm->cgaClusterSize = NCCL_MAX_CGA_CLUSTER_SIZE;
|
||||
}
|
||||
|
||||
minCTAsEnv = ncclParamMinCTAs();
|
||||
if (minCTAsEnv != NCCL_CONFIG_UNDEF_INT) {
|
||||
comm->minCTAs = minCTAsEnv;
|
||||
}
|
||||
|
||||
maxCTAsEnv = ncclParamMaxCTAs();
|
||||
if (maxCTAsEnv != NCCL_CONFIG_UNDEF_INT) {
|
||||
comm->maxCTAs = maxCTAsEnv;
|
||||
}
|
||||
|
||||
/* cap channels if needed */
|
||||
if (comm->minCTAs > MAXCHANNELS) {
|
||||
WARN("minCTAs %d is larger than #channels upper limit %d", comm->minCTAs, MAXCHANNELS);
|
||||
comm->minCTAs = MAXCHANNELS;
|
||||
}
|
||||
|
||||
if (comm->maxCTAs > MAXCHANNELS) {
|
||||
WARN("maxCTAs %d is larger than #channels upper limit %d", comm->maxCTAs, MAXCHANNELS);
|
||||
comm->maxCTAs = MAXCHANNELS;
|
||||
}
|
||||
|
||||
if (comm->minCTAs > comm->maxCTAs) {
|
||||
WARN("minCTAs %d is larger than maxCTAs %d", comm->minCTAs, comm->maxCTAs);
|
||||
ret = ncclInvalidArgument;
|
||||
goto fail;
|
||||
}
|
||||
|
||||
envNetName = getenv("NCCL_NET");
|
||||
if (envNetName)
|
||||
tmpNetName = envNetName;
|
||||
if (tmpNetName != NULL) {
|
||||
int netNameLen = strlen(tmpNetName) + 1;
|
||||
comm->netName = (char*)malloc(netNameLen);
|
||||
memcpy(comm->netName, tmpNetName, netNameLen);
|
||||
} else {
|
||||
comm->netName = NULL;
|
||||
}
|
||||
|
||||
exit:
|
||||
return ret;
|
||||
fail:
|
||||
goto exit;
|
||||
}
|
||||
|
||||
static void ncclCommInitRankUndo(struct ncclAsyncJob* job_) {
|
||||
@@ -1151,6 +1290,7 @@ static ncclResult_t ncclCommInitRankDev(ncclComm_t* newcomm, int nranks, ncclUni
|
||||
CUDACHECKGOTO(cudaFree(NULL), res, fail);
|
||||
|
||||
NCCLCHECKGOTO(PtrCheck(newcomm, "CommInitRank", "newcomm"), res, fail);
|
||||
NCCLCHECKGOTO(PtrCheck(config, "CommInitRank", "config"), res, fail);
|
||||
if (nranks < 1 || myrank < 0 || myrank >= nranks) {
|
||||
WARN("Invalid rank requested : %d/%d", myrank, nranks);
|
||||
res = ncclInvalidArgument;
|
||||
@@ -1201,12 +1341,13 @@ ncclResult_t ncclCommInitRank(ncclComm_t* newcomm, int nranks, ncclUniqueId comm
|
||||
(void)ncclCudaLibraryInit();
|
||||
|
||||
int cudaDev;
|
||||
ncclConfig_t config = NCCL_CONFIG_INITIALIZER;
|
||||
CUDACHECK(cudaGetDevice(&cudaDev));
|
||||
|
||||
NvtxParamsCommInitRank payload{myrank, nranks, cudaDev};
|
||||
NVTX3_FUNC_WITH_PARAMS(CommInitRank, CommInitRankSchema, payload)
|
||||
|
||||
NCCLCHECK(ncclCommInitRankDev(newcomm, nranks, commId, myrank, cudaDev, NULL));
|
||||
NCCLCHECK(ncclCommInitRankDev(newcomm, nranks, commId, myrank, cudaDev, &config));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -1215,6 +1356,7 @@ ncclResult_t ncclCommInitAll(ncclComm_t* comms, int ndev, const int* devlist) {
|
||||
ncclResult_t ret = ncclSuccess;
|
||||
int totalnDev;
|
||||
int *gpuFlags = NULL;
|
||||
ncclConfig_t config = NCCL_CONFIG_INITIALIZER;
|
||||
|
||||
constexpr nvtxPayloadSchemaEntry_t CommInitAllSchema[] = {
|
||||
{0, NVTX_PAYLOAD_ENTRY_TYPE_INT, "No. of devices"}
|
||||
@@ -1258,7 +1400,7 @@ ncclResult_t ncclCommInitAll(ncclComm_t* comms, int ndev, const int* devlist) {
|
||||
NCCLCHECKGOTO(ncclGroupStart(), ret, fail);
|
||||
for (int i=0; i<ndev; i++) {
|
||||
// Ignore return codes .. we need to call ncclGroupEnd to clean up anyway
|
||||
ncclCommInitRankDev(comms+i, ndev, uniqueId, i, devlist ? devlist[i] : i, NULL);
|
||||
ncclCommInitRankDev(comms+i, ndev, uniqueId, i, devlist ? devlist[i] : i, &config);
|
||||
}
|
||||
NCCLCHECKGOTO(ncclGroupEnd(), ret, fail);
|
||||
|
||||
@@ -1283,39 +1425,16 @@ ncclResult_t ncclCommInitRankConfig(ncclComm_t *newcomm, int nranks, ncclUniqueI
|
||||
int cudaDev;
|
||||
ncclResult_t ret = ncclSuccess;
|
||||
ncclConfig_t internalConfig = NCCL_CONFIG_INITIALIZER;
|
||||
ncclConfig_t *internalConfigPtr;
|
||||
size_t realSize;
|
||||
int blockingEnv;
|
||||
|
||||
ncclConfig_t *internalConfigPtr = NULL;
|
||||
NCCLCHECK(ncclGroupStartInternal());
|
||||
internalConfigPtr = &internalConfig;
|
||||
if (config) {
|
||||
memcpy((void*)&realSize, (void*)config, sizeof(size_t));
|
||||
realSize = realSize > sizeof(ncclConfig_t) ? sizeof(ncclConfig_t) : realSize;
|
||||
memcpy((void*)internalConfigPtr, (void*)config, realSize);
|
||||
if (internalConfigPtr->magic != 0xcafebeef) {
|
||||
WARN("ncclConfig_t argument not initialized via NCCL_CONFIG_INITIALIZER");
|
||||
ret = ncclInvalidArgument;
|
||||
goto exit;
|
||||
}
|
||||
}
|
||||
|
||||
/* check input config attributes */
|
||||
if (internalConfigPtr->blocking != 0 && internalConfigPtr->blocking != 1) {
|
||||
WARN("Invalid config blocking attribute value %d", internalConfigPtr->blocking);
|
||||
ret = ncclInvalidArgument;
|
||||
goto exit;
|
||||
}
|
||||
|
||||
/* overwrite configuration from env variable. */
|
||||
blockingEnv = ncclParamCommBlocking();
|
||||
if (blockingEnv != 0 && blockingEnv != 1) {
|
||||
WARN("Invalid NCCL_COMM_BLOCKING value %d", blockingEnv);
|
||||
}
|
||||
if (blockingEnv == 1) internalConfigPtr->blocking = blockingEnv;
|
||||
|
||||
(void)ncclCudaLibraryInit();
|
||||
CUDACHECKGOTO(cudaGetDevice(&cudaDev), ret, exit);
|
||||
CUDACHECKGOTO(cudaGetDevice(&cudaDev), ret, fail);
|
||||
|
||||
if (config == NULL)
|
||||
internalConfigPtr = &internalConfig;
|
||||
else
|
||||
internalConfigPtr = config;
|
||||
NCCLCHECKGOTO(ncclCommInitRankDev(newcomm, nranks, commId, myrank, cudaDev, internalConfigPtr), ret, fail);
|
||||
|
||||
exit:
|
||||
|
||||
Reference in New Issue
Block a user