Merge remote-tracking branch 'nccl/master' into develop
This commit is contained in:
@@ -456,7 +456,7 @@ static ncclResult_t connectNvls(struct ncclComm* comm, int* nvlsHeads, int nHead
|
||||
channel->nvls.out = -1; // NVLS+SHARP not yet implemented.
|
||||
channel->nvls.headRank = headRank;
|
||||
channel->nvls.treeUp = channel->nvls.treeDown[0] = channel->nvls.treeDown[1] = channel->nvls.treeDown[2] = -1;
|
||||
if (comm->collNetSupport && channel->nvls.headRank != -1) channel->nvls.out = comm->nRanks;
|
||||
if (comm->config.collnetEnable && channel->nvls.headRank != -1) channel->nvls.out = comm->nRanks;
|
||||
}
|
||||
if (comm->nNodes == 1) return ncclSuccess;
|
||||
|
||||
@@ -528,7 +528,7 @@ int ncclMinNchannels() {
|
||||
if (ncclParamMinNrings() != -2) minNchannels = ncclParamMinNrings();
|
||||
if (ncclParamMinNchannels() != -2) minNchannels = ncclParamMinNchannels();
|
||||
if (minNchannels > MAXCHANNELS) {
|
||||
WARN("User asked for a minimum of %d channels, limiting to %d", minNchannels, MAXCHANNELS);
|
||||
INFO(NCCL_GRAPH|NCCL_ENV, "User asked for a minimum of %d channels, limiting to %d", minNchannels, MAXCHANNELS);
|
||||
minNchannels = MAXCHANNELS;
|
||||
}
|
||||
if (minNchannels < 0) minNchannels = 0;
|
||||
@@ -544,7 +544,7 @@ int ncclMaxNchannels() {
|
||||
maxNchannels = std::min(maxNchannels, ncclDevMaxChannelsForArgsBytes(ncclParamWorkArgsBytes()));
|
||||
if (maxNchannels > MAXCHANNELS) maxNchannels = MAXCHANNELS;
|
||||
if (maxNchannels < 1) {
|
||||
WARN("User asked for a maximum of %d channels, setting it to 1", maxNchannels);
|
||||
INFO(NCCL_GRAPH|NCCL_ENV, "User asked for a maximum of %d channels, setting it to 1", maxNchannels);
|
||||
maxNchannels = 1;
|
||||
}
|
||||
return maxNchannels;
|
||||
@@ -718,7 +718,7 @@ ncclResult_t ncclTopoPostset(struct ncclComm* comm, int* firstRanks, int* treePa
|
||||
int nNodes = comm->nNodes;
|
||||
int nChannels = comm->nChannels;
|
||||
int minHeadNum = INT_MAX;
|
||||
int shared = parent && parent->nvlsSupport && parent->config.splitShare;
|
||||
int shared = parent && parent->nvlsSupport && parent->shareResources;
|
||||
int maxChannels;
|
||||
int minNchannels, maxNchannels;
|
||||
NCCLCHECK(ncclCalloc(&ringRecv, nNodes*MAXCHANNELS));
|
||||
@@ -839,7 +839,7 @@ ncclResult_t ncclTopoPostset(struct ncclComm* comm, int* firstRanks, int* treePa
|
||||
nChannels = comm->nChannels = std::min(maxChannels, (nChannels <= maxChannels/2) ? nChannels*2 : nChannels);
|
||||
|
||||
// Setup CollNet
|
||||
if (comm->collNetSupport == 1) {
|
||||
if (comm->config.collnetEnable) {
|
||||
struct ncclTopoGraph* collNetChainGraph = graphs[NCCL_ALGO_COLLNET_CHAIN];
|
||||
// Add more channels to saturate intra-node bandwidth, except the 1 PPN case
|
||||
if (collNetChainGraph->bwIntra > collNetChainGraph->bwInter && comm->nRanks > comm->nNodes) {
|
||||
|
||||
+62
-35
@@ -219,7 +219,7 @@ ncclResult_t ncclGetLevel(int* level, const char* disableEnv, const char* levelE
|
||||
const char* str = ncclGetEnv(disableEnv);
|
||||
if (str) {
|
||||
int disable = strtol(str, NULL, 0);
|
||||
if (disable == 1) l = 0;
|
||||
if (disable == 1) l = PATH_LOC;
|
||||
if (l >= 0) INFO(NCCL_ALL, "%s set by environment to %d", disableEnv, disable);
|
||||
}
|
||||
}
|
||||
@@ -252,7 +252,18 @@ ncclResult_t ncclGetLevel(int* level, const char* disableEnv, const char* levelE
|
||||
|
||||
NCCL_PARAM(IgnoreDisabledP2p, "IGNORE_DISABLED_P2P", 0);
|
||||
|
||||
int ncclTopoUserP2pLevel = -1;
|
||||
static int ncclTopoUserP2pLevel = -1; // Initially "uninitialized". When initialized but unset, changes to -2.
|
||||
|
||||
// Gets the user-provided value of NCCL_P2P_LEVEL/NCCL_P2P_DISABLE. If the user did not provide any, the value
|
||||
// of the "level" argument is left unchanged.
|
||||
ncclResult_t ncclGetUserP2pLevel(int* level) {
|
||||
if (ncclTopoUserP2pLevel == -1)
|
||||
NCCLCHECK(ncclGetLevel(&ncclTopoUserP2pLevel, "NCCL_P2P_DISABLE", "NCCL_P2P_LEVEL"));
|
||||
if (ncclTopoUserP2pLevel != -2)
|
||||
*level = ncclTopoUserP2pLevel;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoCheckP2p(struct ncclComm* comm, struct ncclTopoSystem* system, int rank1, int rank2,
|
||||
int* p2p, int *read, int* intermediateRank) {
|
||||
int mnnvl = 0;
|
||||
@@ -280,9 +291,9 @@ ncclResult_t ncclTopoCheckP2p(struct ncclComm* comm, struct ncclTopoSystem* syst
|
||||
|
||||
// Get GPUs from topology
|
||||
int g1, g2;
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank1, &g1));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank1, &g1, /*showWarn=*/true));
|
||||
struct ncclTopoNode* gpu1 = system->nodes[GPU].nodes+g1;
|
||||
if (ncclTopoRankToIndex(system, rank2, &g2) == ncclInternalError) {
|
||||
if (ncclTopoRankToIndex(system, rank2, &g2, /*showWarn=*/false) == ncclInternalError) {
|
||||
// GPU not found, we can't use p2p.
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -305,12 +316,7 @@ ncclResult_t ncclTopoCheckP2p(struct ncclComm* comm, struct ncclTopoSystem* syst
|
||||
int p2pLevel = PATH_SYS;
|
||||
|
||||
// User override
|
||||
if (ncclTopoUserP2pLevel == -1)
|
||||
NCCLCHECK(ncclGetLevel(&ncclTopoUserP2pLevel, "NCCL_P2P_DISABLE", "NCCL_P2P_LEVEL"));
|
||||
if (ncclTopoUserP2pLevel != -2) {
|
||||
p2pLevel = ncclTopoUserP2pLevel;
|
||||
goto compare;
|
||||
}
|
||||
NCCLCHECK(ncclGetUserP2pLevel(&p2pLevel));
|
||||
|
||||
// Don't use P2P through ARM CPUs
|
||||
int arch, vendor, model;
|
||||
@@ -323,7 +329,6 @@ ncclResult_t ncclTopoCheckP2p(struct ncclComm* comm, struct ncclTopoSystem* syst
|
||||
p2pLevel = PATH_PXB;
|
||||
}
|
||||
|
||||
compare:
|
||||
// Compute the PCI distance and compare with the p2pLevel.
|
||||
if (path->type <= p2pLevel) *p2p = 1;
|
||||
|
||||
@@ -393,7 +398,8 @@ NCCL_PARAM(NetGdrRead, "NET_GDR_READ", -2);
|
||||
int ncclTopoUserGdrLevel = -1;
|
||||
const char* ncclTopoGdrModeStr[ncclTopoGdrModeNum] = { "Disabled", "Default", "PCI" };
|
||||
|
||||
NCCL_PARAM(NetGdrC2c, "NET_GDR_C2C", 0);
|
||||
// On C2C platforms use GDRDMA on NICs which are connected to the CPUs
|
||||
NCCL_PARAM(NetGdrC2c, "NET_GDR_C2C", 1);
|
||||
|
||||
ncclResult_t ncclTopoCheckGdr(struct ncclTopoSystem* system, int rank, int64_t netId, int read, enum ncclTopoGdrMode* gdrMode) {
|
||||
*gdrMode = ncclTopoGdrModeDisable;
|
||||
@@ -402,7 +408,7 @@ ncclResult_t ncclTopoCheckGdr(struct ncclTopoSystem* system, int rank, int64_t n
|
||||
int n, g;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netId, &n));
|
||||
struct ncclTopoNode* net = system->nodes[NET].nodes+n;
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &g));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &g, /*showWarn=*/true));
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
|
||||
// Check that both the NIC and GPUs support it
|
||||
@@ -459,29 +465,29 @@ ncclResult_t ncclTopoCheckGdr(struct ncclTopoSystem* system, int rank, int64_t n
|
||||
// In case of PXN, use the intermediate GPU distance instead
|
||||
int proxyRank;
|
||||
NCCLCHECK(ncclTopoGetIntermediateRank(system, gpu->gpu.rank, netId, &proxyRank));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, proxyRank, &g));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, proxyRank, &g, /*showWarn=*/true));
|
||||
gpu = system->nodes[GPU].nodes+g;
|
||||
distance = gpu->paths[NET][n].type;
|
||||
}
|
||||
|
||||
int c;
|
||||
NCCLCHECK(ncclGetLocalCpu(system, g, &c));
|
||||
if (ncclParamNetGdrC2c() && distance == PATH_PHB && gpu->paths[CPU][c].type == PATH_C2C) {
|
||||
// On C2C platforms we can still use GDRDMA on NICs connected to the CPUs
|
||||
INFO(NCCL_NET, "GPU %d / HCA %lx connected to CPU %d via C2C link", rank, netId, c);
|
||||
// On C2C platforms we can still use GDRDMA on NICs connected to the CPUs
|
||||
if (ncclParamNetGdrC2c() && distance == PATH_P2C) {
|
||||
INFO(NCCL_GRAPH | NCCL_NET, "GPU %d / HCA %lx connected via C2C link", rank, netId);
|
||||
distance = PATH_C2C;
|
||||
}
|
||||
|
||||
if (distance > netGdrLevel) {
|
||||
INFO(NCCL_NET,"GPU Direct RDMA Disabled for GPU %d / HCA %lx (distance %d > %d)", rank, netId, distance, netGdrLevel);
|
||||
INFO(NCCL_GRAPH|NCCL_NET,"GPU Direct RDMA Disabled for GPU %d / HCA %lx (distance %d > %d)", rank, netId, distance, netGdrLevel);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Force PCIe mapping if path goes through PCI on a C2C system
|
||||
int c;
|
||||
NCCLCHECK(ncclGetLocalCpu(system, g, &c));
|
||||
if (gpu->paths[CPU][c].type == PATH_C2C && distance != PATH_C2C) *gdrMode = ncclTopoGdrModePci;
|
||||
else *gdrMode = ncclTopoGdrModeDefault;
|
||||
|
||||
INFO(NCCL_NET,"GPU Direct RDMA Enabled for GPU %d / HCA %lx (distance %d <= %d), read %d mode %s", rank, netId, distance, netGdrLevel, read, ncclTopoGdrModeStr[*gdrMode]);
|
||||
INFO(NCCL_GRAPH|NCCL_NET,"GPU Direct RDMA Enabled for GPU %d / HCA %lx (distance %d <= %d), read %d mode %s", rank, netId, distance, netGdrLevel, read, ncclTopoGdrModeStr[*gdrMode]);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -516,7 +522,7 @@ ncclResult_t ncclTopoNeedFlush(struct ncclComm* comm, int64_t netId, int netDev,
|
||||
if (props.forceFlush == 1 || ncclParamNetForceFlush()) return ncclSuccess;
|
||||
int g;
|
||||
struct ncclTopoSystem* system = comm->topo;
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &g));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &g, /*showWarn=*/true));
|
||||
#if defined(__HIP_PLATFORM_AMD__) || defined(__HIPCC__)
|
||||
*flush = 1;
|
||||
#else
|
||||
@@ -546,8 +552,8 @@ ncclResult_t ncclTopoCheckNet(struct ncclTopoSystem* system, int rank1, int rank
|
||||
*net = 1;
|
||||
// First check the current GPU-to-GPU speed.
|
||||
int g1, g2;
|
||||
if (ncclTopoRankToIndex(system, rank1, &g1) != ncclSuccess ||
|
||||
ncclTopoRankToIndex(system, rank2, &g2) != ncclSuccess) {
|
||||
if (ncclTopoRankToIndex(system, rank1, &g1, /*showWarn=*/false) != ncclSuccess ||
|
||||
ncclTopoRankToIndex(system, rank2, &g2, /*showWarn=*/false) != ncclSuccess) {
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -573,7 +579,7 @@ ncclResult_t ncclTopoGetIntermediateRank(struct ncclTopoSystem* system, int rank
|
||||
// Get GPU and NET
|
||||
int n, g;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netId, &n));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &g));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &g, /*showWarn=*/true));
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
struct ncclTopoLinkList* path = gpu->paths[NET]+n;
|
||||
if (path->type == PATH_PXN) {
|
||||
@@ -666,6 +672,8 @@ static bool rcclPathOverride(struct ncclTopoSystem* system, uint64_t distance) {
|
||||
}
|
||||
}
|
||||
|
||||
NCCL_PARAM(PxnC2c, "PXN_C2C", 0);
|
||||
|
||||
ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclComm* comm) {
|
||||
// Precompute paths between GPUs/NICs.
|
||||
|
||||
@@ -724,6 +732,20 @@ ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclComm
|
||||
}
|
||||
}
|
||||
}
|
||||
// update the GPU -> NIC path in the case of C2C + PHB
|
||||
for (int n = 0; n < system->nodes[NET].count; n++) {
|
||||
struct ncclTopoNode* netNode = system->nodes[NET].nodes + n;
|
||||
for (int g = 0; g < system->nodes[GPU].count; g++) {
|
||||
struct ncclTopoNode* gpuNode = system->nodes[GPU].nodes + g;
|
||||
int c;
|
||||
NCCLCHECK(ncclGetLocalCpu(system, g, &c));
|
||||
if (c == -1) continue;
|
||||
if (gpuNode->paths[NET][n].type == PATH_PHB && gpuNode->paths[CPU][c].type == PATH_C2C) {
|
||||
gpuNode->paths[NET][n].type = PATH_P2C;
|
||||
netNode->paths[GPU][g].type = PATH_P2C;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Special handling of gfx94x and gfx950
|
||||
|
||||
@@ -759,15 +781,20 @@ ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclComm
|
||||
// PXN = PCI + NVLink.
|
||||
struct ncclTopoNode* peerNode = system->nodes[GPU].nodes+localGpuIndex;
|
||||
// Only use PXN for NIC n if remote GPU p ...
|
||||
if (peerNode->paths[NET][n].type <= PATH_PXB && // Is connected to the NIC through PCI
|
||||
peerNode->paths[GPU][g].type <= PATH_NVL && // Is connected to us through NVLink
|
||||
NCCL_TOPO_ID_SYSTEM_ID(peerNode->id) == NCCL_TOPO_ID_SYSTEM_ID(gpu->id) && // Is on the same node as us
|
||||
(peerNode->paths[NET][n].bw > gpu->paths[NET][n].bw || // Has either higher BW to that NIC
|
||||
gpu->paths[NET][n].type > PATH_PXB)) // or avoids going through a CPU
|
||||
// We can use that GPU as relay to communicate with that NIC.
|
||||
// Only enabling it in the GPU->NIC direction for now to favor
|
||||
// receiving locally and sending remotely (consistent with net.cc)
|
||||
NCCLCHECK(addInterStep(system, GPU, localGpuIndex, GPU, g, NET, n));
|
||||
if (/* (1) is either connected to the NIC with PXB*/
|
||||
(peerNode->paths[NET][n].type <= PATH_PXB ||
|
||||
/* or with P2C and PxN over C2C is enabled */
|
||||
(ncclParamPxnC2c() && peerNode->paths[NET][n].type == PATH_P2C)) &&
|
||||
/* and (2) is connected to us through NVLink */
|
||||
peerNode->paths[GPU][g].type <= PATH_NVL &&
|
||||
/* and (3) is on the same node as us */
|
||||
NCCL_TOPO_ID_SYSTEM_ID(peerNode->id) == NCCL_TOPO_ID_SYSTEM_ID(gpu->id) &&
|
||||
/* and (4) has either higher bw to that NIC or avoid going through the CPU*/
|
||||
(peerNode->paths[NET][n].bw > gpu->paths[NET][n].bw || gpu->paths[NET][n].type > PATH_PXB))
|
||||
// We can use that GPU as relay to communicate with that NIC.
|
||||
// Only enabling it in the GPU->NIC direction for now to favor
|
||||
// receiving locally and sending remotely (consistent with net.cc)
|
||||
NCCLCHECK(addInterStep(system, GPU, localGpuIndex, GPU, g, NET, n));
|
||||
}
|
||||
}
|
||||
if (gpu->paths[NET][n].type < PATH_PHB) {
|
||||
@@ -904,7 +931,7 @@ static ncclResult_t ncclTopoGetNchannels(struct ncclComm* comm, int g /*local gp
|
||||
int peer;
|
||||
struct ncclTopoSystem* system = comm->topo;
|
||||
struct ncclTopoLinkList* path = NULL;
|
||||
if (ncclTopoRankToIndex(system, peerRank, &peer) == ncclSuccess) {
|
||||
if (ncclTopoRankToIndex(system, peerRank, &peer, /*showWarn=*/false) == ncclSuccess) {
|
||||
// Same rank
|
||||
if (g == peer) {
|
||||
*nChannels = -1;
|
||||
|
||||
+16
-27
@@ -141,6 +141,7 @@ static ncclResult_t ncclTopoFollowPath(struct ncclTopoSystem* system, struct ncc
|
||||
float bw = intra ? graph->bwIntra : graph->bwInter;
|
||||
int type = intra ? graph->typeIntra : graph->typeInter;
|
||||
|
||||
if (path->type >= PATH_DIS) return ncclSuccess;
|
||||
if (mult == 1 && (path->type > type)) return ncclSuccess;
|
||||
if (mult == 1 && (graph->pattern == NCCL_TOPO_PATTERN_BALANCED_TREE ||
|
||||
graph->pattern == NCCL_TOPO_PATTERN_TREE ||
|
||||
@@ -332,8 +333,7 @@ ncclResult_t ncclTopoReplayGetGpu(struct ncclTopoSystem* system, struct ncclTopo
|
||||
*g = i;
|
||||
return ncclSuccess;
|
||||
}
|
||||
if (*g == -1) return ncclInternalError;
|
||||
return ncclSuccess;
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoSearchRecGpu(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, struct ncclTopoNode* gpu, int step, int backToNet, int backToFirstRank, int forcedOrder, int *time);
|
||||
@@ -709,24 +709,12 @@ ncclResult_t ncclTopoSearchRecNet(struct ncclTopoSystem* system, struct ncclTopo
|
||||
}
|
||||
|
||||
// Then try the most local GPUs
|
||||
float maxBw = 0;
|
||||
int minHops = 0xfffffff;
|
||||
struct ncclTopoLinkList* paths = net->paths[GPU];
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
if (paths[g].bw > maxBw) {
|
||||
maxBw = paths[g].bw;
|
||||
minHops = paths[g].count;
|
||||
} else if (paths[g].bw == maxBw && paths[g].count > 0 && paths[g].count < minHops) {
|
||||
minHops = paths[g].count;
|
||||
}
|
||||
}
|
||||
if (maxBw >= bw) {
|
||||
for (int i=0; i<system->nodes[GPU].count; i++) {
|
||||
int g = (graph->nChannels+i)%system->nodes[GPU].count;
|
||||
if (paths[g].bw == maxBw && paths[g].count == minHops) {
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, 0, backToNet, backToFirstRank, 0, time, NET, n, g));
|
||||
}
|
||||
}
|
||||
int localGpus[NCCL_TOPO_MAX_NODES], localGpuCount, pathType;
|
||||
NCCLCHECK(ncclTopoGetLocal(system, NET, n, GPU, localGpus, &localGpuCount, &pathType));
|
||||
// if no GPUs are connected, skip this net
|
||||
if (pathType == PATH_DIS) continue;
|
||||
for (int g = 0; g < localGpuCount; ++g) {
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, 0, backToNet, backToFirstRank, 0, time, NET, n, localGpus[g]));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -813,6 +801,7 @@ struct kvDict kvDictLinkType[] = {
|
||||
{ "PIX", PATH_PIX },
|
||||
{ "PXB", PATH_PXB },
|
||||
{ "PXN", PATH_PXN },
|
||||
{ "P2C", PATH_P2C },
|
||||
{ "PHB", PATH_PHB },
|
||||
{ "SYS", PATH_SYS },
|
||||
{ NULL, 0 }
|
||||
@@ -980,8 +969,8 @@ float sm90SpeedArrayInter[] = { 48.0, 45.0, 42.0, 40.0, 30.0, 24.0, 22.0, 20.0,
|
||||
|
||||
RCCL_PARAM(ModelMatchingDisable, "MODEL_MATCHING_DISABLE", 0);
|
||||
|
||||
float sm100SpeedArrayIntra[] = { 90.0, 80.0, 70.0, 60.0, 50.0, 40.0, 30.0, 24.0, 20.0, 19.0 };
|
||||
float sm100SpeedArrayInter[] = { 48.0, 45.0, 42.0, 40.0, 30.0, 24.0, 22.0, 20.0, 17.5, 15.0, 12.0, 6.0, 3.0, 2.4, 1.2, 0.24, 0.12 };
|
||||
float sm100SpeedArrayIntra[] = { 90.0, 80.0, 70.0, 60.0, 50.0, 40.0, 30.0, 24.0, 20.0, 19.0, 18.0 };
|
||||
float sm100SpeedArrayInter[] = { 47.9, 45.0, 42.0, 40.0, 30.0, 24.0, 22.0, 20.0, 17.5, 15.0, 12.0, 6.0, 3.0, 2.4, 1.2, 0.24, 0.12 };
|
||||
#define NSPEEDSINTRA_SM100 (sizeof(sm100SpeedArrayIntra)/sizeof(float))
|
||||
#define NSPEEDSINTER_SM100 (sizeof(sm100SpeedArrayInter)/sizeof(float))
|
||||
|
||||
@@ -1168,13 +1157,13 @@ search:
|
||||
int maxIntra = system->nodes[NET].count > 0 ? tmpGraph.typeInter : maxTypeIntra;
|
||||
if (tmpGraph.typeIntra < maxIntra && (graph->nChannels == 0 || tmpGraph.typeIntra < graph->typeIntra)) {
|
||||
tmpGraph.typeIntra += 1;
|
||||
goto search;
|
||||
if (tmpGraph.typeIntra < PATH_DIS) goto search;
|
||||
}
|
||||
tmpGraph.typeIntra = minTypeIntra;
|
||||
|
||||
if (system->nodes[NET].count > 0 && tmpGraph.typeInter < maxTypeInter && (graph->nChannels == 0 || tmpGraph.typeInter < graph->typeInter || tmpGraph.typeInter < PATH_PXN)) {
|
||||
tmpGraph.typeInter += 1;
|
||||
goto search;
|
||||
if (tmpGraph.typeInter < PATH_DIS) goto search;
|
||||
}
|
||||
tmpGraph.typeInter = minTypeInter;
|
||||
|
||||
@@ -1232,7 +1221,7 @@ done:
|
||||
}
|
||||
|
||||
if (graph->nChannels == 0 && graph->collNet == 0 && graph->pattern != NCCL_TOPO_PATTERN_NVLS) {
|
||||
WARN("Could not find a path for pattern %d, falling back to simple order", graph->pattern);
|
||||
INFO(NCCL_GRAPH, "Could not find a path for pattern %d, falling back to simple order", graph->pattern);
|
||||
for (int i=0; i<ngpus; i++) graph->intra[i] = system->nodes[GPU].nodes[i].gpu.rank;
|
||||
graph->inter[0] = graph->inter[1] = 0;
|
||||
graph->bwIntra = graph->bwInter = 0.1;
|
||||
@@ -1366,7 +1355,7 @@ ncclResult_t ncclTopoGetNetDev(struct ncclComm* comm, int rank, struct ncclTopoG
|
||||
}
|
||||
if (pxnLevel == 1) {
|
||||
int g, n;
|
||||
NCCLCHECK(ncclTopoRankToIndex(comm->topo, rank, &g));
|
||||
NCCLCHECK(ncclTopoRankToIndex(comm->topo, rank, &g, /*showWarn=*/true));
|
||||
NCCLCHECK(ncclTopoIdToIndex(comm->topo, NET, netId, &n));
|
||||
struct ncclTopoNode* gpu = comm->topo->nodes[GPU].nodes+g;
|
||||
if (gpu->paths[NET][n].type <= PATH_PXN) {
|
||||
@@ -1378,7 +1367,7 @@ ncclResult_t ncclTopoGetNetDev(struct ncclComm* comm, int rank, struct ncclTopoG
|
||||
// Check which local GPU corresponds to that NIC and see if we can use PXN.
|
||||
int n, g1, g2;
|
||||
NCCLCHECK(ncclTopoIdToIndex(comm->topo, NET, netId, &n));
|
||||
NCCLCHECK(ncclTopoRankToIndex(comm->topo, rank, &g1));
|
||||
NCCLCHECK(ncclTopoRankToIndex(comm->topo, rank, &g1, /*showWarn=*/true));
|
||||
NCCLCHECK(ncclTopoGetLocalGpu(comm->topo, netId, &g2));
|
||||
if (g2 != -1) {
|
||||
struct ncclTopoNode* peerGpu = comm->topo->nodes[GPU].nodes+g2;
|
||||
|
||||
+86
-52
@@ -10,12 +10,10 @@
|
||||
#include "topo.h"
|
||||
#include "comm.h"
|
||||
#include "nvmlwrap.h"
|
||||
#include "net.h"
|
||||
#include "coll_net.h"
|
||||
#include "transport.h"
|
||||
#include <sys/stat.h>
|
||||
#include <fcntl.h>
|
||||
#include "xml.h"
|
||||
#include "cpuset.h"
|
||||
#include "bootstrap.h"
|
||||
|
||||
@@ -24,11 +22,11 @@
|
||||
|
||||
const char* topoNodeTypeStr[] = { "GPU", "PCI", "NVS", "CPU", "NIC", "NET" };
|
||||
#if defined(__HIP_PLATFORM_AMD__) || defined(__HIPCC__)
|
||||
const char* topoLinkTypeStr[] = { "LOC", "XGMI", "", "C2C", "PCI", "", "", "", "SYS", "NET" };
|
||||
const char* topoPathTypeStr[] = { "LOC", "XGMI", "NVB", "C2C", "PIX", "PXB", "PXN", "PHB", "SYS", "DIS" };
|
||||
const char* topoLinkTypeStr[] = { "LOC", "XGMI", "", "C2C", "PCI", "", "", "", "", "SYS", "NET" };
|
||||
const char* topoPathTypeStr[] = { "LOC", "XGMI", "NVB", "C2C", "PIX", "PXB", "PXN", "P2C", "PHB", "SYS", "NET", "DIS" };
|
||||
#else
|
||||
const char* topoLinkTypeStr[] = { "LOC", "NVL", "", "C2C", "PCI", "", "", "", "SYS", "NET" };
|
||||
const char* topoPathTypeStr[] = { "LOC", "NVL", "NVB", "C2C", "PIX", "PXB", "PXN", "PHB", "SYS", "NET", "DIS" };
|
||||
const char* topoLinkTypeStr[] = { "LOC", "NVL", "", "C2C", "PCI", "", "", "", "", "SYS", "NET" };
|
||||
const char* topoPathTypeStr[] = { "LOC", "NVL", "NVB", "C2C", "PIX", "PXB", "PXN", "P2C", "PHB", "SYS", "NET", "DIS" };
|
||||
#endif
|
||||
|
||||
/******************************************************************/
|
||||
@@ -257,7 +255,7 @@ ncclResult_t ncclTopoFlattenBcmSwitches(struct ncclTopoSystem* system) {
|
||||
pciSwitch->pci.device |= 0xffff;
|
||||
free(subSwIds);
|
||||
// Restart, as system->nodes[PCI].nodes has changed.
|
||||
s = 0;
|
||||
s = -1; // Will be incremented to 0 in the next loop iteration
|
||||
continue;
|
||||
fail:
|
||||
free(subSwIds);
|
||||
@@ -427,7 +425,9 @@ ncclResult_t ncclTopoAddGpu(struct ncclXmlNode* xmlGpu, struct ncclTopoSystem* s
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
struct kvDict kvDictPciClass[] = { { "0x060400", PCI }, { "0x068000", NVS }, { "0x068001", CPU }, { "0x03", GPU }, { "0x02", NIC }, { "0x120000", GPU }, { NULL, PCI /* Default fallback value */ } };
|
||||
#define PCI_BRIDGE_DEVICE_CLASS "0x060400"
|
||||
|
||||
struct kvDict kvDictPciClass[] = { { PCI_BRIDGE_DEVICE_CLASS, PCI }, { "0x068000", NVS }, { "0x068001", CPU }, { "0x03", GPU }, { "0x02", NIC }, { "0x120000", GPU }, { NULL, PCI /* Default fallback value */ } };
|
||||
struct kvDict kvDictPciGen[] = {
|
||||
{ "2.5 GT/s", 15 }, { "5 GT/s", 30 }, { "8 GT/s", 60 }, { "16 GT/s", 120 }, { "32 GT/s", 240 }, /* Kernel 5.6 and earlier */
|
||||
{ "2.5 GT/s PCIe", 15 }, { "5.0 GT/s PCIe", 30 }, { "8.0 GT/s PCIe", 60 }, { "16.0 GT/s PCIe", 120 }, { "32.0 GT/s PCIe", 240 }, { "64.0 GT/s PCIe", 480 },
|
||||
@@ -786,6 +786,7 @@ static ncclResult_t xmlInitAttrInt(struct ncclXmlNode* node, const char* attrNam
|
||||
if (index == -1) {
|
||||
index = node->nAttrs++;
|
||||
strncpy(node->attrs[index].key, attrName, MAX_STR_LEN);
|
||||
node->attrs[index].key[MAX_STR_LEN] = '\0';
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "%d", value);
|
||||
}
|
||||
return ncclSuccess;
|
||||
@@ -796,6 +797,7 @@ static ncclResult_t xmlInitAttrUint64(struct ncclXmlNode* node, const char* attr
|
||||
if (index == -1) {
|
||||
index = node->nAttrs++;
|
||||
strncpy(node->attrs[index].key, attrName, MAX_STR_LEN);
|
||||
node->attrs[index].key[MAX_STR_LEN] = '\0';
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "0x%lx", value);
|
||||
}
|
||||
return ncclSuccess;
|
||||
@@ -806,6 +808,7 @@ static ncclResult_t xmlInitAttrFloat(struct ncclXmlNode* node, const char* attrN
|
||||
if (index == -1) {
|
||||
index = node->nAttrs++;
|
||||
strncpy(node->attrs[index].key, attrName, MAX_STR_LEN);
|
||||
node->attrs[index].key[MAX_STR_LEN] = '\0';
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "%f", value);
|
||||
}
|
||||
return ncclSuccess;
|
||||
@@ -886,6 +889,17 @@ typedef struct xmlNodeStack {
|
||||
|
||||
} xmlNodeStack;
|
||||
|
||||
ncclResult_t ncclFindFirstPciParent(ncclXmlNode** parent) {
|
||||
ncclXmlNode* newParent = *parent;
|
||||
while (strcmp(newParent->name, "pci") != 0) {
|
||||
newParent = newParent->parent;
|
||||
if (newParent == nullptr) return ncclSuccess;
|
||||
if (strcmp(newParent->name, "system") == 0) return ncclSuccess;
|
||||
}
|
||||
*parent = newParent;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// 1. Find the common parent xmlNode between the given set of nodes
|
||||
ncclResult_t ncclTopoGetPath(ncclXmlNode** nodes, int nNodes, int* path, ncclXmlNode** parent) {
|
||||
// Track a stack of parents per-net node being merged
|
||||
@@ -984,6 +998,7 @@ ncclResult_t ncclTopoGetPath(ncclXmlNode** nodes, int nNodes, int* path, ncclXml
|
||||
}
|
||||
|
||||
out:
|
||||
ncclFindFirstPciParent(&common);
|
||||
*parent = common;
|
||||
free(parents);
|
||||
return ncclSuccess;
|
||||
@@ -1047,13 +1062,19 @@ ncclResult_t ncclTopoMakePciParent(struct ncclXml* xml, struct ncclXmlNode** par
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoMakeVnic(ncclComm_t comm, struct ncclXml* xml, ncclNetVDeviceProps_t* vProps,
|
||||
struct ncclXmlNode** physNetNodes, struct ncclXmlNode** netNode, ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*)) {
|
||||
ncclResult_t ncclTopoMakeVnic(struct ncclXml* xml, ncclNetVDeviceProps_t* vProps,
|
||||
struct ncclXmlNode** physNetNodes, ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*)) {
|
||||
if (vProps->ndevs > NCCL_NET_MAX_DEVS_PER_NIC) {
|
||||
WARN("TOPO/NET : Tried to merge too many NICs. %d > %d", vProps->ndevs, NCCL_NET_MAX_DEVS_PER_NIC);
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
// Don't make vNics of size 1
|
||||
if (vProps->ndevs == 1) {
|
||||
TRACE(NCCL_GRAPH, "TOPO/NET : Skipping vNic of size 1");
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Trigger the merge, then get the new device's properties
|
||||
int vDevIndex = 0;
|
||||
ncclResult_t ret = makeVDevice(&vDevIndex, vProps);
|
||||
@@ -1063,11 +1084,18 @@ struct ncclXmlNode** physNetNodes, struct ncclXmlNode** netNode, ncclResult_t (*
|
||||
return ret;
|
||||
}
|
||||
|
||||
// Mark original NICs as keep="0" in the topology
|
||||
for (int i = 0; i < vProps->ndevs; i++) {
|
||||
int dev = vProps->devs[i];
|
||||
struct ncclXmlNode* netNode = physNetNodes[dev];
|
||||
NCCLCHECK(xmlSetAttrInt(netNode, "keep", 0));
|
||||
}
|
||||
|
||||
INFO(NCCL_GRAPH, "TOPO/NET : Made vNic %d", vDevIndex);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoForceMerge(ncclComm_t comm, struct ncclXml* xml, const char* str, int* placedDevs, ncclNetProperties_t* propsList, struct ncclXmlNode** physNetNodes, int nPhysDevs, ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*)) {
|
||||
ncclResult_t ncclTopoForceMerge(struct ncclXml* xml, char* str, int* placedDevs, ncclNetProperties_t* propsList, struct ncclXmlNode** physNetNodes, int nPhysDevs, ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*)) {
|
||||
ncclResult_t ret = ncclSuccess;
|
||||
INFO(NCCL_ENV|NCCL_NET, "TOPO/NET : Force-fusing NICs using NCCL_NET_FORCE_MERGE=%s", str);
|
||||
char* ncStr;
|
||||
@@ -1105,8 +1133,7 @@ ncclResult_t ncclTopoForceMerge(ncclComm_t comm, struct ncclXml* xml, const char
|
||||
goto fail;
|
||||
}
|
||||
|
||||
struct ncclXmlNode* netNode;
|
||||
ret = ncclTopoMakeVnic(comm, xml, &vProps, physNetNodes, &netNode, makeVDevice);
|
||||
ret = ncclTopoMakeVnic(xml, &vProps, physNetNodes, makeVDevice);
|
||||
if (ret == ncclSuccess) {
|
||||
// Only set that a device is "placed" after successfully making a vNic (it's possible to exit before this)
|
||||
for (int i = 0; i < vProps.ndevs; i++) {
|
||||
@@ -1128,7 +1155,7 @@ fail:
|
||||
goto exit;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoAutoMerge(ncclComm_t comm, struct ncclXml* xml, int mergeLevel, int* placedDevs, ncclNetProperties_t* propsList, struct ncclXmlNode** physNetNodes, int nPhysDevs, ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*)) {
|
||||
ncclResult_t ncclTopoAutoMerge(struct ncclXml* xml, int mergeLevel, int* placedDevs, ncclNetProperties_t* propsList, struct ncclXmlNode** physNetNodes, int nPhysDevs, ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*)) {
|
||||
// Compute the path type between each device
|
||||
int* paths = NULL;
|
||||
ncclResult_t res = ncclSuccess;
|
||||
@@ -1172,8 +1199,7 @@ ncclResult_t ncclTopoAutoMerge(ncclComm_t comm, struct ncclXml* xml, int mergeLe
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
struct ncclXmlNode* netNode;
|
||||
ncclResult_t ret = ncclTopoMakeVnic(comm, xml, &vProps, physNetNodes, &netNode, makeVDevice);
|
||||
ncclResult_t ret = ncclTopoMakeVnic(xml, &vProps, physNetNodes, makeVDevice);
|
||||
|
||||
// Merging failed.
|
||||
// Mark all as unplaced and increase their distance to disconnected (PATH_DIS)
|
||||
@@ -1205,6 +1231,7 @@ struct kvDict nicPathKvList[] = {
|
||||
{ "PIX", PATH_PIX },
|
||||
{ "PXB", PATH_PXB },
|
||||
{ "PXN", PATH_PXN },
|
||||
{ "P2C", PATH_P2C },
|
||||
{ "PHB", PATH_PHB },
|
||||
{ "SYS", PATH_SYS },
|
||||
{ NULL, 0 }
|
||||
@@ -1226,14 +1253,19 @@ ncclResult_t ncclTopoGetVNicParent(struct ncclXml* xml, ncclResult_t (*getProper
|
||||
if (path == PATH_LOC) {
|
||||
*parent = NULL;
|
||||
} else if (parent && strcmp((*parent)->name, "pci") == 0) {
|
||||
// If the common parent is PCI, we must reparent the new NIC under a made up busId
|
||||
NCCLCHECK(ncclTopoMakePciParent(xml, parent, physNetNodes[0]));
|
||||
// Compare PCI class here to avoid NCCL WARN when the "class" attribute doesn't exist
|
||||
const char* c;
|
||||
NCCLCHECK(xmlGetAttrStr(*parent, "class", &c));
|
||||
if (strcmp(c, PCI_BRIDGE_DEVICE_CLASS) == 0) {
|
||||
// If the common parent is a PCI switch, we must reparent the new NIC under a made up pci device with a unique busid
|
||||
NCCLCHECK(ncclTopoMakePciParent(xml, parent, physNetNodes[0]));
|
||||
}
|
||||
}
|
||||
TRACE(NCCL_GRAPH, "Selected parent %s with path %d", (*parent)->name, path);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoMakeVNics(ncclComm_t comm, struct ncclXml* xml, ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*), ncclResult_t (*getProperties)(int, ncclNetProperties_t*), int physicalDevs) {
|
||||
ncclResult_t ncclTopoMakeVNics(struct ncclXml* xml, ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*), ncclResult_t (*getProperties)(int, ncclNetProperties_t*), int physicalDevs) {
|
||||
int* placedDevs = NULL;
|
||||
struct ncclXmlNode** physNetNodes = NULL;
|
||||
if (physicalDevs == 0) return ncclSuccess;
|
||||
@@ -1257,15 +1289,15 @@ ncclResult_t ncclTopoMakeVNics(ncclComm_t comm, struct ncclXml* xml, ncclResult_
|
||||
{ // Avoids warnings related to jumping to "out"
|
||||
const char* mergeLevelEnv = ncclGetEnv("NCCL_NET_MERGE_LEVEL");
|
||||
if (mergeLevelEnv) kvConvertToInt(mergeLevelEnv, &mergeLevel, nicPathKvList);
|
||||
const char* forceMerge = ncclGetEnv("NCCL_NET_FORCE_MERGE");
|
||||
char* forceMerge = (char*) ncclGetEnv("NCCL_NET_FORCE_MERGE");
|
||||
NCCLCHECK(ncclCalloc(&placedDevs, physicalDevs));
|
||||
memset(placedDevs, 0, sizeof(int)*physicalDevs);
|
||||
|
||||
if (forceMerge) {
|
||||
NCCLCHECKGOTO(ncclTopoForceMerge(comm, xml, forceMerge, placedDevs, props, physNetNodes, physicalDevs, makeVDevice), res, out);
|
||||
NCCLCHECKGOTO(ncclTopoForceMerge(xml, forceMerge, placedDevs, props, physNetNodes, physicalDevs, makeVDevice), res, out);
|
||||
}
|
||||
}
|
||||
NCCLCHECKGOTO(ncclTopoAutoMerge(comm, xml, mergeLevel, placedDevs, props, physNetNodes, physicalDevs, makeVDevice), res, out);
|
||||
NCCLCHECKGOTO(ncclTopoAutoMerge(xml, mergeLevel, placedDevs, props, physNetNodes, physicalDevs, makeVDevice), res, out);
|
||||
|
||||
out:
|
||||
free(physNetNodes);
|
||||
@@ -1274,7 +1306,7 @@ out:
|
||||
return res;
|
||||
}
|
||||
|
||||
static ncclResult_t ncclTopoPopulateNics(ncclComm_t comm, ncclXml* xml, int startIndex, int endIndex, ncclResult_t (*getProperties)(int, ncclNetProperties_t*), const char* netName, int coll, int keep, int virtualNics) {
|
||||
static ncclResult_t ncclTopoPopulateNics(ncclXml* xml, int startIndex, int endIndex, ncclResult_t (*getProperties)(int, ncclNetProperties_t*), const char* netName, int coll, int virtualNics, bool dmaBufSupport) {
|
||||
for (int n = startIndex; n < endIndex; n++) {
|
||||
ncclNetProperties_t props;
|
||||
NCCLCHECK(getProperties(n, &props));
|
||||
@@ -1293,15 +1325,17 @@ static ncclResult_t ncclTopoPopulateNics(ncclComm_t comm, ncclXml* xml, int star
|
||||
const char* colAttr;
|
||||
NCCLCHECK(xmlGetAttr(netNode, "coll", &colAttr));
|
||||
|
||||
// If coll == 0 but the netNode is tagged as coll, don't update the keep value
|
||||
if (colAttr == NULL || coll != 0 || strcmp(colAttr,"1") != 0) NCCLCHECK(xmlSetAttrInt(netNode, "keep", keep));
|
||||
NCCLCHECK(xmlSetAttrInt(netNode, "keep", 1));
|
||||
int dev;
|
||||
xmlGetAttrIntDefault(netNode, "dev", &dev, -1);
|
||||
if (dev != -1 && dev != n) INFO(NCCL_GRAPH, "TOPO/NET : Changing %s dev index from %d to %d", netName, dev, n);
|
||||
NCCLCHECK(xmlSetAttrInt(netNode, "dev", n));
|
||||
NCCLCHECK(xmlInitAttrInt(netNode, "latency", props.latency));
|
||||
NCCLCHECK(xmlInitAttrInt(netNode, "speed", props.speed));
|
||||
NCCLCHECK(xmlInitAttrInt(netNode, "port", props.port));
|
||||
NCCLCHECK(xmlInitAttrUint64(netNode, "guid", props.guid));
|
||||
NCCLCHECK(xmlInitAttrInt(netNode, "maxconn", props.maxComms));
|
||||
bool gdrSupport = (props.ptrSupport & NCCL_PTR_CUDA) || (comm->dmaBufSupport && (props.ptrSupport & NCCL_PTR_DMABUF));
|
||||
bool gdrSupport = (props.ptrSupport & NCCL_PTR_CUDA) || (dmaBufSupport && (props.ptrSupport & NCCL_PTR_DMABUF));
|
||||
INFO(NCCL_NET,"NET/%s : GPU Direct RDMA %s for HCA %d '%s'", netName, gdrSupport ? "Enabled" : "Disabled", n, props.name);
|
||||
NCCLCHECK(xmlInitAttrInt(netNode, "gdr", gdrSupport));
|
||||
// Only set coll if it's not 0
|
||||
@@ -1317,30 +1351,22 @@ static ncclResult_t ncclTopoPopulateNics(ncclComm_t comm, ncclXml* xml, int star
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
struct ncclTopoNetState {
|
||||
int nVirtualNics;
|
||||
int nPhysicalNics;
|
||||
const char* name;
|
||||
};
|
||||
|
||||
// Calls to network plugin APIs should be protected. This function should be called inside a per-process lock.
|
||||
static ncclResult_t ncclTopoProcessNet(ncclComm_t comm, ncclXml* xml, int coll, const char* dumpXmlFile, ncclTopoNetState* state, ncclResult_t (*getProperties)(int, ncclNetProperties_t*), ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*), ncclResult_t (*devices)(int*), const char* netName) {
|
||||
ncclResult_t ncclTopoProcessNet(ncclXml* xml, int coll, const char* dumpXmlFile, ncclTopoNetState* state, ncclResult_t (*getProperties)(int, ncclNetProperties_t*), ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*), ncclResult_t (*devices)(int*), const char* netName, bool dmaBufSupport) {
|
||||
int usePhysicalDevices = (dumpXmlFile || makeVDevice == NULL);
|
||||
if (state->nPhysicalNics == -1) NCCLCHECK(devices(&state->nPhysicalNics));
|
||||
// Enumerate physical devices
|
||||
NCCLCHECK(ncclTopoPopulateNics(comm, xml, 0, state->nPhysicalNics, getProperties, netName, coll, 1, 0));
|
||||
NCCLCHECK(ncclTopoPopulateNics(xml, 0, state->nPhysicalNics, getProperties, netName, coll, false, dmaBufSupport));
|
||||
if (!usePhysicalDevices) {
|
||||
if (state->nVirtualNics == -1) {
|
||||
NCCLCHECK(ncclTopoMakeVNics(comm, xml, makeVDevice, getProperties, state->nPhysicalNics));
|
||||
NCCLCHECK(ncclTopoMakeVNics(xml, makeVDevice, getProperties, state->nPhysicalNics));
|
||||
int nDevs;
|
||||
NCCLCHECK(devices(&nDevs));
|
||||
state->nVirtualNics = nDevs - state->nPhysicalNics;
|
||||
}
|
||||
// Remove keep=1 for physical collnets
|
||||
if (state->nVirtualNics > 0) {
|
||||
NCCLCHECK(ncclTopoPopulateNics(comm, xml, 0, state->nPhysicalNics, getProperties, netName, coll, 0, 0));
|
||||
// Populate new devices
|
||||
NCCLCHECK(ncclTopoPopulateNics(comm, xml, state->nPhysicalNics, state->nPhysicalNics+state->nVirtualNics, getProperties, netName, coll, 1, 1));
|
||||
NCCLCHECK(ncclTopoPopulateNics(xml, state->nPhysicalNics, state->nPhysicalNics+state->nVirtualNics, getProperties, netName, coll, true, dmaBufSupport));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1388,6 +1414,15 @@ ncclResult_t ncclTopoGetSystem(struct ncclComm* comm, struct ncclTopoSystem** sy
|
||||
// Try default XML topology location
|
||||
NCCLCHECKGOTO(ncclTopoGetXmlFromFile("/var/run/nvidia-topologyd/virtualTopology.xml", xml, 0), ret, fail);
|
||||
}
|
||||
// Fixup the cpu's host_hashes.
|
||||
struct ncclXmlNode* node;
|
||||
// Update every cpu node's host_hash attribute since those are not
|
||||
// intended to be preserved from the XML files that have been read.
|
||||
NCCLCHECKGOTO(xmlFindTag(xml, "cpu", &node), ret, fail);
|
||||
while (node != nullptr) {
|
||||
NCCLCHECKGOTO(xmlSetAttrLong(node, "host_hash", getHostHash()), ret, fail);
|
||||
NCCLCHECKGOTO(xmlFindNextTag(xml, "cpu", node, &node), ret, fail);
|
||||
}
|
||||
if (xml->maxIndex == 0) {
|
||||
// Create top tag
|
||||
struct ncclXmlNode* top;
|
||||
@@ -1400,7 +1435,6 @@ ncclResult_t ncclTopoGetSystem(struct ncclComm* comm, struct ncclTopoSystem** sy
|
||||
// Detect only the GPU managed by this process. We'll get any others through XML fusion.
|
||||
char busId[NVML_DEVICE_PCI_BUS_ID_BUFFER_SIZE];
|
||||
NCCLCHECKGOTO(int64ToBusId(comm->peerInfo[comm->rank].busId, busId), ret, fail);
|
||||
struct ncclXmlNode* node;
|
||||
NCCLCHECKGOTO(ncclTopoFillGpu(xml, busId, &node), ret, fail);
|
||||
if (node) {
|
||||
NCCLCHECKGOTO(xmlSetAttrInt(node, "keep", 1), ret, fail);
|
||||
@@ -1417,13 +1451,13 @@ ncclResult_t ncclTopoGetSystem(struct ncclComm* comm, struct ncclTopoSystem** sy
|
||||
state = NULL;
|
||||
if (collNetSupport(comm)) {
|
||||
NCCLCHECKGOTO(ncclTopoGetSharedState(&state, comm->ncclCollNet->name, collNetStates), ret, fail);
|
||||
NCCLCHECKGOTO(ncclTopoProcessNet(comm, xml, 1, dumpXmlFile, state,
|
||||
comm->ncclCollNet->getProperties, comm->ncclCollNet->makeVDevice, comm->ncclCollNet->devices, comm->ncclCollNet->name), ret, fail);
|
||||
NCCLCHECKGOTO(ncclTopoProcessNet(xml, 1, dumpXmlFile, state,
|
||||
comm->ncclCollNet->getProperties, comm->ncclCollNet->makeVDevice, comm->ncclCollNet->devices, comm->ncclCollNet->name, comm->dmaBufSupport), ret, fail);
|
||||
}
|
||||
NCCLCHECKGOTO(ncclTopoGetSharedState(&state, comm->ncclNet->name, netStates), ret, fail);
|
||||
// [RCCL] Disabled virtual devices
|
||||
NCCLCHECKGOTO(ncclTopoProcessNet(comm, xml, 0, dumpXmlFile, state,
|
||||
comm->ncclNet->getProperties, nullptr /*comm->ncclNet->makeVDevice*/, comm->ncclNet->devices, comm->ncclNet->name), ret, fail);
|
||||
NCCLCHECKGOTO(ncclTopoProcessNet(xml, 0, dumpXmlFile, state,
|
||||
comm->ncclNet->getProperties, nullptr /*comm->ncclNet->makeVDevice*/, comm->ncclNet->devices, comm->ncclNet->name, comm->dmaBufSupport), ret, fail);
|
||||
pthread_mutex_unlock(&netLock);
|
||||
netLockHeld = 0;
|
||||
|
||||
@@ -1487,7 +1521,7 @@ fail:
|
||||
goto exit;
|
||||
}
|
||||
|
||||
static ncclResult_t ncclTopoGetLocal(struct ncclTopoSystem* system, int type, int index, int resultType,
|
||||
ncclResult_t ncclTopoGetLocal(struct ncclTopoSystem* system, int type, int index, int resultType,
|
||||
int locals[NCCL_TOPO_MAX_NODES], int* localCount, int* pathType) {
|
||||
int minType = PATH_DIS;
|
||||
float maxBw = 0;
|
||||
@@ -1540,7 +1574,7 @@ ncclResult_t getLocalNetCountByBw(struct ncclTopoSystem* system, int gpu, int *c
|
||||
|
||||
ncclResult_t ncclTopoGetLocalNet(struct ncclTopoSystem* system, int rank, int channelId, int64_t* id, int* dev) {
|
||||
int gpu;
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &gpu));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &gpu, /*showWarn=*/true));
|
||||
|
||||
int localNets[NCCL_TOPO_MAX_NODES];
|
||||
int localNetCount;
|
||||
@@ -1607,7 +1641,7 @@ NCCL_PARAM(IgnoreCpuAffinity, "IGNORE_CPU_AFFINITY", 0);
|
||||
ncclResult_t ncclTopoGetCpuAffinity(struct ncclTopoSystem* system, int rank, cpu_set_t* affinity) {
|
||||
struct ncclTopoNode* cpu = NULL, *gpu = NULL;
|
||||
int gpuIndex, cpuIndex;
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &gpuIndex));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &gpuIndex, /*showWarn=*/true));
|
||||
NCCLCHECK(ncclGetLocalCpu(system, gpuIndex, &cpuIndex));
|
||||
gpu = system->nodes[GPU].nodes+gpuIndex;
|
||||
cpu = system->nodes[CPU].nodes+cpuIndex;
|
||||
@@ -1619,8 +1653,8 @@ ncclResult_t ncclTopoGetCpuAffinity(struct ncclTopoSystem* system, int rank, cpu
|
||||
#ifdef ENABLE_TRACE
|
||||
{
|
||||
char affinityStr[sizeof(cpu_set_t)*2];
|
||||
NCCLCHECK(ncclCpusetToStr(&mask, affinityStr));
|
||||
TRACE(NCCL_INIT, "Current affinity for GPU %d is %s", gpu->gpu.dev, affinityStr);
|
||||
TRACE(NCCL_INIT, "Current affinity for GPU %d is %s", gpu->gpu.dev,
|
||||
ncclCpusetToRangeStr(&mask, affinityStr, sizeof(affinityStr)));
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1630,8 +1664,8 @@ ncclResult_t ncclTopoGetCpuAffinity(struct ncclTopoSystem* system, int rank, cpu
|
||||
#ifdef ENABLE_TRACE
|
||||
{
|
||||
char affinityStr[sizeof(cpu_set_t)*2];
|
||||
NCCLCHECK(ncclCpusetToStr(&cpuMask, affinityStr));
|
||||
TRACE(NCCL_INIT, "CPU GPU affinity for GPU %d is %s", gpu->gpu.dev, affinityStr);
|
||||
TRACE(NCCL_INIT, "CPU GPU affinity for GPU %d is %s", gpu->gpu.dev,
|
||||
ncclCpusetToRangeStr(&cpuMask, affinityStr, sizeof(affinityStr)));
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1648,8 +1682,8 @@ ncclResult_t ncclTopoGetCpuAffinity(struct ncclTopoSystem* system, int rank, cpu
|
||||
// If there is a non empty set, use it to set affinity
|
||||
if (CPU_COUNT(&finalMask)) {
|
||||
char affinityStr[sizeof(cpu_set_t)*2];
|
||||
NCCLCHECK(ncclCpusetToStr(&finalMask, affinityStr));
|
||||
INFO(NCCL_INIT, "Setting affinity for GPU %d to %s", gpu->gpu.dev, affinityStr);
|
||||
INFO(NCCL_INIT, "Setting affinity for GPU %d to %s", gpu->gpu.dev,
|
||||
ncclCpusetToRangeStr(&finalMask, affinityStr, sizeof(affinityStr)));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
+22
-8
@@ -10,6 +10,8 @@
|
||||
|
||||
#include "graph.h"
|
||||
#include "core.h"
|
||||
#include "xml.h"
|
||||
#include "net.h"
|
||||
#include "archinfo.h"
|
||||
#include <string.h>
|
||||
|
||||
@@ -57,9 +59,10 @@ extern const char* topoNodeTypeStr[];
|
||||
#define LINK_PCI 4
|
||||
// Skipping 5 for PATH_PXB
|
||||
// Skipping 6 for PATH_PXN
|
||||
// Skipping 7 for PATH_PHB
|
||||
#define LINK_SYS 8
|
||||
#define LINK_NET 9
|
||||
// Skipping 7 for PATH_P2C
|
||||
// Skipping 8 for PATH_PHB
|
||||
#define LINK_SYS 9
|
||||
#define LINK_NET 10
|
||||
extern const char* topoLinkTypeStr[];
|
||||
|
||||
// Local (myself)
|
||||
@@ -83,20 +86,23 @@ extern const char* topoLinkTypeStr[];
|
||||
// Connection between a GPU and a NIC using an intermediate GPU. Used to enable rail-local, aggregated network send/recv operations.
|
||||
#define PATH_PXN 6
|
||||
|
||||
// Connection between a GPU and a NIC using the C2C connection to the CPU and the PCIe connection to the NIC
|
||||
#define PATH_P2C 7
|
||||
|
||||
// Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)
|
||||
#define PATH_PHB 7
|
||||
#define PATH_PHB 8
|
||||
|
||||
// Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)
|
||||
#define PATH_SYS 8
|
||||
#define PATH_SYS 9
|
||||
|
||||
// Connection through the network
|
||||
#define PATH_NET 9
|
||||
#define PATH_NET 10
|
||||
|
||||
// New type of path which should precede PATH_PIX
|
||||
#define PATH_PORT PATH_NVL
|
||||
|
||||
// Disconnected
|
||||
#define PATH_DIS 10
|
||||
#define PATH_DIS 11
|
||||
extern const char* topoPathTypeStr[];
|
||||
|
||||
struct ncclTopoNode;
|
||||
@@ -217,6 +223,13 @@ ncclResult_t ncclTopoGetGpuMinPath(struct ncclTopoSystem* system, int type, int*
|
||||
ncclResult_t ncclTopoGetGpuMaxPath(struct ncclTopoSystem* system, int type, int* max);
|
||||
ncclResult_t ncclTopoSplitNvLink(struct ncclTopoSystem* system, int* splitNvLink);
|
||||
|
||||
struct ncclTopoNetState {
|
||||
int nVirtualNics;
|
||||
int nPhysicalNics;
|
||||
const char* name;
|
||||
};
|
||||
ncclResult_t ncclTopoProcessNet(ncclXml* xml, int coll, const char* dumpXmlFile, ncclTopoNetState* state, ncclResult_t (*getProperties)(int, ncclNetProperties_t*), ncclResult_t (*makeVDevice)(int*, ncclNetVDeviceProps_t*), ncclResult_t (*devices)(int*), const char* netName, bool dmaBufSupport);
|
||||
|
||||
#define NCCL_TOPO_XML_MAX_NODES 8192
|
||||
#define NCCL_GRAPH_XML_MAX_NODES 8192
|
||||
ncclResult_t ncclTopoGetSystemFromXml(struct ncclXml* xml, struct ncclTopoSystem** topoSystem, uint64_t localHostHash);
|
||||
@@ -236,7 +249,7 @@ static ncclResult_t ncclTopoIdToIndex(struct ncclTopoSystem* system, int type, i
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
static ncclResult_t ncclTopoRankToIndex(struct ncclTopoSystem* system, int rank, int* index) {
|
||||
static ncclResult_t ncclTopoRankToIndex(struct ncclTopoSystem* system, int rank, int* index, bool showWarn) {
|
||||
*index = -1;
|
||||
for (int i=0; i<system->nodes[GPU].count; i++) {
|
||||
if (system->nodes[GPU].nodes[i].gpu.rank == rank) {
|
||||
@@ -244,6 +257,7 @@ static ncclResult_t ncclTopoRankToIndex(struct ncclTopoSystem* system, int rank,
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
if (showWarn) WARN("ncclTopoRankToIndex could not find rank %d", rank);
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
|
||||
+57
-34
@@ -18,13 +18,13 @@ static int getNthreads(const char* name, int env, int min, int max, int def, int
|
||||
int nt = env;
|
||||
if (nt > 0) {
|
||||
if (nt % WarpSize != 0) {
|
||||
WARN("Invalid %s %d (must be a multiple of %d)", name, nt, WarpSize);
|
||||
INFO(NCCL_GRAPH|NCCL_ENV, "Invalid %s %d (must be a multiple of %d)", name, nt, WarpSize);
|
||||
nt = max;
|
||||
} else if (nt > max) {
|
||||
WARN("Invalid %s %d (maximum %d).", name, nt, max);
|
||||
INFO(NCCL_GRAPH|NCCL_ENV, "Invalid %s %d (maximum %d).", name, nt, max);
|
||||
nt = max;
|
||||
} else if (nt < min) {
|
||||
WARN("Invalid %s %d (minimum %d).", name, nt, min);
|
||||
INFO(NCCL_GRAPH|NCCL_ENV, "Invalid %s %d (minimum %d).", name, nt, min);
|
||||
nt = min;
|
||||
}
|
||||
} else {
|
||||
@@ -53,11 +53,14 @@ static int getNthreads(const char* name, int env, int min, int max, int def, int
|
||||
// NCCL_PROTO="^LL128;allreduce:LL128"
|
||||
// Enable everything but LL128, but only LL128 for allreduce.
|
||||
ncclResult_t parseList(const char* str, const char* prefixElems[], int nprefixes, const char* elems[], int nelems, int* list) {
|
||||
ncclResult_t ret = ncclSuccess;
|
||||
char* fullStr = strdup(str);
|
||||
char* tmpFullStr;
|
||||
char* fullToken = strtok_r(fullStr, ";", &tmpFullStr);
|
||||
char* subToken = nullptr;
|
||||
char* tokStr = nullptr;
|
||||
while (fullToken) {
|
||||
char* subToken = strdup(fullToken);
|
||||
subToken = strdup(fullToken);
|
||||
char* tmpSubStr;
|
||||
char* prefix = strtok_r(subToken, ":", &tmpSubStr);
|
||||
char* elemList = strtok_r(NULL, ":", &tmpSubStr);
|
||||
@@ -67,7 +70,8 @@ ncclResult_t parseList(const char* str, const char* prefixElems[], int nprefixes
|
||||
// because then all the prefixes before the prefix-less entry would be
|
||||
// overwritten.
|
||||
WARN("All entries except the first must have a prefix: \"%s\"", str);
|
||||
return ncclInvalidUsage;
|
||||
ret = ncclInvalidUsage;
|
||||
goto fail;
|
||||
}
|
||||
elemList = prefix;
|
||||
prefix = NULL;
|
||||
@@ -86,7 +90,7 @@ ncclResult_t parseList(const char* str, const char* prefixElems[], int nprefixes
|
||||
foundPrefix = true;
|
||||
for (int e=0; e<nelems; e++) list[p*nelems+e] = unset;
|
||||
|
||||
char* tokStr = strdup(elemList);
|
||||
tokStr = strdup(elemList);
|
||||
char* tmpStr;
|
||||
char* elem = strtok_r(tokStr, ",", &tmpStr);
|
||||
while (elem) {
|
||||
@@ -99,22 +103,32 @@ ncclResult_t parseList(const char* str, const char* prefixElems[], int nprefixes
|
||||
}
|
||||
if (e==nelems) {
|
||||
WARN("Unrecognized element token \"%s\" when parsing \"%s\"", elem, str);
|
||||
return ncclInvalidUsage;
|
||||
ret = ncclInvalidUsage;
|
||||
goto fail;
|
||||
}
|
||||
elem = strtok_r(NULL, ",", &tmpStr);
|
||||
}
|
||||
free(tokStr);
|
||||
tokStr = nullptr;
|
||||
}
|
||||
if (!foundPrefix) {
|
||||
WARN("Unrecognized prefix token \"%s\" when parsing \"%s\"", prefix, str);
|
||||
return ncclInvalidUsage;
|
||||
ret = ncclInvalidUsage;
|
||||
goto fail;
|
||||
}
|
||||
free(subToken);
|
||||
subToken = nullptr;
|
||||
|
||||
fullToken = strtok_r(NULL, ";", &tmpFullStr);
|
||||
}
|
||||
|
||||
exit:
|
||||
free(tokStr);
|
||||
free(subToken);
|
||||
free(fullStr);
|
||||
return ncclSuccess;
|
||||
return ret;
|
||||
fail:
|
||||
goto exit;
|
||||
}
|
||||
|
||||
// Latencies in us, Bandwidths in GB/s
|
||||
@@ -448,6 +462,8 @@ static float getNetOverhead(struct ncclComm* comm) {
|
||||
return 1.0;
|
||||
}
|
||||
|
||||
NCCL_PARAM(Ll128C2c, "LL128_C2C", 1);
|
||||
|
||||
ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCompCap, struct ncclTopoGraph** graphs) {
|
||||
int simpleDefaultThreads = (graphs[NCCL_ALGO_RING]->bwIntra*graphs[NCCL_ALGO_RING]->nChannels <= PCI_BW) ? 256 : NCCL_SIMPLE_MAX_NTHREADS;
|
||||
comm->maxThreads[NCCL_ALGO_RING][NCCL_PROTO_SIMPLE] =
|
||||
@@ -521,7 +537,14 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
float busBw = comm->topo->baseBw != 0.0 ? comm->topo->baseBw : graphs[a]->nChannels * bw;
|
||||
//INFO(NCCL_INIT, "algo %s proto %s busBw %f baseBw %f bw %f nChannels %d bwIntra %f bwInter %f", ncclAlgoStr[a], ncclProtoStr[p], busBw, comm->topo->baseBw, bw, graphs[a]->nChannels, graphs[a]->bwIntra, graphs[a]->bwInter);
|
||||
|
||||
if (a == NCCL_ALGO_NVLS) bw = std::min(graphs[a]->bwIntra, graphs[a]->bwInter);
|
||||
if (a == NCCL_ALGO_NVLS) {
|
||||
if (coll == ncclFuncAllReduce) {
|
||||
bw = std::min(graphs[a]->bwIntra, graphs[a]->bwInter);
|
||||
} else {
|
||||
// allgather and reducescatter
|
||||
bw = std::min(graphs[a]->bwIntra * (ppn - 1.0f) / ppn, graphs[a]->bwInter * 0.9f);
|
||||
}
|
||||
}
|
||||
if (a == NCCL_ALGO_NVLS_TREE) bw = std::min(graphs[a]->bwIntra, nNodes <= 2 ? graphs[a]->bwInter : graphs[a]->bwInter/2);
|
||||
|
||||
// Various model refinements
|
||||
@@ -543,19 +566,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
if (a == NCCL_ALGO_COLLNET_CHAIN && p != NCCL_PROTO_SIMPLE) busBw = 0; // Not used
|
||||
if (a == NCCL_ALGO_COLLNET_DIRECT && p == NCCL_PROTO_SIMPLE) {
|
||||
if (coll == ncclFuncAllGather || coll == ncclFuncReduceScatter) {
|
||||
busBw = ppn * bw;
|
||||
// AllGather/ReduceScatter requires 1:1 GPU:NIC
|
||||
int nicPerNode = comm->collNetHeadsNum;
|
||||
if (coll == ncclFuncAllGather && comm->nNodes > 1) {
|
||||
if (!comm->ncclCollNet || !comm->ncclCollNet->iallgather || ppn > nicPerNode) busBw = 0;
|
||||
}
|
||||
if (coll == ncclFuncReduceScatter && comm->nNodes > 1) {
|
||||
if (!comm->ncclCollNet || !comm->ncclCollNet->ireducescatter || ppn > nicPerNode) busBw = 0;
|
||||
}
|
||||
// Measured corrective ratio needed at 1 ppn and 8ppn. Here we hackishly
|
||||
// interpolate the two.
|
||||
float w = (ppn-1)/(8-1);
|
||||
busBw *= w*0.85 + (1-w)*0.95;
|
||||
busBw = ppn * std::min(graphs[a]->bwIntra, graphs[a]->bwInter * 0.9f);
|
||||
} else {
|
||||
// Collnet+Direct requires all GPUs to have a local NIC to work at full speed
|
||||
float factor = ppn / (1.0*graphs[a]->nChannels); // GPU/NIC ratio
|
||||
@@ -564,8 +575,27 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
if (minCompCap >= 90) busBw *= .85;
|
||||
}
|
||||
}
|
||||
// disable collnet for allgather/reducescatter if #localranks > #heads
|
||||
// AllGather/ReduceScatter requires 1:1 GPU:NIC
|
||||
if ((a == NCCL_ALGO_NVLS || a == NCCL_ALGO_COLLNET_DIRECT) && p == NCCL_PROTO_SIMPLE && (coll == ncclFuncAllGather || coll == ncclFuncReduceScatter) && comm->nNodes > 1) {
|
||||
int nHeads = 0;
|
||||
if (coll == ncclFuncAllGather && comm->nNodes > 1 && (!comm->ncclCollNet || !comm->ncclCollNet->iallgather)) busBw = 0.0f;
|
||||
if (coll == ncclFuncReduceScatter && comm->nNodes > 1 && (!comm->ncclCollNet || !comm->ncclCollNet->ireducescatter)) busBw = 0.0f;
|
||||
if (comm->config.collnetEnable)
|
||||
nHeads = comm->collNetHeadsNum;
|
||||
else
|
||||
busBw = 0.0f;
|
||||
if (busBw > 0.0f) {
|
||||
for (int r = 0; r < comm->nRanks; r++) {
|
||||
int node = comm->rankToNode[r];
|
||||
if (comm->nodeRanks[node].localRanks > nHeads) {
|
||||
busBw = 0.0f;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
// Convert bus BW to algorithm BW
|
||||
if (!(a != NCCL_ALGO_RING && (coll == ncclFuncAllGather || coll == ncclFuncReduceScatter))) {
|
||||
float ratio = 1.0f;
|
||||
@@ -689,7 +719,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
// Disable NVLS Tree on a single node
|
||||
if (comm->nNodes == 1 && a == NCCL_ALGO_NVLS_TREE) disable = 1;
|
||||
// Disable Collnet+Direct, Collnet+Chain or Collnet+NVLS if collnet is not supported.
|
||||
if (comm->collNetSupport == 0 &&
|
||||
if (comm->config.collnetEnable == 0 &&
|
||||
(a == NCCL_ALGO_COLLNET_DIRECT ||
|
||||
a == NCCL_ALGO_COLLNET_CHAIN ||
|
||||
(a == NCCL_ALGO_NVLS && comm->nNodes > 1))) disable = 1;
|
||||
@@ -716,17 +746,10 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
#else
|
||||
// Enable LL128 by default only on Volta/Ampere/Hopper+NVLink. Other cases are not tested and may cause silent data corruption.
|
||||
pEnable = 1;
|
||||
pEnable &= (graphs[a]->typeInter <= PATH_PXB || (minCompCap >= 90 && graphs[a]->typeInter <= PATH_PXN));
|
||||
pEnable &= (graphs[a]->typeInter <= PATH_PXB || (minCompCap >= 90 && graphs[a]->typeInter <= (ncclParamLl128C2c() ? PATH_P2C : PATH_PXN)));
|
||||
pEnable &= (graphs[a]->typeIntra <= PATH_NVB);
|
||||
pEnable &= (minCompCap == maxCompCap);
|
||||
switch (minCompCap) {
|
||||
case 70: pEnable &= 1; break;
|
||||
case 80: pEnable &= 1; break;
|
||||
case 90: pEnable &= !(CUDART_VERSION == 11080 && c == ncclFuncAllReduce && a == NCCL_ALGO_RING && comm->nRanks == 2); break;
|
||||
case 100: pEnable &= 1; break;
|
||||
case 120: pEnable &= 1; break;
|
||||
default: pEnable &= 0; break;
|
||||
}
|
||||
pEnable &= !(minCompCap < 70 || (minCompCap == 90 && CUDART_VERSION == 11080 && c == ncclFuncAllReduce && a == NCCL_ALGO_RING && comm->nRanks == 2));
|
||||
#endif
|
||||
}
|
||||
if (pEnable == 0) comm->bandwidths[c][a][p] = 0;
|
||||
|
||||
+29
-8
@@ -42,7 +42,13 @@ ncclResult_t xmlGetValue(FILE* file, char* value, char* last) {
|
||||
#if INT_OK
|
||||
int o = 0;
|
||||
do {
|
||||
value[o++] = c;
|
||||
value[o] = c;
|
||||
if (o == MAX_STR_LEN-1) {
|
||||
value[o] = '\0';
|
||||
WARN("Error : value %s too long (max %d)", value, MAX_STR_LEN);
|
||||
return ncclInternalError;
|
||||
}
|
||||
o++;
|
||||
NCCLCHECK(xmlGetChar(file, &c));
|
||||
} while (c >= '0' && c <= '9');
|
||||
value[o] = '\0';
|
||||
@@ -54,10 +60,17 @@ ncclResult_t xmlGetValue(FILE* file, char* value, char* last) {
|
||||
#endif
|
||||
}
|
||||
int o = 0;
|
||||
char quote = c; // Remember which quote type we started with
|
||||
do {
|
||||
NCCLCHECK(xmlGetChar(file, &c));
|
||||
value[o++] = c;
|
||||
} while (c != '"');
|
||||
value[o] = c;
|
||||
if (o == MAX_STR_LEN-1) {
|
||||
value[o] = '\0';
|
||||
WARN("Error : value %s too long (max %d)", value, MAX_STR_LEN);
|
||||
return ncclInternalError;
|
||||
}
|
||||
o++;
|
||||
} while (c != quote);
|
||||
value[o-1] = '\0';
|
||||
NCCLCHECK(xmlGetChar(file, last));
|
||||
return ncclSuccess;
|
||||
@@ -270,7 +283,7 @@ ncclResult_t ncclTopoDumpXmlRec(int indent, FILE* file, struct ncclXmlNode* node
|
||||
ncclResult_t ncclTopoDumpXmlToFile(const char* xmlTopoFile, struct ncclXml* xml) {
|
||||
FILE* file = fopen(xmlTopoFile, "w");
|
||||
if (file == NULL) {
|
||||
WARN("Unable to open %s, not dumping topology.", xmlTopoFile);
|
||||
INFO(NCCL_GRAPH|NCCL_ENV, "Unable to open %s, not dumping topology.", xmlTopoFile);
|
||||
return ncclSuccess;
|
||||
}
|
||||
NCCLCHECK(ncclTopoDumpXmlRec(0, file, xml->nodes));
|
||||
@@ -385,7 +398,7 @@ ncclResult_t ncclTopoGetXmlFromFile(const char* xmlTopoFile, struct ncclXml* xml
|
||||
FILE* file = fopen(xmlTopoFile, "r");
|
||||
if (file == NULL) {
|
||||
if (warn) {
|
||||
WARN("Could not open XML topology file %s : %s", xmlTopoFile, strerror(errno));
|
||||
INFO(NCCL_GRAPH|NCCL_ENV, "Could not open XML topology file %s : %s", xmlTopoFile, strerror(errno));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -835,7 +848,7 @@ ncclResult_t ncclTopoGetXmlFromGpu(struct ncclXmlNode* pciNode, uint32_t rocmDev
|
||||
int maxNvLinks = (sm < 60) ? 0 : (sm < 70) ? 4 : (sm < 80) ? 6 : (sm < 90) ? 12 : 18;
|
||||
|
||||
if (maxNvLinks > 0 && nvmlDev == NULL) {
|
||||
WARN("No NVML device handle. Skipping nvlink detection.");
|
||||
INFO(NCCL_GRAPH, "No NVML device handle. Skipping nvlink detection.");
|
||||
maxNvLinks = 0;
|
||||
}
|
||||
|
||||
@@ -1052,8 +1065,16 @@ ncclResult_t ncclTopoTrimXmlRec(struct ncclXmlNode* node, int* keep) {
|
||||
NCCLCHECK(ncclTopoTrimXmlRec(subs[s], &k));
|
||||
*keep += k;
|
||||
}
|
||||
if (*keep == 0 && // Trim PCI switches or CPU with no used GPU/NIC under them.
|
||||
(strcmp(node->name, "pci") == 0 || strcmp(node->name, "cpu") == 0)) {
|
||||
// Remove node if it has no children and no keep attribute
|
||||
if (*keep == 0 && // Trim PCI switches, CPUs with no used GPU/NIC under them, or pruned NICs
|
||||
(strcmp(node->name, "pci") == 0 || strcmp(node->name, "cpu") == 0 || strcmp(node->name, "nic") == 0 || strcmp(node->name, "net") == 0)) {
|
||||
#ifdef ENABLE_TRACE
|
||||
const char* name;
|
||||
const char* busid;
|
||||
NCCLCHECK(xmlGetAttr(node, "name", &name));
|
||||
NCCLCHECK(xmlGetAttr(node, "busid", &busid));
|
||||
TRACE(NCCL_GRAPH, "Removing node %s %s %s\n", node->name, name, busid);
|
||||
#endif
|
||||
NCCLCHECK(xmlRemoveNode(node));
|
||||
}
|
||||
}
|
||||
|
||||
+7
-4
@@ -121,6 +121,13 @@ static ncclResult_t xmlGetAttrIntDefault(struct ncclXmlNode* node, const char* a
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetAttrUint64(struct ncclXmlNode* node, const char* attrName, uint64_t* value) {
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttrStr(node, attrName, &str));
|
||||
*value = strtoull(str, NULL, 0);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetAttrLong(struct ncclXmlNode* node, const char* attrName, int64_t* value) {
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttrStr(node, attrName, &str));
|
||||
@@ -128,7 +135,6 @@ static ncclResult_t xmlGetAttrLong(struct ncclXmlNode* node, const char* attrNam
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
|
||||
static ncclResult_t xmlGetAttrFloat(struct ncclXmlNode* node, const char* attrName, float* value) {
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttrStr(node, attrName, &str));
|
||||
@@ -258,7 +264,6 @@ static ncclResult_t xmlSetAttrInt(struct ncclXmlNode* node, const char* attrName
|
||||
node->attrs[index].key[MAX_STR_LEN] = '\0';
|
||||
}
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "%d", value);
|
||||
node->attrs[index].value[MAX_STR_LEN] = '\0';
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -271,7 +276,6 @@ static ncclResult_t xmlSetAttrFloat(struct ncclXmlNode* node, const char* attrNa
|
||||
node->attrs[index].key[MAX_STR_LEN] = '\0';
|
||||
}
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "%g", value);
|
||||
node->attrs[index].value[MAX_STR_LEN] = '\0';
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -284,7 +288,6 @@ static ncclResult_t xmlSetAttrLong(struct ncclXmlNode* node, const char* attrNam
|
||||
node->attrs[index].key[MAX_STR_LEN] = '\0';
|
||||
}
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "%#lx", value);
|
||||
node->attrs[index].value[MAX_STR_LEN] = '\0';
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user