Merge remote-tracking branch 'nccl/master' into develop
This commit is contained in:
+18
-8
@@ -20,6 +20,7 @@
|
||||
ncclResult_t ncclTopoPreset(struct ncclComm* comm, struct ncclTopoGraph** graphs, struct ncclTopoRanks* topoRanks) {
|
||||
int rank = comm->rank;
|
||||
int localRanks = comm->topo->nodes[GPU].count;
|
||||
int nvlsRanks = comm->MNNVL ? comm->clique.size : localRanks;
|
||||
int nChannels = comm->nChannels;
|
||||
|
||||
topoRanks->nvlsHeadNum = 0;
|
||||
@@ -74,7 +75,7 @@ ncclResult_t ncclTopoPreset(struct ncclComm* comm, struct ncclTopoGraph** graphs
|
||||
// Get nvls heads and the number of heads. Duplicate head is not allowed.
|
||||
for (int c = 0; c < graphs[NCCL_ALGO_NVLS]->nChannels; ++c) {
|
||||
bool addHead = true;
|
||||
int* nvlsIntra = graphs[NCCL_ALGO_NVLS]->intra + c * localRanks;
|
||||
int* nvlsIntra = graphs[NCCL_ALGO_NVLS]->intra + c * nvlsRanks;
|
||||
|
||||
for (int dup = 0; dup < topoRanks->nvlsHeadNum; dup++) {
|
||||
if (topoRanks->nvlsHeads[dup] == nvlsIntra[0]) {
|
||||
@@ -455,8 +456,7 @@ static ncclResult_t connectNvls(struct ncclComm* comm, int* nvlsHeads, int nHead
|
||||
channel->nvls.nNodes = comm->nNodes;
|
||||
if (comm->collNetSupport && channel->nvls.headRank != -1) channel->nvls.out = comm->nRanks;
|
||||
}
|
||||
// MNNVL: NVLS not yet supported
|
||||
if (comm->nNodes == 1 || comm->MNNVL) return ncclSuccess;
|
||||
if (comm->nNodes == 1) return ncclSuccess;
|
||||
|
||||
// Connect Trees
|
||||
int tree0Parent, tree0Child0, tree0Child1, tree1Parent, tree1Child0, tree1Child1;
|
||||
@@ -508,9 +508,9 @@ static ncclResult_t connectNvls(struct ncclComm* comm, int* nvlsHeads, int nHead
|
||||
|
||||
struct ncclNvls* nvls0 = &comm->channels[0].nvls;
|
||||
struct ncclNvls* nvls1 = &comm->channels[1].nvls;
|
||||
INFO(NCCL_GRAPH, "NVLS Trees : %d/%d->%d->%d %d/%d->%d->%d",
|
||||
nvls0->treeDown[0], nvls0->treeDown[1], comm->rank, nvls0->treeUp,
|
||||
nvls1->treeDown[0], nvls1->treeDown[1], comm->rank, nvls1->treeUp);
|
||||
INFO(NCCL_GRAPH, "NVLS Trees : %d/%d/%d->%d->%d %d/%d/%d->%d->%d",
|
||||
nvls0->treeDown[0], nvls0->treeDown[1], nvls0->treeDown[2], comm->rank, nvls0->treeUp,
|
||||
nvls1->treeDown[0], nvls1->treeDown[1], nvls1->treeDown[2], comm->rank, nvls1->treeUp);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -561,13 +561,14 @@ void exchangeValues(int* v0, int* v1) {
|
||||
*v0 = tmp;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoPostset(struct ncclComm* comm, int* firstRanks, int* treePatterns, struct ncclTopoRanks** allTopoRanks, int* rings, struct ncclTopoGraph** graphs, int nc) {
|
||||
ncclResult_t ncclTopoPostset(struct ncclComm* comm, int* firstRanks, int* treePatterns, struct ncclTopoRanks** allTopoRanks, int* rings, struct ncclTopoGraph** graphs, struct ncclComm* parent, int nc) {
|
||||
// Gather data from all ranks
|
||||
int *ringRecv, *ringSend, *ringPrev, *ringNext, *treeToParent, *treeToChild0, *treeToChild1, *nvlsHeads;
|
||||
int nranks = comm->nRanks;
|
||||
int nNodes = comm->nNodes;
|
||||
int nChannels = comm->nChannels;
|
||||
int minHeadNum = INT_MAX;
|
||||
int shared = parent && parent->nvlsSupport && parent->config.splitShare;
|
||||
NCCLCHECK(ncclCalloc(&ringRecv, nNodes*MAXCHANNELS));
|
||||
NCCLCHECK(ncclCalloc(&ringSend, nNodes*MAXCHANNELS));
|
||||
NCCLCHECK(ncclCalloc(&ringPrev, nranks*MAXCHANNELS));
|
||||
@@ -578,7 +579,7 @@ ncclResult_t ncclTopoPostset(struct ncclComm* comm, int* firstRanks, int* treePa
|
||||
NCCLCHECK(ncclCalloc(&nvlsHeads, nNodes*MAXCHANNELS));
|
||||
|
||||
// Alternate rings to avoid crossing rails
|
||||
if (graphs[NCCL_ALGO_RING]->crossNic && (comm->nNodes % 2) == 0 && (nChannels % 2) == 0) {
|
||||
if (graphs[NCCL_ALGO_RING]->crossNic && (nChannels % 2) == 0) {
|
||||
for (int r=0; r<comm->nRanks; r++) {
|
||||
if (comm->rankToNode[r] % 2 == 1) {
|
||||
// Exchange rings
|
||||
@@ -705,11 +706,20 @@ ncclResult_t ncclTopoPostset(struct ncclComm* comm, int* firstRanks, int* treePa
|
||||
}
|
||||
|
||||
comm->collChannels = comm->nChannels;
|
||||
#if CUDART_VERSION >= 12010
|
||||
// Support maximal channel usage for aggregation
|
||||
if (shared && comm->nvlsChannels > parent->nvlsResources->nChannels) {
|
||||
comm->nvlsChannels = parent->nvlsResources->nChannels;
|
||||
}
|
||||
if (comm->nChannels < comm->nvlsChannels) {
|
||||
nChannels = comm->nChannels = copyChannels(comm, comm->nChannels, comm->nvlsChannels, ringPrev, ringNext);
|
||||
}
|
||||
NCCLCHECK(connectNvls(comm, nvlsHeads, minHeadNum));
|
||||
#endif
|
||||
if (shared && comm->nChannels > parent->sharedRes->tpNChannels) {
|
||||
nChannels = comm->nChannels = parent->sharedRes->tpNChannels;
|
||||
comm->collChannels = std::min(comm->collChannels, comm->nChannels);
|
||||
}
|
||||
|
||||
// Create rings array and check all is fine
|
||||
NCCLCHECK(ncclBuildRings(nChannels, rings, comm->rank, comm->nRanks, ringPrev, ringNext));
|
||||
|
||||
+35
-34
@@ -60,6 +60,7 @@ static ncclResult_t ncclTopoSetPaths(struct ncclTopoNode* baseNode, struct ncclT
|
||||
struct ncclTopoNode* remNode = link->remNode;
|
||||
if (remNode->paths[baseNode->type] == NULL) {
|
||||
NCCLCHECK(ncclCalloc(remNode->paths+baseNode->type, system->nodes[baseNode->type].count));
|
||||
for (int i=0; i<system->nodes[baseNode->type].count; i++) remNode->paths[baseNode->type][i].type = PATH_DIS;
|
||||
}
|
||||
struct ncclTopoLinkList* remPath;
|
||||
NCCLCHECK(getPath(system, remNode, baseNode->type, baseNode->id, &remPath));
|
||||
@@ -115,11 +116,12 @@ static ncclResult_t ncclTopoSetPaths(struct ncclTopoNode* baseNode, struct ncclT
|
||||
}
|
||||
|
||||
static void printNodePaths(struct ncclTopoSystem* system, struct ncclTopoNode* node) {
|
||||
char line[2048];
|
||||
const int linesize = 2048;
|
||||
char line[linesize];
|
||||
#ifdef ENABLE_TRACE
|
||||
INFO(NCCL_GRAPH, "Paths from %s/%lX :", topoNodeTypeStr[node->type], node->id);
|
||||
#else
|
||||
sprintf(line, "%s/%lX :", topoNodeTypeStr[node->type], node->id);
|
||||
snprintf(line, linesize, "%s/%lX :", topoNodeTypeStr[node->type], node->id);
|
||||
int offset = strlen(line);
|
||||
#endif
|
||||
for (int t=0; t<NCCL_TOPO_NODE_TYPES; t++) {
|
||||
@@ -131,12 +133,12 @@ static void printNodePaths(struct ncclTopoSystem* system, struct ncclTopoNode* n
|
||||
for (int i=0; i<node->paths[t][n].count; i++) {
|
||||
struct ncclTopoLink* link = node->paths[t][n].list[i];
|
||||
struct ncclTopoNode* remNode = link->remNode;
|
||||
sprintf(line+offset, "--%s(%g)->%s/%lX", topoLinkTypeStr[link->type], link->bw, topoNodeTypeStr[remNode->type], remNode->id);
|
||||
snprintf(line+offset, linesize-offset, "--%s(%g)->%s/%lx-%lx", topoLinkTypeStr[link->type], link->bw, topoNodeTypeStr[remNode->type], NCCL_TOPO_ID_SYSTEM_ID(remNode->id), NCCL_TOPO_ID_LOCAL_ID(remNode->id));
|
||||
offset = strlen(line);
|
||||
}
|
||||
INFO(NCCL_GRAPH, "%s (%f)", line, node->paths[t][n].bw);
|
||||
#else
|
||||
sprintf(line+offset, "%s/%lX (%d/%f/%s) ", topoNodeTypeStr[t], system->nodes[t].nodes[n].id, node->paths[t][n].count, node->paths[t][n].bw, topoPathTypeStr[node->paths[t][n].type]);
|
||||
snprintf(line+offset, linesize-offset, "%s/%lx-%lx (%d/%.1f/%s) ", topoNodeTypeStr[t], NCCL_TOPO_ID_SYSTEM_ID(system->nodes[t].nodes[n].id), NCCL_TOPO_ID_LOCAL_ID(system->nodes[t].nodes[n].id), node->paths[t][n].count, node->paths[t][n].bw, topoPathTypeStr[node->paths[t][n].type]);
|
||||
offset = strlen(line);
|
||||
#endif
|
||||
}
|
||||
@@ -369,12 +371,12 @@ ncclResult_t ncclTopoCheckMNNVL(struct ncclTopoSystem* system, struct ncclPeerIn
|
||||
NCCL_PARAM(NetGdrRead, "NET_GDR_READ", -2);
|
||||
int ncclTopoUserGdrLevel = -1;
|
||||
|
||||
ncclResult_t ncclTopoCheckGdr(struct ncclTopoSystem* system, int64_t busId, int netDev, int read, int* useGdr) {
|
||||
ncclResult_t ncclTopoCheckGdr(struct ncclTopoSystem* system, int64_t busId, int64_t netId, int read, int* useGdr) {
|
||||
*useGdr = 0;
|
||||
|
||||
// Get GPU and NET
|
||||
int n, g;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netDev, &n));
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netId, &n));
|
||||
struct ncclTopoNode* net = system->nodes[NET].nodes+n;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, GPU, busId, &g));
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
@@ -432,18 +434,18 @@ ncclResult_t ncclTopoCheckGdr(struct ncclTopoSystem* system, int64_t busId, int
|
||||
if (distance == PATH_PXN) {
|
||||
// In case of PXN, use the intermediate GPU distance instead
|
||||
int proxyRank, g;
|
||||
NCCLCHECK(ncclTopoGetIntermediateRank(system, gpu->gpu.rank, netDev, &proxyRank));
|
||||
NCCLCHECK(ncclTopoGetIntermediateRank(system, gpu->gpu.rank, netId, &proxyRank));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, proxyRank, &g));
|
||||
struct ncclTopoNode* proxyGpu = system->nodes[GPU].nodes+g;
|
||||
distance = proxyGpu->paths[NET][n].type;
|
||||
}
|
||||
if (distance > netGdrLevel) {
|
||||
INFO(NCCL_NET,"GPU Direct RDMA Disabled for GPU %lx / HCA %d (distance %d > %d)", busId, netDev, distance, netGdrLevel);
|
||||
INFO(NCCL_NET,"GPU Direct RDMA Disabled for GPU %lx / HCA %lx (distance %d > %d)", busId, netId, distance, netGdrLevel);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
*useGdr = 1;
|
||||
INFO(NCCL_NET,"GPU Direct RDMA Enabled for GPU %lx / HCA %d (distance %d <= %d), read %d", busId, netDev, distance, netGdrLevel, read);
|
||||
INFO(NCCL_NET,"GPU Direct RDMA Enabled for GPU %lx / HCA %lx (distance %d <= %d), read %d", busId, netId, distance, netGdrLevel, read);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -497,10 +499,10 @@ ncclResult_t ncclTopoCheckNet(struct ncclTopoSystem* system, int64_t id1, int64_
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetIntermediateRank(struct ncclTopoSystem* system, int rank, int netDev, int* intermediateRank) {
|
||||
ncclResult_t ncclTopoGetIntermediateRank(struct ncclTopoSystem* system, int rank, int64_t netId, int* intermediateRank) {
|
||||
// Get GPU and NET
|
||||
int n, g;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netDev, &n));
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netId, &n));
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &g));
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
struct ncclTopoLinkList* path = gpu->paths[NET]+n;
|
||||
@@ -512,7 +514,7 @@ ncclResult_t ncclTopoGetIntermediateRank(struct ncclTopoSystem* system, int rank
|
||||
type = node->type;
|
||||
}
|
||||
if (type != GPU) {
|
||||
WARN("Could not find intermediate GPU between GPU rank %d and NIC %d", rank, netDev);
|
||||
WARN("Could not find intermediate GPU between GPU rank %d and NIC %lx", rank, netId);
|
||||
return ncclInternalError;
|
||||
}
|
||||
*intermediateRank = node->gpu.rank;
|
||||
@@ -548,11 +550,12 @@ ncclResult_t ncclTopoGetPxnRanks(struct ncclComm* comm, int** intermediateRanks,
|
||||
int nr = 0;
|
||||
int* ranks = NULL;
|
||||
for (int rank=0; rank<comm->nRanks; rank++) {
|
||||
int netDev, proxyRank;
|
||||
NCCLCHECK(ncclTopoGetNetDev(comm, comm->rank, NULL, 0, rank, &netDev, &proxyRank));
|
||||
int64_t netId;
|
||||
int proxyRank;
|
||||
NCCLCHECK(ncclTopoGetNetDev(comm, comm->rank, NULL, 0, rank, &netId, NULL, &proxyRank));
|
||||
if (proxyRank == comm->rank) continue;
|
||||
int useGdr;
|
||||
NCCLCHECK(ncclTopoCheckGdr(comm->topo, comm->busId, netDev, 1, &useGdr));
|
||||
NCCLCHECK(ncclTopoCheckGdr(comm->topo, comm->busId, netId, 1, &useGdr));
|
||||
if (useGdr == 0) continue;
|
||||
int found = 0;
|
||||
for (int r=0; r<nr; r++) {
|
||||
@@ -679,13 +682,14 @@ ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclComm
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
if (ncclPxnDisable(comm) != 1) {
|
||||
int localGpuIndex;
|
||||
NCCLCHECK(ncclTopoGetLocalGpu(system, system->nodes[NET].nodes[n].id, &localGpuIndex));
|
||||
NCCLCHECK(ncclTopoGetLocalGpu(system, netNode->id, &localGpuIndex));
|
||||
if (localGpuIndex != g && localGpuIndex != -1) {
|
||||
// PXN = PCI + NVLink.
|
||||
struct ncclTopoNode* peerNode = system->nodes[GPU].nodes+localGpuIndex;
|
||||
// Only use PXN for NIC n if remote GPU p ...
|
||||
if (peerNode->paths[NET][n].type <= PATH_PXB && // Is connected to the NIC through PCI
|
||||
peerNode->paths[GPU][g].type <= PATH_NVL && // Is connected to us through NVLink
|
||||
NCCL_TOPO_ID_SYSTEM_ID(peerNode->id) == NCCL_TOPO_ID_SYSTEM_ID(gpu->id) && // Is on the same node as us
|
||||
(peerNode->paths[NET][n].bw > gpu->paths[NET][n].bw || // Has either higher BW to that NIC
|
||||
gpu->paths[NET][n].type > PATH_PXB)) // or avoids going through a CPU
|
||||
// We can use that GPU as relay to communicate with that NIC.
|
||||
@@ -694,15 +698,17 @@ ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclComm
|
||||
NCCLCHECK(addInterStep(system, GPU, localGpuIndex, GPU, g, NET, n));
|
||||
}
|
||||
}
|
||||
// Update path when we dont want to / can't use GPU Direct RDMA.
|
||||
int gdr;
|
||||
NCCLCHECK(ncclTopoCheckGdr(system, system->nodes[GPU].nodes[g].id, netNode->id, 0, &gdr));
|
||||
if (gdr == 0) {
|
||||
// We cannot use GPU Direct RDMA, divert all traffic through the CPU local to the GPU
|
||||
int localCpu;
|
||||
NCCLCHECK(getLocalCpu(system, g, &localCpu));
|
||||
NCCLCHECK(addInterStep(system, CPU, localCpu, NET, n, GPU, g));
|
||||
NCCLCHECK(addInterStep(system, CPU, localCpu, GPU, g, NET, n));
|
||||
if (gpu->paths[NET][n].type < PATH_PHB) {
|
||||
// Update path when we dont want to / can't use GPU Direct RDMA.
|
||||
int gdr;
|
||||
NCCLCHECK(ncclTopoCheckGdr(system, system->nodes[GPU].nodes[g].id, netNode->id, 0, &gdr));
|
||||
if (gdr == 0) {
|
||||
// We cannot use GPU Direct RDMA, divert all traffic through the CPU local to the GPU
|
||||
int localCpu;
|
||||
NCCLCHECK(getLocalCpu(system, g, &localCpu));
|
||||
NCCLCHECK(addInterStep(system, CPU, localCpu, NET, n, GPU, g));
|
||||
NCCLCHECK(addInterStep(system, CPU, localCpu, GPU, g, NET, n));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -785,9 +791,9 @@ ncclResult_t ncclTopoTrimSystem(struct ncclTopoSystem* system, struct ncclComm*
|
||||
}
|
||||
if (allXgmi) system->type |= RCCL_TOPO_XGMI_ALL;
|
||||
for (int g = 0; g < system->nodes[GPU].count; g++) {
|
||||
int net;
|
||||
NCCLCHECK(ncclTopoGetLocalNet(system, system->nodes[GPU].nodes[g].gpu.rank, 0, &net));
|
||||
NCCLCHECK(ncclTopoCheckGdr(system, system->nodes[GPU].nodes[g].id, net, 1, &gdr));
|
||||
int64_t netId;
|
||||
NCCLCHECK(ncclTopoGetLocalNet(system, system->nodes[GPU].nodes[g].gpu.rank, 0, &netId, nullptr));
|
||||
NCCLCHECK(ncclTopoCheckGdr(system, system->nodes[GPU].nodes[g].id, netId, 1, &gdr));
|
||||
if (!gdr) break;
|
||||
}
|
||||
if (gdr && !allXgmi) {
|
||||
@@ -802,7 +808,7 @@ ncclResult_t ncclTopoTrimSystem(struct ncclTopoSystem* system, struct ncclComm*
|
||||
}
|
||||
|
||||
comm->localRanks = system->nodes[GPU].count;
|
||||
if ((system->nodes[GPU].count == comm->nRanks && remove) || comm->MNNVL) {
|
||||
if (system->nodes[GPU].count == comm->nRanks && remove) {
|
||||
for (int n=system->nodes[NET].count-1; n>=0; n--)
|
||||
NCCLCHECK(ncclTopoRemoveNode(system, NET, n));
|
||||
}
|
||||
@@ -838,11 +844,6 @@ static ncclResult_t ncclTopoGetNchannels(struct ncclComm* comm, int g /*local gp
|
||||
} else {
|
||||
*nChannels = 2;
|
||||
}
|
||||
} else if (comm->MNNVL) {
|
||||
// MNNVL assume all GPUs are connected via NVLink
|
||||
path = system->nodes[GPU].nodes[g].paths[GPU]+((g+1)%system->nodes[GPU].count);
|
||||
float nvlBw = ncclTopoNVLinkBw(system->nodes[GPU].nodes[g].gpu.cudaCompCap);
|
||||
*nChannels = 2*std::max(1, (int)(path->bw / nvlBw));
|
||||
} else {
|
||||
// Remote rank, use network
|
||||
int nNetChannels = ncclParamNChannelsPerNetPeer();
|
||||
|
||||
@@ -1031,9 +1031,9 @@ end:
|
||||
graph->bwIntra = graph->bwInter = system->totalBw/nChannels;
|
||||
if (graph->id == 1) {
|
||||
for (int i=0; i<graph->nChannels; i++) {
|
||||
int net;
|
||||
ncclTopoGetLocalNet(system, graph->intra[i*ngpus+1], i, &net);
|
||||
graph->inter[i*2+1] = net;
|
||||
int64_t netId;
|
||||
ncclTopoGetLocalNet(system, graph->intra[i*ngpus+1], i, &netId, nullptr);
|
||||
graph->inter[i*2+1] = netId;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+68
-51
@@ -5,6 +5,7 @@
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include "comm.h"
|
||||
#include "core.h"
|
||||
#include "graph.h"
|
||||
#include "topo.h"
|
||||
@@ -42,6 +43,7 @@ ncclResult_t ncclTopoSearchInit(struct ncclTopoSystem* system) {
|
||||
int inter = system->nodes[NET].count;
|
||||
if (inter == 0 && system->nodes[GPU].count == 1) {
|
||||
system->maxBw = LOC_BW;
|
||||
system->totalBw = LOC_BW;
|
||||
return ncclSuccess;
|
||||
}
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
@@ -118,7 +120,6 @@ static ncclResult_t ncclTopoFollowPath(struct ncclTopoSystem* system, struct ncc
|
||||
WARN("No path computed to go from %s/%d to %s/%d", topoNodeTypeStr[type1], index1, topoNodeTypeStr[type2], index2);
|
||||
return ncclInternalError;
|
||||
}
|
||||
if (path->count == 0 ) return ncclSuccess;
|
||||
|
||||
// Now check link type
|
||||
*node = NULL;
|
||||
@@ -220,7 +221,7 @@ static ncclResult_t getNetIndex(struct ncclTopoSystem* system, int64_t id, int*
|
||||
}
|
||||
|
||||
static ncclResult_t getNetPaths(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoLinkList** netPaths) {
|
||||
int netId = graph->inter[graph->nChannels*2];
|
||||
int64_t netId = graph->inter[graph->nChannels*2];
|
||||
int n;
|
||||
NCCLCHECK(getNetIndex(system, netId, &n));
|
||||
*netPaths=system->nodes[NET].nodes[n].paths[GPU];
|
||||
@@ -264,6 +265,8 @@ ncclResult_t ncclTopoSearchNextGpuSort(struct ncclTopoSystem* system, struct ncc
|
||||
for (int i=0; i<count; i++) next[i] = scores[i].g;
|
||||
}
|
||||
|
||||
*countPtr = count;
|
||||
|
||||
if (system->nodes[NVS].count) {
|
||||
// NVSwitches prefer when we talk to a limited set of peers. Try to use neighbors first.
|
||||
int index = gpu-system->nodes[GPU].nodes;
|
||||
@@ -280,16 +283,18 @@ ncclResult_t ncclTopoSearchNextGpuSort(struct ncclTopoSystem* system, struct ncc
|
||||
} else {
|
||||
firstGpus[0] = nextGpu; firstGpuCount = 1;
|
||||
}
|
||||
if (nextGpu == prevGpu && firstGpuCount == 2) firstGpuCount = 1;
|
||||
int firstGpuRealCount = 0;
|
||||
for (int g=0; g<firstGpuCount; g++) {
|
||||
for (i=0; i<count && next[i] != firstGpus[g]; i++);
|
||||
if (i<count) {
|
||||
for (; i>0; i--) next[i] = next[i-1];
|
||||
next[0] = firstGpus[g];
|
||||
firstGpuRealCount++;
|
||||
}
|
||||
}
|
||||
*countPtr = firstGpuRealCount;
|
||||
}
|
||||
|
||||
*countPtr = count;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -404,7 +409,6 @@ ncclResult_t ncclTopoCompareGraphs(struct ncclTopoSystem* system, struct ncclTop
|
||||
return ncclSuccess;
|
||||
}
|
||||
// 2. Try to get better bandwidth
|
||||
// Give a 5% perf bonus to paths not crossing nics
|
||||
if (graph->nChannels*graph->bwIntra > refGraph->nChannels*refGraph->bwIntra) {
|
||||
*copy = 1;
|
||||
return ncclSuccess;
|
||||
@@ -441,8 +445,8 @@ ncclResult_t ncclTopoSelectNets(struct ncclTopoSystem* system, int typeInter, in
|
||||
localNetCount = 0;
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
for (int c = 0; c<MAXCHANNELS; c++) {
|
||||
int netId;
|
||||
NCCLCHECK(ncclTopoGetLocalNet(system, gpu->gpu.rank, c, &netId));
|
||||
int64_t netId;
|
||||
NCCLCHECK(ncclTopoGetLocalNet(system, gpu->gpu.rank, c, &netId, NULL));
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netId, localNets+localNetCount));
|
||||
if (localNetCount > 0 && localNets[localNetCount] == localNets[0]) break;
|
||||
localNetCount++;
|
||||
@@ -463,7 +467,7 @@ ncclResult_t ncclTopoSelectNets(struct ncclTopoSystem* system, int typeInter, in
|
||||
localNetCount = 0;
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
struct ncclTopoLinkList* paths = gpu->paths[NET];
|
||||
for (int n=0; n<system->nodes[NET].count; n++) {
|
||||
for (int n=0; n<system->nodes[NET].count && n<MAXCHANNELS; n++) {
|
||||
if (paths[n].type == t) localNets[localNetCount++] = n;
|
||||
}
|
||||
// Append NICs to list
|
||||
@@ -752,22 +756,25 @@ struct kvDict kvDictLinkType[] = {
|
||||
|
||||
ncclResult_t ncclTopoGetChannelFromXml(struct ncclXmlNode *xmlChannel, int c, struct ncclTopoSystem* system, struct ncclTopoGraph* graph) {
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
int* inter = graph->inter+2*c;
|
||||
int64_t* inter = graph->inter+2*c;
|
||||
int* intra = graph->intra+ngpus*c;
|
||||
int n=0, g=0;
|
||||
for (int s=0; s<xmlChannel->nSubs; s++) {
|
||||
struct ncclXmlNode* sub = xmlChannel->subs[s];
|
||||
int dev;
|
||||
NCCLCHECK(xmlGetAttrInt(sub, "dev", &dev));
|
||||
int64_t dev;
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttrStr(sub, "dev", &str));
|
||||
dev = strtol(str, NULL, 16);
|
||||
if (strcmp(sub->name, "net") == 0) {
|
||||
inter[n++] = dev;
|
||||
} else if (strcmp(sub->name, "gpu") == 0) {
|
||||
int rank = -1;
|
||||
for (int g=0; g<ngpus; g++) {
|
||||
if (system->nodes[GPU].nodes[g].gpu.dev == dev) rank = system->nodes[GPU].nodes[g].gpu.rank;
|
||||
int systemId = NCCL_TOPO_ID_SYSTEM_ID(system->nodes[GPU].nodes[g].id);
|
||||
if (NCCL_TOPO_ID(systemId, system->nodes[GPU].nodes[g].gpu.dev) == dev) rank = system->nodes[GPU].nodes[g].gpu.rank;
|
||||
}
|
||||
if (rank == -1) {
|
||||
WARN("XML Import Channel : dev %d not found.", dev);
|
||||
WARN("XML Import Channel : dev %ld not found.", dev);
|
||||
return ncclSystemError;
|
||||
}
|
||||
intra[g++] = rank;
|
||||
@@ -813,29 +820,33 @@ ncclResult_t ncclTopoGetGraphFromXml(struct ncclXmlNode *xmlGraphs, struct ncclT
|
||||
ncclResult_t ncclTopoGetXmlFromChannel(struct ncclTopoGraph* graph, int c, struct ncclTopoSystem* system, struct ncclXml *xml, struct ncclXmlNode* parent) {
|
||||
struct ncclXmlNode* xmlChannel;
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
int* inter = graph->inter+2*c;
|
||||
int64_t* inter = graph->inter+2*c;
|
||||
int* intra = graph->intra+ngpus*c;
|
||||
NCCLCHECK(xmlAddNode(xml, parent, "channel", &xmlChannel));
|
||||
struct ncclXmlNode* node;
|
||||
if (system->nodes[NET].count) {
|
||||
NCCLCHECK(xmlAddNode(xml, xmlChannel, "net", &node));
|
||||
NCCLCHECK(xmlSetAttrInt(node, "dev", inter[0]));
|
||||
NCCLCHECK(xmlSetAttrLong(node, "dev", inter[0]));
|
||||
}
|
||||
for (int g=0; g<ngpus; g++) {
|
||||
NCCLCHECK(xmlAddNode(xml, xmlChannel, "gpu", &node));
|
||||
int dev = -1;
|
||||
int64_t dev = -1;
|
||||
for (int i=0; i<ngpus; i++) {
|
||||
if (system->nodes[GPU].nodes[i].gpu.rank == intra[g]) dev = system->nodes[GPU].nodes[i].gpu.dev;
|
||||
if (system->nodes[GPU].nodes[i].gpu.rank == intra[g]) {
|
||||
int systemId = NCCL_TOPO_ID_SYSTEM_ID(system->nodes[GPU].nodes[i].id);
|
||||
dev = NCCL_TOPO_ID(systemId, system->nodes[GPU].nodes[i].gpu.dev);
|
||||
}
|
||||
}
|
||||
if (dev == -1) {
|
||||
WARN("XML Export Channel : rank %d not found.", intra[g]);
|
||||
return ncclInternalError;
|
||||
}
|
||||
NCCLCHECK(xmlSetAttrInt(node, "dev", dev));
|
||||
NCCLCHECK(xmlSetAttrLong(node, "dev", dev));
|
||||
if (graph->id == 3) break; // NVLS graphs only use the first GPU
|
||||
}
|
||||
if (system->nodes[NET].count) {
|
||||
NCCLCHECK(xmlAddNode(xml, xmlChannel, "net", &node));
|
||||
NCCLCHECK(xmlSetAttrInt(node, "dev", inter[1]));
|
||||
NCCLCHECK(xmlSetAttrLong(node, "dev", inter[1]));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -879,7 +890,7 @@ ncclResult_t ncclTopoDupChannels(struct ncclTopoGraph* graph, int ccMin, int ngp
|
||||
|
||||
int dupChannels = std::min(graph->nChannels*2, graph->maxChannels);
|
||||
memcpy(graph->intra+graph->nChannels*ngpus, graph->intra, (dupChannels-graph->nChannels)*ngpus*sizeof(int));
|
||||
memcpy(graph->inter+graph->nChannels*2,graph->inter, (dupChannels-graph->nChannels)*2*sizeof(int));
|
||||
memcpy(graph->inter+graph->nChannels*2,graph->inter, (dupChannels-graph->nChannels)*2*sizeof(int64_t));
|
||||
graph->bwIntra /= DIVUP(dupChannels, graph->nChannels);
|
||||
graph->bwInter /= DIVUP(dupChannels, graph->nChannels);
|
||||
graph->nChannels = dupChannels;
|
||||
@@ -897,7 +908,7 @@ float speedArrayInter[] = { 48.0, 30.0, 28.0, 24.0, 20.0, 18.0, 15.0, 12.0, 10.0
|
||||
#define NSPEEDSINTRA (sizeof(speedArrayIntra)/sizeof(float))
|
||||
#define NSPEEDSINTER (sizeof(speedArrayInter)/sizeof(float))
|
||||
|
||||
float sm90SpeedArrayIntra[] = { 60.0, 50.0, 40.0, 30.0, 24.0, 20.0, 15.0, 12.0, 6.0, 3.0 };
|
||||
float sm90SpeedArrayIntra[] = { 60.0, 50.0, 40.0, 30.0, 24.0, 20.0, 15.0, 12.0, 11.0, 6.0, 3.0 };
|
||||
float sm90SpeedArrayInter[] = { 48.0, 45.0, 42.0, 40.0, 30.0, 24.0, 22.0, 20.0, 17.5, 15.0, 12.0, 6.0, 3.0, 2.4, 1.2, 0.24, 0.12 };
|
||||
#define NSPEEDSINTRA_SM90 (sizeof(sm90SpeedArrayIntra)/sizeof(float))
|
||||
#define NSPEEDSINTER_SM90 (sizeof(sm90SpeedArrayInter)/sizeof(float))
|
||||
@@ -929,7 +940,7 @@ ncclResult_t ncclTopoCompute(ncclTopoSystem* system, struct ncclTopoGraph* graph
|
||||
if (str) {
|
||||
INFO(NCCL_ENV, "NCCL_GRAPH_FILE set by environment to %s", str);
|
||||
struct ncclXml* xml;
|
||||
NCCLCHECK(ncclCalloc(&xml, 1));
|
||||
NCCLCHECK(xmlAlloc(&xml, NCCL_GRAPH_XML_MAX_NODES));
|
||||
NCCLCHECK(ncclTopoGetXmlGraphFromFile(str, xml));
|
||||
int nChannels;
|
||||
NCCLCHECK(ncclTopoGetGraphFromXml(xml->nodes, system, graph, &nChannels));
|
||||
@@ -1009,7 +1020,7 @@ ncclResult_t ncclTopoCompute(ncclTopoSystem* system, struct ncclTopoGraph* graph
|
||||
int speedIndex = 0;
|
||||
float maxBw = system->maxBw;
|
||||
float totalBw = system->totalBw;
|
||||
if (ngpus == 1 || graph->pattern != NCCL_TOPO_PATTERN_RING) totalBw *= ngpus*1.0/(ngpus-1);
|
||||
if (ngpus > 1 && graph->pattern != NCCL_TOPO_PATTERN_RING) totalBw *= ngpus*1.0/(ngpus-1);
|
||||
while ((speedArray[speedIndex] > maxBw || speedArray[speedIndex]*graph->minChannels > totalBw) && speedIndex < nspeeds-1) speedIndex++;
|
||||
tmpGraph.bwIntra = tmpGraph.bwInter = speedArray[speedIndex];
|
||||
int64_t globalTimeout = NCCL_SEARCH_GLOBAL_TIMEOUT;
|
||||
@@ -1028,7 +1039,7 @@ search:
|
||||
for (int g=0; g<ngpus; g++) {
|
||||
printf("%d ", graph->intra[c*ngpus+g]);
|
||||
}
|
||||
printf("[%d %d]", graph->inter[c*2+0], graph->inter[c*2+1]);
|
||||
printf("[%lx %lx]", graph->inter[c*2+0], graph->inter[c*2+1]);
|
||||
printf("\n");
|
||||
}
|
||||
#endif
|
||||
@@ -1143,7 +1154,7 @@ ncclResult_t ncclTopoPrintGraph(struct ncclTopoSystem* system, struct ncclTopoGr
|
||||
sprintf(line, "%2d :", c);
|
||||
int offset = strlen(line);
|
||||
if (system->nodes[NET].count > 0 && system->nodes[GPU].count != system->nRanks && !graph->nIntraChannels) {
|
||||
sprintf(line+offset, " %s/%d", topoNodeTypeStr[NET], graph->inter[2*c]);
|
||||
sprintf(line+offset, " %s/%lx", topoNodeTypeStr[NET], graph->inter[2*c]);
|
||||
offset = strlen(line);
|
||||
}
|
||||
for (int i=0; i<ngpus; i++) {
|
||||
@@ -1161,7 +1172,7 @@ ncclResult_t ncclTopoPrintGraph(struct ncclTopoSystem* system, struct ncclTopoGr
|
||||
}
|
||||
}
|
||||
if (system->nodes[NET].count > 0 && system->nodes[GPU].count != system->nRanks && !graph->nIntraChannels) {
|
||||
sprintf(line+offset, " %s/%d", topoNodeTypeStr[NET], graph->inter[2*c+1]);
|
||||
sprintf(line+offset, " %s/%lx", topoNodeTypeStr[NET], graph->inter[2*c+1]);
|
||||
offset = strlen(line);
|
||||
}
|
||||
INFO(NCCL_GRAPH, "%s", line);
|
||||
@@ -1174,7 +1185,7 @@ ncclResult_t ncclTopoDumpGraphs(struct ncclTopoSystem* system, int ngraphs, stru
|
||||
if (str) {
|
||||
INFO(NCCL_ENV, "NCCL_GRAPH_DUMP_FILE set by environment to %s", str);
|
||||
struct ncclXml* xml;
|
||||
NCCLCHECK(ncclCalloc(&xml, 1));
|
||||
NCCLCHECK(xmlAlloc(&xml, NCCL_GRAPH_XML_MAX_NODES));
|
||||
NCCLCHECK(ncclTopoGetXmlFromGraphs(ngraphs, graphs, system, xml));
|
||||
NCCLCHECK(ncclTopoDumpXmlToFile(str, xml));
|
||||
free(xml);
|
||||
@@ -1184,11 +1195,11 @@ ncclResult_t ncclTopoDumpGraphs(struct ncclTopoSystem* system, int ngraphs, stru
|
||||
|
||||
#include "comm.h"
|
||||
// NVLS channels aren't compute channels. Find which NIC corresponds to our rank being the head
|
||||
ncclResult_t getNvlsNetDev(struct ncclComm* comm, struct ncclTopoGraph* graph, int channelId, int* dev) {
|
||||
ncclResult_t getNvlsNetDev(struct ncclComm* comm, struct ncclTopoGraph* graph, int channelId, int64_t* netId) {
|
||||
ncclResult_t ret = ncclSuccess;
|
||||
int localRanks = comm->topo->nodes[GPU].count;
|
||||
int netNum = 0;
|
||||
int net[MAXCHANNELS];
|
||||
int64_t net[MAXCHANNELS];
|
||||
|
||||
for (int c = 0; c < graph->nChannels; c++) {
|
||||
if (graph->intra[c * localRanks] == comm->rank) {
|
||||
@@ -1196,7 +1207,7 @@ ncclResult_t getNvlsNetDev(struct ncclComm* comm, struct ncclTopoGraph* graph, i
|
||||
}
|
||||
}
|
||||
if (netNum) {
|
||||
*dev = net[channelId % netNum];
|
||||
*netId = net[channelId % netNum];
|
||||
} else {
|
||||
ret = ncclInternalError;
|
||||
goto fail;
|
||||
@@ -1212,23 +1223,30 @@ fail:
|
||||
// 0: don't use PXN for P2P, 1: use PXN if needed, 2: use PXN as much as possible to maximize aggregation
|
||||
NCCL_PARAM(P2pPxnLevel, "P2P_PXN_LEVEL", 2);
|
||||
|
||||
ncclResult_t ncclTopoGetNetDev(struct ncclComm* comm, int rank, struct ncclTopoGraph* graph, int channelId, int peerRank, int* dev, int* proxyRank) {
|
||||
ncclResult_t ncclTopoGetNetDev(struct ncclComm* comm, int rank, struct ncclTopoGraph* graph, int channelId, int peerRank, int64_t* id, int* dev, int* proxyRank) {
|
||||
int64_t netId = -1;
|
||||
int netDev = -1;
|
||||
if (graph) {
|
||||
// Honor the net device in the graph
|
||||
int channel = channelId%graph->nChannels;
|
||||
int ngpus = comm->topo->nodes[GPU].count;
|
||||
int index = graph->intra[channel*ngpus] == rank ? 0 : 1;
|
||||
if (graph->pattern != NCCL_TOPO_PATTERN_NVLS) {
|
||||
*dev = graph->inter[channel*2+index];
|
||||
netId = graph->inter[channel*2+index];
|
||||
} else {
|
||||
NCCLCHECK(getNvlsNetDev(comm, graph, channelId, dev));
|
||||
NCCLCHECK(getNvlsNetDev(comm, graph, channelId, &netId));
|
||||
}
|
||||
NCCLCHECK(ncclTopoGetIntermediateRank(comm->topo, rank, *dev, proxyRank));
|
||||
NCCLCHECK(ncclTopoIdToNetDev(comm->topo, netId, &netDev));
|
||||
if (dev) *dev = netDev;
|
||||
if (id) *id = netId;
|
||||
NCCLCHECK(ncclTopoGetIntermediateRank(comm->topo, rank, netId, proxyRank));
|
||||
} else if (peerRank == -1) {
|
||||
return ncclInternalError;
|
||||
} else {
|
||||
// Start with our local NIC and local Rank
|
||||
NCCLCHECK(ncclTopoGetLocalNet(comm->topo, rank, channelId, dev));
|
||||
NCCLCHECK(ncclTopoGetLocalNet(comm->topo, rank, channelId, &netId, &netDev));
|
||||
if (dev) *dev = netDev;
|
||||
if (id) *id = netId;
|
||||
*proxyRank = rank;
|
||||
|
||||
int pxnLevel = ncclPxnDisable(comm) == 1 ? 0 : ncclParamP2pPxnLevel();
|
||||
@@ -1238,38 +1256,35 @@ ncclResult_t ncclTopoGetNetDev(struct ncclComm* comm, int rank, struct ncclTopoG
|
||||
int cudaDev = comm->peerInfo[peerRank].cudaDev;
|
||||
int localRank;
|
||||
if (ncclTopoDevToRank(comm->topo, cudaDev, &localRank) != ncclSuccess) return ncclSuccess;
|
||||
int netDev;
|
||||
NCCLCHECK(ncclTopoGetLocalNet(comm->topo, localRank, channelId, &netDev));
|
||||
NCCLCHECK(ncclTopoGetLocalNet(comm->topo, localRank, channelId, &netId, &netDev));
|
||||
|
||||
int n;
|
||||
// Check that device exists on our node
|
||||
if (ncclParamCrossNic() == 0) {
|
||||
if (ncclTopoIdToIndex(comm->topo, NET, netDev, &n) != ncclSuccess) {
|
||||
WARN("Rank %d requires NIC %d but that NIC is not available for rank %d", peerRank, netDev, rank);
|
||||
return ncclInvalidUsage;
|
||||
}
|
||||
*dev = netDev;
|
||||
if (dev) *dev = netDev;
|
||||
if (id) *id = netId;
|
||||
}
|
||||
if (pxnLevel == 1) {
|
||||
int g, n;
|
||||
NCCLCHECK(ncclTopoRankToIndex(comm->topo, rank, &g));
|
||||
NCCLCHECK(ncclTopoIdToIndex(comm->topo, NET, netDev, &n));
|
||||
NCCLCHECK(ncclTopoIdToIndex(comm->topo, NET, netId, &n));
|
||||
struct ncclTopoNode* gpu = comm->topo->nodes[GPU].nodes+g;
|
||||
if (gpu->paths[NET][n].type <= PATH_PXN) {
|
||||
*dev = netDev;
|
||||
if (dev) *dev = netDev;
|
||||
if (id) *id = netId;
|
||||
NCCLCHECK(ncclTopoGetIntermediateRank(comm->topo, rank, *dev, proxyRank));
|
||||
}
|
||||
} else if (pxnLevel == 2) {
|
||||
// Check which local GPU corresponds to that NIC and see if we can use PXN.
|
||||
int n, g1, g2;
|
||||
NCCLCHECK(ncclTopoIdToIndex(comm->topo, NET, netDev, &n));
|
||||
NCCLCHECK(ncclTopoIdToIndex(comm->topo, NET, netId, &n));
|
||||
NCCLCHECK(ncclTopoRankToIndex(comm->topo, rank, &g1));
|
||||
NCCLCHECK(ncclTopoGetLocalGpu(comm->topo, netDev, &g2));
|
||||
NCCLCHECK(ncclTopoGetLocalGpu(comm->topo, netId, &g2));
|
||||
if (g2 != -1) {
|
||||
struct ncclTopoNode* peerGpu = comm->topo->nodes[GPU].nodes+g2;
|
||||
if (peerGpu->paths[GPU][g1].type <= PATH_NVL && peerGpu->paths[NET][n].type <= PATH_PXB) {
|
||||
*proxyRank = peerGpu->gpu.rank;
|
||||
*dev = netDev;
|
||||
if (dev) *dev = netDev;
|
||||
if (id) *id = netId;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
@@ -1279,8 +1294,9 @@ ncclResult_t ncclTopoGetNetDev(struct ncclComm* comm, int rank, struct ncclTopoG
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetIntraNetDev(struct ncclTopoSystem* system, int rank, struct ncclTopoGraph* graph, int channelId, int type, int* dev) {
|
||||
*dev = -1;
|
||||
ncclResult_t ncclTopoGetIntraNetDev(struct ncclTopoSystem* system, int rank, struct ncclTopoGraph* graph, int channelId, int type, int64_t* id, int* dev) {
|
||||
if (dev) *dev = -1;
|
||||
if (id) *id = -1;
|
||||
if (graph && graph->nIntraChannels) {
|
||||
int n1 = -1;
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
@@ -1293,7 +1309,8 @@ ncclResult_t ncclTopoGetIntraNetDev(struct ncclTopoSystem* system, int rank, str
|
||||
}
|
||||
}
|
||||
if (n1 >= 0 && n1 < nnets) {
|
||||
*dev = n1;
|
||||
if (dev) *dev = n1;
|
||||
if (id) *id = NCCL_TOPO_ID(system->systemId, n1);
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
|
||||
+101
-43
@@ -16,6 +16,7 @@
|
||||
#include <fcntl.h>
|
||||
#include "xml.h"
|
||||
#include "cpuset.h"
|
||||
#include "bootstrap.h"
|
||||
|
||||
#define BUSID_SIZE (sizeof("0000:00:00.0"))
|
||||
#define BUSID_REDUCED_SIZE (sizeof("0000:00"))
|
||||
@@ -26,7 +27,7 @@ const char* topoLinkTypeStr[] = { "LOC", "XGMI", "", "PCI", "", "",
|
||||
const char* topoPathTypeStr[] = { "LOC", "XGMI", "NVB", "PIX", "PXB", "PXN", "PHB", "SYS", "DIS" };
|
||||
#else
|
||||
const char* topoLinkTypeStr[] = { "LOC", "NVL", "", "PCI", "", "", "", "SYS", "NET" };
|
||||
const char* topoPathTypeStr[] = { "LOC", "NVL", "NVB", "PIX", "PXB", "PXN", "PHB", "SYS", "DIS" };
|
||||
const char* topoPathTypeStr[] = { "LOC", "NVL", "NVB", "PIX", "PXB", "PXN", "PHB", "SYS", "NET", "DIS" };
|
||||
#endif
|
||||
|
||||
/******************************************************************/
|
||||
@@ -162,9 +163,13 @@ ncclResult_t ncclTopoRemoveNode(struct ncclTopoSystem* system, int type, int ind
|
||||
ncclResult_t ncclTopoConnectNodes(struct ncclTopoNode* node, struct ncclTopoNode* remNode, int type, float bw) {
|
||||
// Aggregate links into higher bw for NVLink
|
||||
struct ncclTopoLink* link;
|
||||
for (link = node->links; link->remNode; link++) {
|
||||
for (link = node->links; link - node->links != NCCL_TOPO_MAX_LINKS && link->remNode; link++) {
|
||||
if (link->remNode == remNode && link->type == type) break;
|
||||
}
|
||||
if (link - node->links == NCCL_TOPO_MAX_LINKS) {
|
||||
WARN("Error : too many Topo links (max %d)", NCCL_TOPO_MAX_LINKS);
|
||||
return ncclInternalError;
|
||||
}
|
||||
if (link->remNode == NULL) node->nlinks++;
|
||||
link->type = type;
|
||||
link->remNode = remNode;
|
||||
@@ -224,6 +229,10 @@ ncclResult_t ncclTopoFlattenBcmSwitches(struct ncclTopoSystem* system) {
|
||||
struct ncclTopoNode* remNode = sub->links[l].remNode;
|
||||
if (remNode == pciSwitch) continue;
|
||||
// Add link from parent PCI switch -> PCI device
|
||||
if (pciSwitch->nlinks == NCCL_TOPO_MAX_LINKS) {
|
||||
WARN("Error : too many Topo links (max %d)", NCCL_TOPO_MAX_LINKS);
|
||||
return ncclInternalError;
|
||||
}
|
||||
memcpy(pciSwitch->links+pciSwitch->nlinks, sub->links+l, sizeof(struct ncclTopoLink));
|
||||
pciSwitch->nlinks++;
|
||||
// Update link from PCI device -> parent PCI switch
|
||||
@@ -249,11 +258,13 @@ ncclResult_t ncclTopoFlattenBcmSwitches(struct ncclTopoSystem* system) {
|
||||
ncclResult_t ncclTopoConnectCpus(struct ncclTopoSystem* system) {
|
||||
// And connect all CPU nodes together
|
||||
for (int n=0; n<system->nodes[CPU].count; n++) {
|
||||
struct ncclTopoNode* cpu1 = system->nodes[CPU].nodes+n;
|
||||
for (int p=0; p<system->nodes[CPU].count; p++) {
|
||||
if (n == p) continue;
|
||||
struct ncclTopoNode* cpu2 = system->nodes[CPU].nodes+p;
|
||||
if (n == p || (NCCL_TOPO_ID_SYSTEM_ID(cpu1->id) != NCCL_TOPO_ID_SYSTEM_ID(cpu2->id))) continue;
|
||||
float bw;
|
||||
NCCLCHECK(ncclTopoGetInterCpuBw(system->nodes[CPU].nodes+n, &bw));
|
||||
NCCLCHECK(ncclTopoConnectNodes(system->nodes[CPU].nodes+n, system->nodes[CPU].nodes+p, LINK_SYS, bw));
|
||||
NCCLCHECK(ncclTopoGetInterCpuBw(cpu1, &bw));
|
||||
NCCLCHECK(ncclTopoConnectNodes(cpu1, cpu2, LINK_SYS, bw));
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
@@ -261,13 +272,13 @@ ncclResult_t ncclTopoConnectCpus(struct ncclTopoSystem* system) {
|
||||
|
||||
static ncclResult_t ncclTopoPrintRec(struct ncclTopoNode* node, struct ncclTopoNode* prevNode, char* line, int offset) {
|
||||
if (node->type == GPU) {
|
||||
sprintf(line+offset, "%s/%lX (%d)", topoNodeTypeStr[node->type], node->id, node->gpu.rank);
|
||||
sprintf(line+offset, "%s/%lx-%lx (%d)", topoNodeTypeStr[node->type], NCCL_TOPO_ID_SYSTEM_ID(node->id), NCCL_TOPO_ID_LOCAL_ID(node->id), node->gpu.rank);
|
||||
} else if (node->type == CPU) {
|
||||
sprintf(line+offset, "%s/%lX (%d/%d/%d)", topoNodeTypeStr[node->type], node->id, node->cpu.arch, node->cpu.vendor, node->cpu.model);
|
||||
sprintf(line+offset, "%s/%lx-%lx (%d/%d/%d)", topoNodeTypeStr[node->type], NCCL_TOPO_ID_SYSTEM_ID(node->id), NCCL_TOPO_ID_LOCAL_ID(node->id), node->cpu.arch, node->cpu.vendor, node->cpu.model);
|
||||
} else if (node->type == PCI) {
|
||||
sprintf(line+offset, "%s/%lX (%lx)", topoNodeTypeStr[node->type], node->id, node->pci.device);
|
||||
sprintf(line+offset, "%s/%lx-%lx (%lx)", topoNodeTypeStr[node->type], NCCL_TOPO_ID_SYSTEM_ID(node->id), NCCL_TOPO_ID_LOCAL_ID(node->id), node->pci.device);
|
||||
} else {
|
||||
sprintf(line+offset, "%s/%lX", topoNodeTypeStr[node->type], node->id);
|
||||
sprintf(line+offset, "%s/%lx-%lx", topoNodeTypeStr[node->type], NCCL_TOPO_ID_SYSTEM_ID(node->id), NCCL_TOPO_ID_LOCAL_ID(node->id));
|
||||
}
|
||||
INFO(NCCL_GRAPH, "%s", line);
|
||||
for (int i=0; i<offset; i++) line[i] = ' ';
|
||||
@@ -334,12 +345,13 @@ ncclResult_t ncclTopoSortSystem(struct ncclTopoSystem* system) {
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoAddNet(struct ncclXmlNode* xmlNet, struct ncclTopoSystem* system, struct ncclTopoNode* nic, int64_t busId) {
|
||||
ncclResult_t ncclTopoAddNet(struct ncclXmlNode* xmlNet, struct ncclTopoSystem* system, struct ncclTopoNode* nic, int systemId, int64_t busId) {
|
||||
int dev;
|
||||
NCCLCHECK(xmlGetAttrInt(xmlNet, "dev", &dev));
|
||||
|
||||
struct ncclTopoNode* net;
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &net, NET, dev));
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &net, NET, NCCL_TOPO_ID(systemId, dev)));
|
||||
net->net.dev = dev;
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttr(xmlNet, "guid", &str));
|
||||
if (str) sscanf(str, "0x%lx", &net->net.asic);
|
||||
@@ -355,7 +367,6 @@ ncclResult_t ncclTopoAddNet(struct ncclXmlNode* xmlNet, struct ncclTopoSystem* s
|
||||
NCCLCHECK(xmlGetAttrIntDefault(xmlNet, "gdr", &net->net.gdrSupport, 0));
|
||||
NCCLCHECK(xmlGetAttrIntDefault(xmlNet, "maxconn", &net->net.maxChannels, MAXCHANNELS));
|
||||
NCCLCHECK(xmlGetAttrIntDefault(xmlNet, "coll", &net->net.collSupport, 0));
|
||||
NCCLCHECK(xmlGetAttrIntDefault(xmlNet, "dev", &net->net.dev, 0));
|
||||
net->net.busId = busId;
|
||||
ncclDebugNoWarn = 0;
|
||||
|
||||
@@ -364,14 +375,14 @@ ncclResult_t ncclTopoAddNet(struct ncclXmlNode* xmlNet, struct ncclTopoSystem* s
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoAddNic(struct ncclXmlNode* xmlNic, struct ncclTopoSystem* system, struct ncclTopoNode* nic, int64_t busId) {
|
||||
ncclResult_t ncclTopoAddNic(struct ncclXmlNode* xmlNic, struct ncclTopoSystem* system, struct ncclTopoNode* nic, int systemId, int64_t busId) {
|
||||
for (int s=0; s<xmlNic->nSubs; s++) {
|
||||
struct ncclXmlNode* xmlNet = xmlNic->subs[s];
|
||||
if (strcmp(xmlNet->name, "net") != 0) continue;
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(xmlNet, "dev", &index));
|
||||
if (index == -1) continue;
|
||||
NCCLCHECK(ncclTopoAddNet(xmlNet, system, nic, busId));
|
||||
NCCLCHECK(ncclTopoAddNet(xmlNet, system, nic, systemId, busId));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -405,7 +416,7 @@ struct kvDict kvDictPciGen[] = {
|
||||
{ "2.5 GT/s", 15 }, { "5 GT/s", 30 }, { "8 GT/s", 60 }, { "16 GT/s", 120 }, { "32 GT/s", 240 }, /* Kernel 5.6 and earlier */
|
||||
{ "2.5 GT/s PCIe", 15 }, { "5.0 GT/s PCIe", 30 }, { "8.0 GT/s PCIe", 60 }, { "16.0 GT/s PCIe", 120 }, { "32.0 GT/s PCIe", 240 }, { "64.0 GT/s PCIe", 480 },
|
||||
{ NULL, 60 /* Default fallback */ } }; // x100 Mbps per lane
|
||||
ncclResult_t ncclTopoAddPci(struct ncclXmlNode* xmlPci, struct ncclTopoSystem* system, struct ncclTopoNode* parent) {
|
||||
ncclResult_t ncclTopoAddPci(struct ncclXmlNode* xmlPci, struct ncclTopoSystem* system, struct ncclTopoNode* parent, int systemId) {
|
||||
const char* str;
|
||||
|
||||
int type;
|
||||
@@ -424,7 +435,7 @@ ncclResult_t ncclTopoAddPci(struct ncclXmlNode* xmlPci, struct ncclTopoSystem* s
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(xmlGpu, "rank", &index));
|
||||
if (index == -1) return ncclSuccess;
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &node, type, busId));
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &node, type, NCCL_TOPO_ID(systemId, busId)));
|
||||
NCCLCHECK(ncclTopoAddGpu(xmlGpu, system, node));
|
||||
}
|
||||
struct ncclXmlNode* xmlNic = NULL;
|
||||
@@ -434,14 +445,15 @@ ncclResult_t ncclTopoAddPci(struct ncclXmlNode* xmlPci, struct ncclTopoSystem* s
|
||||
// Ignore sub device ID and merge multi-port NICs into one PCI device.
|
||||
busId &= 0xfffffffffffffff0;
|
||||
struct ncclTopoNode* nicNode = NULL;
|
||||
NCCLCHECK(ncclTopoGetNode(system, &nicNode, type, busId));
|
||||
int64_t id = NCCL_TOPO_ID(systemId, busId);
|
||||
NCCLCHECK(ncclTopoGetNode(system, &nicNode, type, id));
|
||||
if (nicNode == NULL) {
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &nicNode, type, busId));
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &nicNode, type, id));
|
||||
node = nicNode; // Connect it to parent later on
|
||||
}
|
||||
NCCLCHECK(ncclTopoAddNic(xmlNic, system, nicNode, busId));
|
||||
NCCLCHECK(ncclTopoAddNic(xmlNic, system, nicNode, systemId, busId));
|
||||
} else if (type == PCI) {
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &node, type, busId));
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &node, type, NCCL_TOPO_ID(systemId, busId)));
|
||||
NCCLCHECK(xmlGetAttr(xmlPci, "vendor", &str));
|
||||
if (str) node->pci.device += strtol(str, NULL, 0) << 48;
|
||||
NCCLCHECK(xmlGetAttr(xmlPci, "device", &str));
|
||||
@@ -453,7 +465,7 @@ ncclResult_t ncclTopoAddPci(struct ncclXmlNode* xmlPci, struct ncclTopoSystem* s
|
||||
|
||||
for (int s=0; s<xmlPci->nSubs; s++) {
|
||||
struct ncclXmlNode* xmlSubPci = xmlPci->subs[s];
|
||||
NCCLCHECK(ncclTopoAddPci(xmlSubPci, system, node));
|
||||
NCCLCHECK(ncclTopoAddPci(xmlSubPci, system, node, systemId));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -475,11 +487,25 @@ ncclResult_t ncclTopoAddPci(struct ncclXmlNode* xmlPci, struct ncclTopoSystem* s
|
||||
struct kvDict kvDictCpuArch[] = { { "x86_64", NCCL_TOPO_CPU_ARCH_X86 }, { "arm64", NCCL_TOPO_CPU_ARCH_ARM }, { "ppc64", NCCL_TOPO_CPU_ARCH_POWER }, { NULL, 0 } };
|
||||
struct kvDict kvDictCpuVendor[] = { { "GenuineIntel", NCCL_TOPO_CPU_VENDOR_INTEL }, { "AuthenticAMD", NCCL_TOPO_CPU_VENDOR_AMD }, { "CentaurHauls", NCCL_TOPO_CPU_VENDOR_ZHAOXIN }, { " Shanghai ", NCCL_TOPO_CPU_VENDOR_ZHAOXIN }, { NULL, 0 } };
|
||||
|
||||
ncclResult_t ncclGetSystemId(struct ncclTopoSystem* system, struct ncclXmlNode* xmlCpu, int* systemIdPtr) {
|
||||
const char* hostHashStr;
|
||||
NCCLCHECK(xmlGetAttr(xmlCpu, "host_hash", &hostHashStr));
|
||||
uint64_t hostHash = hostHashStr ? strtoull(hostHashStr, NULL, 16) : 0;
|
||||
int systemId;
|
||||
for (systemId=0; systemId<system->nHosts; systemId++) if (system->hostHashes[systemId] == hostHash) break;
|
||||
if (systemId == system->nHosts) system->hostHashes[system->nHosts++] = hostHash;
|
||||
*systemIdPtr = systemId;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
|
||||
ncclResult_t ncclTopoAddCpu(struct ncclXmlNode* xmlCpu, struct ncclTopoSystem* system) {
|
||||
int numaId;
|
||||
NCCLCHECK(xmlGetAttrInt(xmlCpu, "numaid", &numaId));
|
||||
int systemId;
|
||||
NCCLCHECK(ncclGetSystemId(system, xmlCpu, &systemId));
|
||||
struct ncclTopoNode* cpu;
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &cpu, CPU, numaId));
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &cpu, CPU, NCCL_TOPO_ID(systemId, numaId)));
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttr(xmlCpu, "affinity", &str));
|
||||
if (str != NULL) {
|
||||
@@ -512,16 +538,16 @@ ncclResult_t ncclTopoAddCpu(struct ncclXmlNode* xmlCpu, struct ncclTopoSystem* s
|
||||
}
|
||||
for (int s=0; s<xmlCpu->nSubs; s++) {
|
||||
struct ncclXmlNode* node = xmlCpu->subs[s];
|
||||
if (strcmp(node->name, "pci") == 0) NCCLCHECK(ncclTopoAddPci(node, system, cpu));
|
||||
if (strcmp(node->name, "pci") == 0) NCCLCHECK(ncclTopoAddPci(node, system, cpu, systemId));
|
||||
if (strcmp(node->name, "nic") == 0) {
|
||||
struct ncclTopoNode* nic = NULL;
|
||||
NCCLCHECK(ncclTopoGetNode(system, &nic, NIC, 0));
|
||||
if (nic == NULL) {
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &nic, NIC, 0));
|
||||
NCCLCHECK(ncclTopoCreateNode(system, &nic, NIC, NCCL_TOPO_ID(systemId, 0)));
|
||||
NCCLCHECK(ncclTopoConnectNodes(cpu, nic, LINK_PCI, LOC_BW));
|
||||
NCCLCHECK(ncclTopoConnectNodes(nic, cpu, LINK_PCI, LOC_BW));
|
||||
}
|
||||
NCCLCHECK(ncclTopoAddNic(node, system, nic, 0));
|
||||
NCCLCHECK(ncclTopoAddNic(node, system, nic, systemId, 0));
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
@@ -579,11 +605,12 @@ ncclResult_t ncclTopoAddXGMI(struct ncclXmlNode* node, struct ncclTopoSystem* sy
|
||||
return ncclSuccess;
|
||||
}
|
||||
#else
|
||||
ncclResult_t ncclTopoAddNvLinks(struct ncclXmlNode* node, struct ncclTopoSystem* system, const char* parentBusId) {
|
||||
ncclResult_t ncclTopoAddNvLinks(struct ncclXmlNode* node, struct ncclTopoSystem* system, const char* parentBusId, int systemId) {
|
||||
if (strcmp(node->name, "nvlink") == 0) {
|
||||
struct ncclTopoNode* gpu = NULL;
|
||||
int64_t pBusId;
|
||||
NCCLCHECK(busIdToInt64(parentBusId, &pBusId));
|
||||
pBusId = NCCL_TOPO_ID(systemId, pBusId);
|
||||
NCCLCHECK(ncclTopoGetNode(system, &gpu, GPU, pBusId));
|
||||
if (gpu == NULL) {
|
||||
WARN("Add NVLink error : could not find GPU %lx", pBusId);
|
||||
@@ -602,7 +629,7 @@ ncclResult_t ncclTopoAddNvLinks(struct ncclXmlNode* node, struct ncclTopoSystem*
|
||||
NCCLCHECK(xmlGetAttrStr(node, "target", &target));
|
||||
int64_t busId;
|
||||
NCCLCHECK(busIdToInt64(target, &busId));
|
||||
NCCLCHECK(ncclTopoGetNode(system, &remote, GPU, busId));
|
||||
NCCLCHECK(ncclTopoGetNode(system, &remote, GPU, NCCL_TOPO_ID(systemId, busId)));
|
||||
} else if (targetType == CPU) {
|
||||
// NVL connection to the local CPU
|
||||
NCCLCHECK(findLocalCpu(gpu, &remote));
|
||||
@@ -621,21 +648,25 @@ ncclResult_t ncclTopoAddNvLinks(struct ncclXmlNode* node, struct ncclTopoSystem*
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (strcmp(node->name, "cpu") == 0) {
|
||||
NCCLCHECK(ncclGetSystemId(system, node, &systemId));
|
||||
}
|
||||
const char* busId;
|
||||
NCCLCHECK(xmlGetAttr(node, "busid", &busId));
|
||||
for (int s=0; s<node->nSubs; s++) {
|
||||
NCCLCHECK(ncclTopoAddNvLinks(node->subs[s], system, busId ? busId : parentBusId));
|
||||
NCCLCHECK(ncclTopoAddNvLinks(node->subs[s], system, busId ? busId : parentBusId, systemId));
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
#endif
|
||||
|
||||
ncclResult_t ncclTopoAddC2c(struct ncclXmlNode* node, struct ncclTopoSystem* system, const char* parentBusId) {
|
||||
ncclResult_t ncclTopoAddC2c(struct ncclXmlNode* node, struct ncclTopoSystem* system, const char* parentBusId, int systemId) {
|
||||
if (strcmp(node->name, "c2c") == 0) {
|
||||
struct ncclTopoNode* gpu = NULL;
|
||||
int64_t pBusId;
|
||||
NCCLCHECK(busIdToInt64(parentBusId, &pBusId));
|
||||
pBusId = NCCL_TOPO_ID(systemId, pBusId);
|
||||
NCCLCHECK(ncclTopoGetNode(system, &gpu, GPU, pBusId));
|
||||
if (gpu == NULL) {
|
||||
WARN("Add NVLink error : could not find GPU %lx", pBusId);
|
||||
@@ -652,29 +683,35 @@ ncclResult_t ncclTopoAddC2c(struct ncclXmlNode* node, struct ncclTopoSystem* sys
|
||||
NCCLCHECK(ncclTopoConnectNodes(gpu, cpu, LINK_NVL, c2cBw));
|
||||
NCCLCHECK(ncclTopoConnectNodes(cpu, gpu, LINK_NVL, c2cBw));
|
||||
} else {
|
||||
if (strcmp(node->name, "cpu") == 0) {
|
||||
NCCLCHECK(ncclGetSystemId(system, node, &systemId));
|
||||
}
|
||||
const char* busId;
|
||||
NCCLCHECK(xmlGetAttr(node, "busid", &busId));
|
||||
for (int s=0; s<node->nSubs; s++) {
|
||||
NCCLCHECK(ncclTopoAddC2c(node->subs[s], system, busId ? busId : parentBusId));
|
||||
NCCLCHECK(ncclTopoAddC2c(node->subs[s], system, busId ? busId : parentBusId, systemId));
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetSystemFromXml(struct ncclXml* xml, struct ncclTopoSystem** topoSystem) {
|
||||
ncclResult_t ncclTopoGetSystemFromXml(struct ncclXml* xml, struct ncclTopoSystem** topoSystem, const uint64_t localHostHash) {
|
||||
NCCLCHECK(ncclCalloc(topoSystem, 1));
|
||||
struct ncclTopoSystem* system = *topoSystem;
|
||||
struct ncclXmlNode* topNode;
|
||||
NCCLCHECK(xmlFindTag(xml, "system", &topNode));
|
||||
for (int s=0; s<topNode->nSubs; s++) {
|
||||
struct ncclXmlNode* node = topNode->subs[s];
|
||||
if (strcmp(node->name, "cpu") == 0) NCCLCHECK(ncclTopoAddCpu(node, *topoSystem));
|
||||
}
|
||||
for (int systemId=0; systemId<system->nHosts; systemId++) if (system->hostHashes[systemId] == localHostHash) system->systemId = systemId;
|
||||
|
||||
#if defined(__HIP_PLATFORM_AMD__) || defined(__HIPCC__)
|
||||
NCCLCHECK(ncclTopoAddXGMI(topNode, *topoSystem, NULL));
|
||||
#else
|
||||
NCCLCHECK(ncclTopoAddNvLinks(topNode, *topoSystem, NULL));
|
||||
NCCLCHECK(ncclTopoAddNvLinks(topNode, *topoSystem, NULL, 0));
|
||||
#endif
|
||||
NCCLCHECK(ncclTopoAddC2c(topNode, *topoSystem, NULL));
|
||||
NCCLCHECK(ncclTopoAddC2c(topNode, *topoSystem, NULL, 0));
|
||||
|
||||
NCCLCHECK(ncclTopoFlattenBcmSwitches(*topoSystem));
|
||||
NCCLCHECK(ncclTopoConnectCpus(*topoSystem));
|
||||
@@ -720,7 +757,7 @@ static ncclResult_t xmlInitAttrFloat(struct ncclXmlNode* node, const char* attrN
|
||||
|
||||
ncclResult_t ncclTopoGetSystem(struct ncclComm* comm, struct ncclTopoSystem** system) {
|
||||
struct ncclXml* xml;
|
||||
NCCLCHECK(ncclCalloc(&xml, 1));
|
||||
NCCLCHECK(xmlAlloc(&xml, NCCL_TOPO_XML_MAX_NODES));
|
||||
const char* xmlTopoFile = ncclGetEnv("NCCL_TOPO_FILE");
|
||||
if (xmlTopoFile) {
|
||||
INFO(NCCL_ENV, "NCCL_TOPO_FILE set by environment to %s", xmlTopoFile);
|
||||
@@ -794,13 +831,32 @@ ncclResult_t ncclTopoGetSystem(struct ncclComm* comm, struct ncclTopoSystem** sy
|
||||
// Remove XML branches which don't have a node with keep="1" (typically when importing a topology)
|
||||
NCCLCHECK(ncclTopoTrimXml(xml));
|
||||
|
||||
if (comm->MNNVL) {
|
||||
// MNNVL clique support
|
||||
char* mem;
|
||||
NCCLCHECK(ncclCalloc(&mem, comm->clique.size * xmlMemSize(NCCL_TOPO_XML_MAX_NODES)));
|
||||
struct ncclXml* rankXml = (struct ncclXml*)(mem+xmlMemSize(NCCL_TOPO_XML_MAX_NODES)*comm->cliqueRank);
|
||||
memcpy(rankXml, xml, xmlMemSize(NCCL_TOPO_XML_MAX_NODES));
|
||||
NCCLCHECK(ncclTopoConvertXml(rankXml, (uintptr_t)xml->nodes, 1));
|
||||
NCCLCHECK(bootstrapIntraNodeAllGather(comm->bootstrap, comm->clique.ranks, comm->cliqueRank, comm->clique.size, mem, xmlMemSize(NCCL_TOPO_XML_MAX_NODES)));
|
||||
struct ncclXml* cliqueXml;
|
||||
NCCLCHECK(xmlAlloc(&cliqueXml, comm->clique.size*NCCL_TOPO_XML_MAX_NODES));
|
||||
for (int i = 0; i < comm->clique.size; i++) {
|
||||
struct ncclXml* peerXml = (struct ncclXml*)(mem+xmlMemSize(NCCL_TOPO_XML_MAX_NODES)*i);
|
||||
NCCLCHECK(ncclTopoConvertXml(peerXml, (uintptr_t)peerXml->nodes, 0));
|
||||
NCCLCHECK(ncclTopoFuseXml(cliqueXml, peerXml));
|
||||
}
|
||||
free(xml);
|
||||
xml = cliqueXml;
|
||||
}
|
||||
|
||||
xmlTopoFile = ncclGetEnv("NCCL_TOPO_DUMP_FILE");
|
||||
if (xmlTopoFile && comm->rank == ncclParamTopoDumpFileRank()) {
|
||||
INFO(NCCL_ENV, "NCCL_TOPO_DUMP_FILE set by environment to %s", xmlTopoFile);
|
||||
NCCLCHECK(ncclTopoDumpXmlToFile(xmlTopoFile, xml));
|
||||
}
|
||||
|
||||
NCCLCHECK(ncclTopoGetSystemFromXml(xml, system));
|
||||
NCCLCHECK(ncclTopoGetSystemFromXml(xml, system, comm->peerInfo[comm->rank].hostHash));
|
||||
free(xml);
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -848,7 +904,7 @@ ncclResult_t getLocalNetCountByBw(struct ncclTopoSystem* system, int gpu, int *c
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetLocalNet(struct ncclTopoSystem* system, int rank, int channelId, int* id) {
|
||||
ncclResult_t ncclTopoGetLocalNet(struct ncclTopoSystem* system, int rank, int channelId, int64_t* id, int* dev) {
|
||||
int gpu;
|
||||
NCCLCHECK(ncclTopoRankToIndex(system, rank, &gpu));
|
||||
int* localNets = NULL;
|
||||
@@ -871,19 +927,21 @@ ncclResult_t ncclTopoGetLocalNet(struct ncclTopoSystem* system, int rank, int ch
|
||||
}
|
||||
if (isPow2(localNetCount)) net = mirrorBits(net, localNetCount);
|
||||
if (localNetCount == 0) {
|
||||
*id = -1;
|
||||
if (id) *id = -1;
|
||||
if (dev) *dev = -1;
|
||||
} else {
|
||||
net += channelId%(DIVUP(localNetCount,localGpuCount));
|
||||
*id = system->nodes[NET].nodes[localNets[net%localNetCount]].id;
|
||||
if (id) *id = system->nodes[NET].nodes[localNets[net%localNetCount]].id;
|
||||
if (dev) *dev = system->nodes[NET].nodes[localNets[net%localNetCount]].net.dev;
|
||||
}
|
||||
free(localNets);
|
||||
free(localGpus);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetLocalGpu(struct ncclTopoSystem* system, int net, int* gpuIndex) {
|
||||
ncclResult_t ncclTopoGetLocalGpu(struct ncclTopoSystem* system, int64_t netId, int* gpuIndex) {
|
||||
int netIndex;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, net, &netIndex));
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netId, &netIndex));
|
||||
int* localGpus = NULL;
|
||||
int localGpuCount;
|
||||
NCCLCHECK(ncclTopoGetLocal(system, NET, netIndex, GPU, &localGpus, &localGpuCount, NULL));
|
||||
@@ -891,9 +949,9 @@ ncclResult_t ncclTopoGetLocalGpu(struct ncclTopoSystem* system, int net, int* gp
|
||||
for (int lg=0; lg<localGpuCount; lg++) {
|
||||
int g = localGpus[lg];
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
int id;
|
||||
NCCLCHECK(ncclTopoGetLocalNet(system, gpu->gpu.rank, c, &id));
|
||||
if (net == id) {
|
||||
int64_t id;
|
||||
NCCLCHECK(ncclTopoGetLocalNet(system, gpu->gpu.rank, c, &id, NULL));
|
||||
if (netId == id) {
|
||||
*gpuIndex = g;
|
||||
free(localGpus);
|
||||
return ncclSuccess;
|
||||
|
||||
+27
-6
@@ -94,7 +94,7 @@ struct ncclTopoLink {
|
||||
float bw;
|
||||
struct ncclTopoNode* remNode;
|
||||
};
|
||||
#define NCCL_TOPO_MAX_LINKS 128 //Changed the value from 32 to 128 for CPX mode
|
||||
#define NCCL_TOPO_MAX_LINKS 128
|
||||
|
||||
#define NCCL_TOPO_MAX_HOPS (NCCL_TOPO_MAX_NODES*NCCL_TOPO_NODE_TYPES)
|
||||
|
||||
@@ -110,6 +110,10 @@ struct ncclTopoLinkList {
|
||||
|
||||
#define NCCL_TOPO_UNDEF (-1)
|
||||
|
||||
#define NCCL_TOPO_ID_SYSTEM_ID(id) (id >> 56)
|
||||
#define NCCL_TOPO_ID_LOCAL_ID(id) (id & 0x00ffffffffffffff)
|
||||
#define NCCL_TOPO_ID(systemid, localid) (((int64_t)systemid << 56) + localid)
|
||||
|
||||
#define RCCL_TOPO_CR8G 1
|
||||
#define RCCL_TOPO_4P2H_ROME 2
|
||||
#define RCCL_TOPO_GDR_ALL 4
|
||||
@@ -132,6 +136,7 @@ struct ncclTopoNode {
|
||||
int cu;
|
||||
}gpu;
|
||||
struct {
|
||||
int dev; // Plugin dev number
|
||||
uint64_t asic;
|
||||
int port;
|
||||
float bw;
|
||||
@@ -140,7 +145,6 @@ struct ncclTopoNode {
|
||||
int collSupport;
|
||||
int maxChannels;
|
||||
int64_t busId;
|
||||
int dev;
|
||||
}net;
|
||||
struct {
|
||||
int arch;
|
||||
@@ -166,6 +170,9 @@ struct ncclTopoNodeSet {
|
||||
};
|
||||
|
||||
struct ncclTopoSystem {
|
||||
int systemId;
|
||||
uint64_t hostHashes[NCCL_TOPO_MAX_NODES];
|
||||
int nHosts;
|
||||
struct ncclTopoNodeSet nodes[NCCL_TOPO_NODE_TYPES];
|
||||
float maxBw;
|
||||
float totalBw;
|
||||
@@ -181,8 +188,7 @@ struct ncclTopoSystem {
|
||||
float baseBw;
|
||||
bool mscclEnabled;
|
||||
|
||||
// [RCCL] Track hostIdx and number of hosts to support rail-optimized rings/trees
|
||||
int nHosts;
|
||||
// [RCCL] Track hostIdx to support rail-optimized rings/trees
|
||||
int hostIdx;
|
||||
};
|
||||
|
||||
@@ -192,9 +198,11 @@ ncclResult_t ncclTopoRemoveNode(struct ncclTopoSystem* system, int type, int id)
|
||||
ncclResult_t ncclTopoConnectNodes(struct ncclTopoNode* node, struct ncclTopoNode* remNode, int type, float bw);
|
||||
ncclResult_t ncclTopoPrintPaths(struct ncclTopoSystem* system);
|
||||
ncclResult_t ncclTopoLoadSystem(const char* xmlTopoFile, struct ncclTopoSystem* system);
|
||||
ncclResult_t ncclTopoGetIntermediateRank(struct ncclTopoSystem* system, int rank, int netDev, int* intermediateRank);
|
||||
ncclResult_t ncclTopoGetIntermediateRank(struct ncclTopoSystem* system, int rank, int64_t netId, int* intermediateRank);
|
||||
|
||||
ncclResult_t ncclTopoGetSystemFromXml(struct ncclXml* xml, struct ncclTopoSystem** topoSystem);
|
||||
#define NCCL_TOPO_XML_MAX_NODES 256
|
||||
#define NCCL_GRAPH_XML_MAX_NODES 4096
|
||||
ncclResult_t ncclTopoGetSystemFromXml(struct ncclXml* xml, struct ncclTopoSystem** topoSystem, uint64_t localHostHash);
|
||||
ncclResult_t ncclTopoGetGraphFromXml(struct ncclXmlNode *xmlGraphs, struct ncclTopoSystem* system, struct ncclTopoGraph* graph, int* nChannels);
|
||||
ncclResult_t ncclTopoGetXmlFromGraphs(int ngraphs, struct ncclTopoGraph** graphs, struct ncclTopoSystem* system, struct ncclXml *xml);
|
||||
|
||||
@@ -225,6 +233,7 @@ static ncclResult_t ncclTopoRankToIndex(struct ncclTopoSystem* system, int rank,
|
||||
static ncclResult_t ncclTopoDevToRank(struct ncclTopoSystem* system, int dev, int* rank) {
|
||||
*rank = -1;
|
||||
for (int i=0; i<system->nodes[GPU].count; i++) {
|
||||
if (NCCL_TOPO_ID_SYSTEM_ID(system->nodes[GPU].nodes[i].id) != system->systemId) continue; // Only consider GPUs on our node
|
||||
if (system->nodes[GPU].nodes[i].gpu.dev == dev) {
|
||||
*rank = system->nodes[GPU].nodes[i].gpu.rank;
|
||||
return ncclSuccess;
|
||||
@@ -233,6 +242,18 @@ static ncclResult_t ncclTopoDevToRank(struct ncclTopoSystem* system, int dev, in
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
static ncclResult_t ncclTopoIdToNetDev(struct ncclTopoSystem* system, int64_t id, int* netDev) {
|
||||
*netDev = -1;
|
||||
for (int i=0; i<system->nodes[NET].count; i++) {
|
||||
if (system->nodes[NET].nodes[i].id == id) {
|
||||
*netDev = system->nodes[NET].nodes[i].net.dev;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
WARN("Could not find NET with id %lx\n", id);
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
// Returns XGMI speed in GB/s
|
||||
static float ncclTopoXGMISpeed(const char* gcn) {
|
||||
if (IsArchMatch(gcn, "gfx90a"))
|
||||
|
||||
+7
-10
@@ -269,7 +269,7 @@ static struct tuningModel rcclTuningModel[] = {
|
||||
static const double llMaxBws[3][3] = {
|
||||
/* Volta-N1/Intel-N2/Intel-N4) */ {39.0, 39.0, 20.4},
|
||||
/* Ampere-N1/AMD-N2/AMD-N4) */ {87.7, 22.5 /*avg of ring & tree*/, 19.0},
|
||||
/* Hopper-N1/AMD-N2/AMD-N4) */ {87.7, 22.5 /*avg of ring & tree*/, 19.0}
|
||||
/* Hopper-N1/AMD-N2/AMD-N4) */ {141.0, 45.0 /*avg of ring & tree*/, 35.0}
|
||||
};
|
||||
|
||||
static const double perChMaxRingLL128Bws[3][3] = {
|
||||
@@ -325,8 +325,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
getNthreads("NCCL_LL128_NTHREADS", ncclParamLl128Nthreads(), NCCL_LL128_MAX_NTHREADS/4, NCCL_LL128_MAX_NTHREADS, NCCL_LL128_MAX_NTHREADS);
|
||||
#endif
|
||||
|
||||
// MNNVL support - treat as a single NVLink connected node
|
||||
int nNodes = comm->MNNVL ? 1 : comm->nNodes;
|
||||
int nNodes = comm->nNodes;
|
||||
int nRanks = comm->nRanks;
|
||||
if (nRanks <= 1) return ncclSuccess;
|
||||
|
||||
@@ -381,7 +380,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
busBw *= rcclTuningModel[comm->topo->tuning].bwRatio[1][a][p];
|
||||
if (a == NCCL_ALGO_RING && p == NCCL_PROTO_LL && (coll == ncclFuncBroadcast || coll == ncclFuncReduce) && IsArchMatch(comm->topo->nodes[GPU].nodes[0].gpu.gcn, "gfx94") && comm->topo->nodes[GPU].count == comm->topo->nRanks) { busBw = busBw * 1.65; }
|
||||
#else
|
||||
if (a == NCCL_ALGO_RING && p == NCCL_PROTO_LL) { busBw = std::min(llMaxBw, busBw * ((nNodes > 1 || coll == ncclFuncAllReduce || coll == ncclFuncReduce) ? 1.0/4.0 : 1.0/3.0)); }
|
||||
if (a == NCCL_ALGO_RING && p == NCCL_PROTO_LL) { busBw = std::min(llMaxBw, busBw * .5); }
|
||||
if (a == NCCL_ALGO_RING && p == NCCL_PROTO_LL128) busBw = std::min(busBw * (ppn < 2 ? 0.7 : 0.92 /*120.0/128.0*/), graphs[a]->nChannels*perChMaxRingLL128Bw);
|
||||
if (a == NCCL_ALGO_TREE) busBw = std::min(busBw*.92, graphs[a]->nChannels*perChMaxTreeBw);
|
||||
if (a == NCCL_ALGO_TREE && p == NCCL_PROTO_LL) busBw = std::min(busBw*1.0/3.8, llMaxBw);
|
||||
@@ -393,7 +392,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
if (coll == ncclFuncAllGather || coll == ncclFuncReduceScatter) {
|
||||
busBw = ppn * bw;
|
||||
// AllGather/ReduceScatter requires 1:1 GPU:NIC
|
||||
int nicPerNode = comm->collNetHeadsUniqueNum;
|
||||
int nicPerNode = comm->collNetHeadsNum;
|
||||
if (coll == ncclFuncAllGather && comm->nNodes > 1) {
|
||||
if (!comm->ncclCollNet || !comm->ncclCollNet->iallgather || ppn > nicPerNode) busBw = 0;
|
||||
}
|
||||
@@ -485,15 +484,13 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
NCCLCHECK(parseList(algoStr, ncclAlgoStr, NCCL_NUM_ALGORITHMS, algoEnable));
|
||||
}
|
||||
|
||||
// MNNVL: NVLS not yet supported
|
||||
if (comm->nNodes == 1 || comm->MNNVL) algoEnable[NCCL_ALGO_NVLS_TREE] = 0;
|
||||
if (comm->nNodes == 1) algoEnable[NCCL_ALGO_NVLS_TREE] = 0;
|
||||
|
||||
// Disable CollNet if it is not supported
|
||||
if (comm->collNetSupport == 0) {
|
||||
algoEnable[NCCL_ALGO_COLLNET_DIRECT] = 0;
|
||||
algoEnable[NCCL_ALGO_COLLNET_CHAIN] = 0;
|
||||
// MNNVL: NVLS not yet supported
|
||||
if (comm->nNodes > 1 || comm->MNNVL) algoEnable[NCCL_ALGO_NVLS] = 0;
|
||||
if (nNodes > 1) algoEnable[NCCL_ALGO_NVLS] = 0;
|
||||
// If user has hard set NCCL_ALGO=COLLNET, ignore it
|
||||
if (algoEnable[NCCL_ALGO_RING] == 0 && algoEnable[NCCL_ALGO_TREE] == 0 &&
|
||||
algoEnable[NCCL_ALGO_NVLS] == 0 && algoEnable[NCCL_ALGO_NVLS_TREE] == 0) {
|
||||
@@ -662,7 +659,7 @@ ncclResult_t ncclTopoGetAlgoTime(struct ncclInfo* info, int algorithm, int proto
|
||||
#else
|
||||
if (algorithm == NCCL_ALGO_TREE && logSize < 23) bw *= treeCorrectionFactor[protocol][logSize];
|
||||
if (info->nChannels != 0) bw = bw / info->comm->nChannels * info->nChannels;
|
||||
if (algorithm == NCCL_ALGO_RING && protocol == NCCL_PROTO_SIMPLE && (!info->comm->MNNVL && info->comm->nNodes > 1)
|
||||
if (algorithm == NCCL_ALGO_RING && protocol == NCCL_PROTO_SIMPLE && info->comm->nNodes > 1
|
||||
&& info->coll == ncclFuncAllReduce && info->nBytes/(info->comm->nChannels*info->comm->nRanks) >= 64) {
|
||||
lat *= info->comm->minCompCap < 80 ? 1.9 : 1.4; // Plateau effect of ring
|
||||
}
|
||||
|
||||
+86
-3
@@ -175,8 +175,8 @@ struct xmlHandler {
|
||||
ncclResult_t xmlLoadSub(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head, struct xmlHandler handlers[], int nHandlers) {
|
||||
if (head && head->type == NODE_TYPE_SINGLE) return ncclSuccess;
|
||||
while (1) {
|
||||
if (xml->maxIndex == MAX_NODES) {
|
||||
WARN("Error : XML parser is limited to 1024 nodes");
|
||||
if (xml->maxIndex == xml->maxNodes) {
|
||||
WARN("Error : XML parser is limited to %d nodes", xml->maxNodes);
|
||||
return ncclInternalError;
|
||||
}
|
||||
struct ncclXmlNode* node = xml->nodes+xml->maxIndex;
|
||||
@@ -201,7 +201,13 @@ ncclResult_t xmlLoadSub(FILE* file, struct ncclXml* xml, struct ncclXmlNode* hea
|
||||
int found = 0;
|
||||
for (int h=0; h<nHandlers; h++) {
|
||||
if (strcmp(node->name, handlers[h].name) == 0) {
|
||||
if (head) head->subs[head->nSubs++] = node;
|
||||
if (head) {
|
||||
if (head->nSubs == MAX_SUBS) {
|
||||
WARN("Error : XML parser is limited to %d subnodes", MAX_SUBS);
|
||||
return ncclInternalError;
|
||||
}
|
||||
head->subs[head->nSubs++] = node;
|
||||
}
|
||||
node->parent = head;
|
||||
node->nSubs = 0;
|
||||
xml->maxIndex++;
|
||||
@@ -221,6 +227,23 @@ ncclResult_t xmlLoadSub(FILE* file, struct ncclXml* xml, struct ncclXmlNode* hea
|
||||
/* XML Writer */
|
||||
/**************/
|
||||
|
||||
// exp == 1 -- serialize; exp == 0 -- deserialize
|
||||
ncclResult_t ncclTopoConvertXml(struct ncclXml* xml, uintptr_t base, int exp) {
|
||||
for (int n = 0; n < xml->maxIndex; n++) {
|
||||
struct ncclXmlNode *node = &xml->nodes[n];
|
||||
|
||||
// For "parent", we shift the base by 1 so that we can distinguish actual
|
||||
// NULL pointers from pointers pointing to the first node.
|
||||
if (node->parent)
|
||||
node->parent = (struct ncclXmlNode *) (exp ? ((uintptr_t)node->parent - base + 1) : (base - 1 + (uintptr_t)node->parent));
|
||||
|
||||
for (int s = 0; s < node->nSubs; s++) {
|
||||
node->subs[s] = (struct ncclXmlNode *) (exp ? ((uintptr_t)node->subs[s] - base) : (base + (uintptr_t)node->subs[s]));
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoDumpXmlRec(int indent, FILE* file, struct ncclXmlNode* node) {
|
||||
for (int i=0; i<indent; i++) fprintf(file, " ");
|
||||
fprintf(file, "<%s", node->name);
|
||||
@@ -252,6 +275,60 @@ ncclResult_t ncclTopoDumpXmlToFile(const char* xmlTopoFile, struct ncclXml* xml)
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoFuseXml(struct ncclXml* dst, struct ncclXml* src) {
|
||||
struct ncclXmlNode* topNode;
|
||||
NCCLCHECK(xmlFindTag(dst, "system", &topNode));
|
||||
|
||||
if (topNode == NULL) {
|
||||
xmlAddTree(dst, NULL, src->nodes);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Fuse the CPUs with the first XML
|
||||
struct ncclXmlNode* srcCpu;
|
||||
NCCLCHECK(xmlFindTag(src, "cpu", &srcCpu));
|
||||
while (srcCpu) {
|
||||
const char* srcNumaId;
|
||||
const char* srcHostHash;
|
||||
NCCLCHECK(xmlGetAttr(srcCpu, "numaid", &srcNumaId));
|
||||
if (srcNumaId == NULL) {
|
||||
WARN("TopoFuseXmls : could not find CPU numa ID.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
xmlGetAttr(srcCpu, "host_hash", &srcHostHash);
|
||||
if (srcHostHash == NULL)
|
||||
srcHostHash = "0";
|
||||
|
||||
// Search through the destination for a duplicate. Note that
|
||||
// this makes the complexity of this whole function O(n^2), but n
|
||||
// is expected to be small.
|
||||
struct ncclXmlNode* dstCpu;
|
||||
NCCLCHECK(xmlFindTag(dst, "cpu", &dstCpu));
|
||||
while (dstCpu) {
|
||||
const char* dstNumaId;
|
||||
const char* dstHostHash;
|
||||
NCCLCHECK(xmlGetAttr(dstCpu, "numaid", &dstNumaId));
|
||||
if (dstNumaId == NULL) {
|
||||
WARN("TopoFuseXmls : could not find CPU numa ID.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
xmlGetAttr(dstCpu, "host_hash", &dstHostHash);
|
||||
if (dstHostHash == NULL)
|
||||
dstHostHash = "0";
|
||||
if (strcmp(srcNumaId, dstNumaId) == 0 && strcmp(srcHostHash, dstHostHash) == 0)
|
||||
break;
|
||||
|
||||
NCCLCHECK(xmlFindNextTag(dst, "cpu", dstCpu, &dstCpu));
|
||||
}
|
||||
// Only add the CPU if no duplicate was found
|
||||
if (dstCpu == NULL)
|
||||
NCCLCHECK(xmlAddTree(dst, topNode, srcCpu));
|
||||
NCCLCHECK(xmlFindNextTag(src, "cpu", srcCpu, &srcCpu));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
|
||||
/****************************************/
|
||||
/* Parser rules for our specific format */
|
||||
/****************************************/
|
||||
@@ -577,6 +654,7 @@ ncclResult_t ncclTopoGetXmlFromSys(struct ncclXmlNode* pciNode, struct ncclXml*
|
||||
NCCLCHECK(xmlGetSubKv(topNode, "cpu", &parent, "numaid", numaIdStr));
|
||||
if (parent == NULL) {
|
||||
NCCLCHECK(xmlAddNode(xml, topNode, "cpu", &parent));
|
||||
NCCLCHECK(xmlSetAttrLong(parent, "host_hash", getHostHash()));
|
||||
NCCLCHECK(xmlSetAttr(parent, "numaid", numaIdStr));
|
||||
}
|
||||
} else if (slashCount == 2) {
|
||||
@@ -602,6 +680,7 @@ ncclResult_t ncclTopoGetXmlFromSys(struct ncclXmlNode* pciNode, struct ncclXml*
|
||||
struct ncclXmlNode* topNode;
|
||||
NCCLCHECK(xmlFindTag(xml, "system", &topNode));
|
||||
NCCLCHECK(xmlAddNode(xml, topNode, "cpu", &parent));
|
||||
NCCLCHECK(xmlSetAttrLong(parent, "host_hash", getHostHash()));
|
||||
NCCLCHECK(xmlSetAttr(parent, "numaid", "-1"));
|
||||
NCCLCHECK(ncclTopoGetXmlFromCpu(parent, xml));
|
||||
}
|
||||
@@ -616,6 +695,10 @@ ncclResult_t ncclTopoGetXmlFromSys(struct ncclXmlNode* pciNode, struct ncclXml*
|
||||
NCCLCHECK(xmlGetAttr(parent->subs[s], "busid", &busId));
|
||||
if (busId != NULL && strcmp(newBusId, busId) < 0) { subIndex = s; break; }
|
||||
}
|
||||
if (parent->nSubs == MAX_SUBS) {
|
||||
WARN("Error : XML parser is limited to %d subnodes", MAX_SUBS);
|
||||
return ncclInternalError;
|
||||
}
|
||||
for (int s = parent->nSubs; s > subIndex; s--) parent->subs[s] = parent->subs[s-1];
|
||||
parent->subs[subIndex] = pciNode;
|
||||
parent->nSubs++;
|
||||
|
||||
+84
-7
@@ -11,14 +11,14 @@
|
||||
#include "nccl.h"
|
||||
#include "debug.h"
|
||||
#include "checks.h"
|
||||
#include "alloc.h"
|
||||
#include <stdlib.h>
|
||||
#include "archinfo.h"
|
||||
|
||||
// A few constraints to make the implementation easy
|
||||
#define MAX_STR_LEN 255
|
||||
#define MAX_ATTR_COUNT 16
|
||||
#define MAX_SUBS 512 //Changed the value from 32 to 512 for CPX mode
|
||||
#define MAX_NODES 8192 //Changed the value from 1024 to 8192 for CPX mode
|
||||
#define MAX_SUBS 512 //Changed the value from 128 to 512 for CPX mode
|
||||
|
||||
#define NODE_TYPE_NONE 0
|
||||
#define NODE_TYPE_OPEN 1
|
||||
@@ -39,8 +39,8 @@ struct ncclXmlNode {
|
||||
};
|
||||
|
||||
struct ncclXml {
|
||||
struct ncclXmlNode nodes[MAX_NODES];
|
||||
int maxIndex;
|
||||
int maxIndex, maxNodes;
|
||||
struct ncclXmlNode nodes[1];
|
||||
};
|
||||
|
||||
/* File functions */
|
||||
@@ -57,6 +57,11 @@ ncclResult_t ncclTopoFillNet(struct ncclXml* xml, const char* pciPath, const cha
|
||||
/* Remove unneeded parts */
|
||||
ncclResult_t ncclTopoTrimXml(struct ncclXml* xml);
|
||||
|
||||
/* Fuse multiple system XMLs into one, skipping duplicate CPUs */
|
||||
ncclResult_t ncclTopoFuseXml(struct ncclXml* dst, struct ncclXml* src);
|
||||
/* Relocate pointers in XML to (de-)serialize the structure */
|
||||
ncclResult_t ncclTopoConvertXml(struct ncclXml* xml, uintptr_t base, int exp);
|
||||
|
||||
ncclResult_t ncclTopoGetStrFromSys(const char* path, const char* fileName, char* strValue);
|
||||
|
||||
/**************/
|
||||
@@ -64,6 +69,17 @@ ncclResult_t ncclTopoGetStrFromSys(const char* path, const char* fileName, char*
|
||||
/* Functions */
|
||||
/**************/
|
||||
|
||||
static size_t xmlMemSize(int maxNodes) {
|
||||
return offsetof(struct ncclXml, nodes) + sizeof(struct ncclXmlNode)*maxNodes;
|
||||
}
|
||||
static ncclResult_t xmlAlloc(struct ncclXml** xml, int maxNodes) {
|
||||
char* mem;
|
||||
NCCLCHECK(ncclCalloc(&mem, xmlMemSize(maxNodes)));
|
||||
*xml = (struct ncclXml*)mem;
|
||||
(*xml)->maxNodes = maxNodes;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetAttrIndex(struct ncclXmlNode* node, const char* attrName, int* index) {
|
||||
*index = -1;
|
||||
const int nAttrs = node->nAttrs;
|
||||
@@ -105,6 +121,13 @@ static ncclResult_t xmlGetAttrIntDefault(struct ncclXmlNode* node, const char* a
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetAttrLong(struct ncclXmlNode* node, const char* attrName, int64_t* value) {
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttrStr(node, attrName, &str));
|
||||
*value = strtol(str, NULL, 0);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
|
||||
static ncclResult_t xmlGetAttrFloat(struct ncclXmlNode* node, const char* attrName, float* value) {
|
||||
const char* str;
|
||||
@@ -125,6 +148,18 @@ static ncclResult_t xmlFindTag(struct ncclXml* xml, const char* tagName, struct
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlFindNextTag(struct ncclXml* xml, const char* tagName, struct ncclXmlNode* prev, struct ncclXmlNode** node) {
|
||||
*node = NULL;
|
||||
for (int i=prev-xml->nodes+1; i<xml->maxIndex; i++) {
|
||||
struct ncclXmlNode* n = xml->nodes+i;
|
||||
if (strcmp(n->name, tagName) == 0) {
|
||||
*node = n;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlFindTagKv(struct ncclXml* xml, const char* tagName, struct ncclXmlNode** node, const char* attrName, const char* attrValue) {
|
||||
*node = NULL;
|
||||
for (int i=0; i<xml->maxIndex; i++) {
|
||||
@@ -192,6 +227,19 @@ static ncclResult_t xmlSetAttrFloat(struct ncclXmlNode* node, const char* attrNa
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlSetAttrLong(struct ncclXmlNode* node, const char* attrName, const int64_t value) {
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(node, attrName, &index));
|
||||
if (index == -1) {
|
||||
index = node->nAttrs++;
|
||||
strncpy(node->attrs[index].key, attrName, MAX_STR_LEN);
|
||||
node->attrs[index].key[MAX_STR_LEN] = '\0';
|
||||
}
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "%#lx", value);
|
||||
node->attrs[index].value[MAX_STR_LEN] = '\0';
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlUnsetAttr(struct ncclXmlNode* node, const char* attrName) {
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(node, attrName, &index));
|
||||
@@ -238,8 +286,8 @@ static ncclResult_t xmlGetSubKvInt(struct ncclXmlNode* node, const char* subName
|
||||
}
|
||||
|
||||
static ncclResult_t xmlAddNode(struct ncclXml* xml, struct ncclXmlNode* parent, const char* subName, struct ncclXmlNode** sub) {
|
||||
if (xml->maxIndex == MAX_NODES) {
|
||||
WARN("Error : too many XML nodes (max %d)", MAX_NODES);
|
||||
if (xml->maxIndex == xml->maxNodes) {
|
||||
WARN("Error : too many XML nodes (max %d)", xml->maxNodes);
|
||||
return ncclInternalError;
|
||||
}
|
||||
struct ncclXmlNode* s = xml->nodes+xml->maxIndex++;
|
||||
@@ -247,7 +295,13 @@ static ncclResult_t xmlAddNode(struct ncclXml* xml, struct ncclXmlNode* parent,
|
||||
s->nAttrs = 0;
|
||||
*sub = s;
|
||||
s->parent = parent;
|
||||
if (parent) parent->subs[parent->nSubs++] = s;
|
||||
if (parent) {
|
||||
if (parent->nSubs == MAX_SUBS) {
|
||||
WARN("Error : too many XML subnodes (max %d)", MAX_SUBS);
|
||||
return ncclInternalError;
|
||||
}
|
||||
parent->subs[parent->nSubs++] = s;
|
||||
}
|
||||
strncpy(s->name, subName, MAX_STR_LEN);
|
||||
s->name[MAX_STR_LEN] = '\0';
|
||||
return ncclSuccess;
|
||||
@@ -266,6 +320,29 @@ static ncclResult_t xmlRemoveNode(struct ncclXmlNode* node) {
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlAddTree(struct ncclXml* dst, struct ncclXmlNode* parent, struct ncclXmlNode* srcNode) {
|
||||
if (dst->maxIndex == dst->maxNodes) {
|
||||
WARN("Error : too many XML nodes (max %d)", dst->maxNodes);
|
||||
return ncclInternalError;
|
||||
}
|
||||
struct ncclXmlNode* dstNode = dst->nodes+dst->maxIndex++;
|
||||
*dstNode = *srcNode;
|
||||
dstNode->parent = parent;
|
||||
if (parent) {
|
||||
if (parent->nSubs == MAX_SUBS) {
|
||||
WARN("Error : too many XML subnodes (max %d)", MAX_SUBS);
|
||||
return ncclInternalError;
|
||||
}
|
||||
parent->subs[parent->nSubs++] = dstNode;
|
||||
}
|
||||
dstNode->nSubs = 0;
|
||||
// Recursively copy the subtree(s)
|
||||
for (int i=0; i<srcNode->nSubs; i++)
|
||||
NCCLCHECK(xmlAddTree(dst, dstNode, srcNode->subs[i]));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
|
||||
// Dictionary for STR -> INT conversions. No dictionary size information,
|
||||
// there needs to be a last element with str == NULL.
|
||||
struct kvDict {
|
||||
|
||||
Reference in New Issue
Block a user