Merge remote-tracking branch 'nccl/master' into v2.6.4_merge
[ROCm/rccl commit: fa36fd9ef9]
This commit is contained in:
@@ -1,5 +1,5 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2016-2019, NVIDIA CORPORATION. All rights reserved.
|
||||
* Copyright (c) 2016-2020, NVIDIA CORPORATION. All rights reserved.
|
||||
* Modifications Copyright (c) 2019-2020 Advanced Micro Devices, Inc. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
@@ -15,7 +15,7 @@
|
||||
/******************************************************************/
|
||||
|
||||
ncclResult_t ncclTopoPreset(struct ncclComm* comm,
|
||||
struct ncclTopoGraph* treeGraph, struct ncclTopoGraph* ringGraph,
|
||||
struct ncclTopoGraph* treeGraph, struct ncclTopoGraph* ringGraph, struct ncclTopoGraph* collNetGraph,
|
||||
struct ncclTopoRanks* topoRanks) {
|
||||
int rank = comm->rank;
|
||||
int localRanks = comm->localRanks;
|
||||
@@ -28,9 +28,14 @@ ncclResult_t ncclTopoPreset(struct ncclComm* comm,
|
||||
for (int i=0; i<NCCL_MAX_TREE_ARITY; i++) channel->treeUp.down[i] = -1;
|
||||
channel->treeDn.up = -1;
|
||||
for (int i=0; i<NCCL_MAX_TREE_ARITY; i++) channel->treeDn.down[i] = -1;
|
||||
channel->collTreeUp.up = -1;
|
||||
for (int i=0; i<NCCL_MAX_TREE_ARITY; i++) channel->collTreeUp.down[i] = -1;
|
||||
channel->collTreeDn.up = -1;
|
||||
for (int i=0; i<NCCL_MAX_TREE_ARITY; i++) channel->collTreeDn.down[i] = -1;
|
||||
|
||||
int* ringIntra = ringGraph->intra+c*localRanks;
|
||||
int* treeIntra = treeGraph->intra+c*localRanks;
|
||||
int* collNetIntra = collNetGraph->intra+c*localRanks;
|
||||
|
||||
for (int i=0; i<localRanks; i++) {
|
||||
if (ringIntra[i] == rank) {
|
||||
@@ -58,6 +63,16 @@ ncclResult_t ncclTopoPreset(struct ncclComm* comm,
|
||||
channel->treeUp.down[0] = sym ? channel->treeDn.down[0] : channel->treeDn.up ;
|
||||
channel->treeUp.up = sym ? channel->treeDn.up : channel->treeDn.down[0];
|
||||
}
|
||||
if (collNetIntra[i] == rank) {
|
||||
int prev = (i-1+localRanks)%localRanks, next = (i+1)%localRanks;
|
||||
|
||||
// CollTrees are always symmetric, i.e.
|
||||
// up/down go in reverse directions
|
||||
channel->collTreeDn.up = collNetIntra[prev];
|
||||
channel->collTreeDn.down[0] = collNetIntra[next];
|
||||
channel->collTreeUp.down[0] = channel->collTreeDn.down[0];
|
||||
channel->collTreeUp.up = channel->collTreeDn.up;
|
||||
}
|
||||
}
|
||||
topoRanks->ringPrev[c] = channel->ring.prev;
|
||||
topoRanks->ringNext[c] = channel->ring.next;
|
||||
@@ -175,6 +190,40 @@ static ncclResult_t connectTrees(struct ncclComm* comm, int* treeUpRecv, int* tr
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoConnectCollNet(struct ncclComm* comm, struct ncclTopoGraph* collNetGraph, int rank) {
|
||||
int nranks = comm->nRanks;
|
||||
int depth = nranks/comm->nNodes;
|
||||
int sendIndex = collNetGraph->pattern == NCCL_TOPO_PATTERN_TREE ? 0 : 1; // send GPU index depends on topo pattern
|
||||
int sendEndIndex = (sendIndex+comm->localRanks-1)%comm->localRanks;
|
||||
for (int c=0; c<comm->nChannels/2; c++) {
|
||||
struct ncclChannel* channel = comm->channels+c;
|
||||
// Set root of collTree to id nranks
|
||||
if (rank == collNetGraph->intra[sendIndex+c*comm->localRanks]) { // is master
|
||||
channel->collTreeUp.up = channel->collTreeDn.up = nranks;
|
||||
}
|
||||
if (rank == collNetGraph->intra[sendEndIndex+c*comm->localRanks]) { // is bottom of intra-node chain
|
||||
channel->collTreeUp.down[0] = channel->collTreeDn.down[0] = -1;
|
||||
}
|
||||
channel->collTreeUp.depth = channel->collTreeDn.depth = depth;
|
||||
INFO(NCCL_GRAPH, "CollNet Channel %d rank %d up %d down %d", c, rank, channel->collTreeUp.up, channel->collTreeUp.down[0]);
|
||||
}
|
||||
int recvIndex = 0; // recv GPU index is always 0
|
||||
int recvEndIndex = (recvIndex+comm->localRanks-1)%comm->localRanks;
|
||||
for (int c=0; c<comm->nChannels/2; c++) {
|
||||
struct ncclChannel* channel = comm->channels+comm->nChannels/2+c;
|
||||
// Set root of collTree to id nranks
|
||||
if (rank == collNetGraph->intra[recvIndex+c*comm->localRanks]) { // is master
|
||||
channel->collTreeUp.up = channel->collTreeDn.up = nranks;
|
||||
}
|
||||
if (rank == collNetGraph->intra[recvEndIndex+c*comm->localRanks]) { // is bottom of intra-node chain
|
||||
channel->collTreeUp.down[0] = channel->collTreeDn.down[0] = -1;
|
||||
}
|
||||
channel->collTreeUp.depth = channel->collTreeDn.depth = depth;
|
||||
INFO(NCCL_GRAPH, "CollNet Channel %d rank %d up %d down %d", comm->nChannels/2+c, rank, channel->collTreeDn.up, channel->collTreeDn.down[0]);
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Legacy naming
|
||||
NCCL_PARAM(MinNrings, "MIN_NRINGS", -2);
|
||||
NCCL_PARAM(MaxNrings, "MAX_NRINGS", -2);
|
||||
|
||||
+174
-116
@@ -1,5 +1,6 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2018-2019, NVIDIA CORPORATION. All rights reserved.
|
||||
* Copyright (c) 2018-2020, NVIDIA CORPORATION. All rights reserved.
|
||||
* Modifications Copyright (c) 2019-2020 Advanced Micro Devices, Inc. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
@@ -42,7 +43,7 @@ static ncclResult_t ncclTopoSetPaths(struct ncclTopoNode* baseNode, struct ncclT
|
||||
NCCLCHECK(getPath(system, baseNode, baseNode->type, baseNode->id, &basePath));
|
||||
basePath->count = 0;
|
||||
basePath->width = LOC_WIDTH;
|
||||
basePath->type = LINK_LOC;
|
||||
basePath->type = PATH_LOC;
|
||||
|
||||
while (nodeList.count) {
|
||||
nextNodeList.count = 0;
|
||||
@@ -58,7 +59,7 @@ static ncclResult_t ncclTopoSetPaths(struct ncclTopoNode* baseNode, struct ncclT
|
||||
}
|
||||
struct ncclTopoLinkList* remPath;
|
||||
NCCLCHECK(getPath(system, remNode, baseNode->type, baseNode->id, &remPath));
|
||||
int width = std::min(path->width, link->width);
|
||||
float width = std::min(path->width, link->width);
|
||||
if (remPath->width < width) {
|
||||
// Find reverse link
|
||||
for (int l=0; l<remNode->nlinks; l++) {
|
||||
@@ -68,8 +69,8 @@ static ncclResult_t ncclTopoSetPaths(struct ncclTopoNode* baseNode, struct ncclT
|
||||
}
|
||||
}
|
||||
if (remPath->list[0] == NULL) {
|
||||
WARN("Failed to find reverse path from remNode id %d type %d nlinks %d to node id %d type %d",
|
||||
remNode->id, remNode->type, remNode->nlinks, node->id, node->type);
|
||||
WARN("Failed to find reverse path from remNode %d/%lx nlinks %d to node %d/%lx",
|
||||
remNode->type, remNode->id, remNode->nlinks, node->type, node->id);
|
||||
return ncclInternalError;
|
||||
}
|
||||
// Copy the rest of the path
|
||||
@@ -77,9 +78,17 @@ static ncclResult_t ncclTopoSetPaths(struct ncclTopoNode* baseNode, struct ncclT
|
||||
remPath->count = path->count + 1;
|
||||
remPath->width = width;
|
||||
|
||||
// Consider the path is QPI when going through the CPU
|
||||
// Also don't consider LINK_NET as we only care about the NIC->GPU path.
|
||||
int type = remNode->type == CPU ? LINK_QPI : link->type == LINK_NET ? 0 : link->type;
|
||||
// Start with path type = link type. PATH and LINK types are supposed to match.
|
||||
// Don't consider LINK_NET as we only care about the NIC->GPU path.
|
||||
int type = link->type == LINK_NET ? 0 : link->type;
|
||||
// Differentiate between one and multiple PCI switches
|
||||
if (type == PATH_PIX && (node->type == PCI || link->remNode->type == PCI) && remPath->count > 3) type = PATH_PXB;
|
||||
// Consider a path going through the CPU as PATH_PHB
|
||||
if (link->type == LINK_PCI && (node->type == CPU || link->remNode->type == CPU)) type = PATH_PHB;
|
||||
// Ignore Power CPU in an NVLink path
|
||||
if (path->type == PATH_NVL && type == PATH_SYS && link->remNode->type == CPU &&
|
||||
link->remNode->cpu.arch == NCCL_TOPO_CPU_ARCH_POWER) type = 0;
|
||||
|
||||
remPath->type = std::max(path->type, type);
|
||||
|
||||
// Add to the list for the next iteration if not already in the list
|
||||
@@ -117,9 +126,9 @@ static void printNodePaths(struct ncclTopoSystem* system, struct ncclTopoNode* n
|
||||
sprintf(line+offset, "--%s->%s/%lX", topoLinkTypeStr[link->type], topoNodeTypeStr[remNode->type], remNode->id);
|
||||
offset = strlen(line);
|
||||
}
|
||||
INFO(NCCL_GRAPH, "%s (%d)", line, node->paths[t][n].width);
|
||||
INFO(NCCL_GRAPH, "%s (%f)", line, node->paths[t][n].width);
|
||||
#else
|
||||
sprintf(line+offset, "%s/%lX (%d/%d/%d) ", topoNodeTypeStr[t], system->nodes[t].nodes[n].id, node->paths[t][n].count, node->paths[t][n].width, node->paths[t][n].type);
|
||||
sprintf(line+offset, "%s/%lX (%d/%f/%s) ", topoNodeTypeStr[t], system->nodes[t].nodes[n].id, node->paths[t][n].count, node->paths[t][n].width, topoPathTypeStr[node->paths[t][n].type]);
|
||||
offset = strlen(line);
|
||||
#endif
|
||||
}
|
||||
@@ -171,7 +180,7 @@ static ncclResult_t addCpuStep(struct ncclTopoSystem* system, int c, int t1, int
|
||||
|
||||
// Update path characteristics
|
||||
srcNode->paths[t2][i2].count = l;
|
||||
srcNode->paths[t2][i2].type = LINK_QPI;
|
||||
srcNode->paths[t2][i2].type = std::max(srcNode->paths[CPU][c].type, cpuNode->paths[t2][i2].type);
|
||||
srcNode->paths[t2][i2].width = std::min(srcNode->paths[CPU][c].width, cpuNode->paths[t2][i2].width);
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -194,6 +203,131 @@ static void ncclTopoRemovePathType(struct ncclTopoSystem* system, int nodeType)
|
||||
}
|
||||
}
|
||||
|
||||
static const int levelsOldToNew[] = { PATH_LOC, PATH_PIX, PATH_PXB, PATH_PHB, PATH_SYS, PATH_SYS };
|
||||
ncclResult_t ncclGetLevel(int* level, const char* disableEnv, const char* levelEnv) {
|
||||
if (*level == -1) {
|
||||
int l = -1;
|
||||
if (disableEnv) {
|
||||
char* str = getenv(disableEnv);
|
||||
if (str) {
|
||||
int disable = strtol(str, NULL, 0);
|
||||
if (disable == 1) l = 0;
|
||||
}
|
||||
}
|
||||
if (l == -1) {
|
||||
char* str = getenv(levelEnv);
|
||||
if (str) {
|
||||
for (int i=0; i<PATH_NET; i++) {
|
||||
if (strcmp(str, topoPathTypeStr[i]) == 0) {
|
||||
l = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Old style numbering
|
||||
if (l == -1 && str[0] >= '0' && str[0] <= '9') {
|
||||
int oldLevel = strtol(str, NULL, 0);
|
||||
const int maxOldLevel = sizeof(levelsOldToNew)/sizeof(int) - 1;
|
||||
if (oldLevel > maxOldLevel) oldLevel = maxOldLevel;
|
||||
l = levelsOldToNew[oldLevel];
|
||||
}
|
||||
}
|
||||
}
|
||||
if (l >= 0) INFO(NCCL_GRAPH, "%s set from environment to %s", levelEnv, topoPathTypeStr[l]);
|
||||
*level = l >= 0 ? l : -2;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
int ncclTopoUserP2pLevel = -1;
|
||||
ncclResult_t ncclTopoCheckP2p(struct ncclTopoSystem* system, int64_t id1, int64_t id2, int* p2p) {
|
||||
*p2p = 0;
|
||||
|
||||
// Get GPUs from topology
|
||||
int g1, g2;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, GPU, id1, &g1));
|
||||
struct ncclTopoNode* gpu1 = system->nodes[GPU].nodes+g1;
|
||||
if (ncclTopoIdToIndex(system, GPU, id2, &g2) == ncclInternalError) {
|
||||
// GPU not found, we can't use p2p.
|
||||
return ncclSuccess;
|
||||
}
|
||||
struct ncclTopoLinkList* path = gpu1->paths[GPU]+g2;
|
||||
|
||||
// In general, use P2P whenever we can.
|
||||
int p2pLevel = PATH_SYS;
|
||||
|
||||
// Don't use P2P through ARM CPUs
|
||||
int arch, vendor, model;
|
||||
NCCLCHECK(ncclTopoCpuType(system, &arch, &vendor, &model));
|
||||
if (arch == NCCL_TOPO_CPU_ARCH_ARM) p2pLevel = PATH_PXB;
|
||||
if (arch == NCCL_TOPO_CPU_ARCH_X86 &&
|
||||
vendor == NCCL_TOPO_CPU_VENDOR_INTEL &&
|
||||
model == NCCL_TOPO_CPU_TYPE_BDW) p2pLevel = PATH_PXB;
|
||||
|
||||
// User override
|
||||
NCCLCHECK(ncclGetLevel(&ncclTopoUserP2pLevel, "NCCL_P2P_DISABLE", "NCCL_P2P_LEVEL"));
|
||||
if (ncclTopoUserP2pLevel != -2) p2pLevel = ncclTopoUserP2pLevel;
|
||||
|
||||
// Compute the PCI distance and compare with the p2pLevel.
|
||||
if (path->type <= p2pLevel) *p2p = 1;
|
||||
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
NCCL_PARAM(NetGdrRead, "NET_GDR_READ", -2);
|
||||
int ncclTopoUserGdrLevel = -1;
|
||||
|
||||
ncclResult_t ncclTopoCheckGdr(struct ncclTopoSystem* system, int64_t busId, int netDev, int read, int* useGdr) {
|
||||
*useGdr = 0;
|
||||
|
||||
// Get GPU and NET
|
||||
int n, g;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, NET, netDev, &n));
|
||||
struct ncclTopoNode* net = system->nodes[NET].nodes+n;
|
||||
NCCLCHECK(ncclTopoIdToIndex(system, GPU, busId, &g));
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
|
||||
// Check that both the NIC and GPUs support it
|
||||
if (net->net.gdrSupport == 0) return ncclSuccess;
|
||||
if (gpu->gpu.gdrSupport == 0) return ncclSuccess;
|
||||
|
||||
if (read) { // For reads (sends) only enable under certain conditions
|
||||
int gdrReadParam = ncclParamNetGdrRead();
|
||||
if (gdrReadParam == 0) return ncclSuccess;
|
||||
#if defined(__HIP_PLATFORM_HCC__) || defined(__HCC__) || defined(__HIPCC__)
|
||||
return ncclSuccess;
|
||||
#else
|
||||
if (gdrReadParam < 0) {
|
||||
int nvlink = 0;
|
||||
// Since we don't know whether there are other communicators,
|
||||
// it's better to keep things local if we have a single GPU.
|
||||
if (system->nodes[GPU].count == 1) nvlink = 1;
|
||||
for (int i=0; i<system->nodes[GPU].count; i++) {
|
||||
if (i == g) continue;
|
||||
if (gpu->paths[GPU][i].type == PATH_NVL) {
|
||||
nvlink = 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!nvlink) return ncclSuccess;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
// Check if we are close enough that it makes sense to enable GDR
|
||||
int netGdrLevel = PATH_PXB;
|
||||
NCCLCHECK(ncclGetLevel(&ncclTopoUserGdrLevel, NULL, "NCCL_NET_GDR_LEVEL"));
|
||||
if (ncclTopoUserGdrLevel != -2) netGdrLevel = ncclTopoUserGdrLevel;
|
||||
int distance = gpu->paths[NET][n].type;
|
||||
if (distance > netGdrLevel) {
|
||||
INFO(NCCL_NET,"GPU Direct RDMA Disabled for GPU %lx / HCA %d (distance %d > %d)", busId, netDev, distance, netGdrLevel);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
*useGdr = 1;
|
||||
INFO(NCCL_NET,"GPU Direct RDMA Enabled for GPU %lx / HCA %d (distance %d <= %d), read %d", busId, netDev, distance, netGdrLevel, read);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclPeerInfo* peerInfos) {
|
||||
// Precompute paths between GPUs/NICs.
|
||||
|
||||
@@ -210,26 +344,29 @@ ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclPeer
|
||||
// Compute paths to GPU g
|
||||
NCCLCHECK(ncclTopoSetPaths(system->nodes[GPU].nodes+g, system));
|
||||
|
||||
// Update path when we don't want to / can't use GPU Direct P2P
|
||||
for (int p=0; p<system->nodes[GPU].count; p++) {
|
||||
int p2p;
|
||||
NCCLCHECK(ncclTopoCheckP2p(system, system->nodes[GPU].nodes[p].id, system->nodes[GPU].nodes[g].id, &p2p));
|
||||
if (p2p == 0) {
|
||||
// Divert all traffic through the CPU
|
||||
int cpu;
|
||||
NCCLCHECK(getLocalCpu(system, g, &cpu));
|
||||
NCCLCHECK(addCpuStep(system, cpu, GPU, p, GPU, g));
|
||||
}
|
||||
}
|
||||
|
||||
if (peerInfos == NULL) continue;
|
||||
// Update paths from GPUs p to GPU g when we can't or don't want to use P2P or even SHM
|
||||
struct ncclPeerInfo* dstInfo = peerInfos+system->nodes[GPU].nodes[g].rank;
|
||||
// Remove GPUs we can't talk to because of containers.
|
||||
struct ncclPeerInfo* dstInfo = peerInfos+system->nodes[GPU].nodes[g].gpu.rank;
|
||||
for (int p=0; p<system->nodes[GPU].count; p++) {
|
||||
if (p == g) continue;
|
||||
struct ncclPeerInfo* srcInfo = peerInfos+system->nodes[GPU].nodes[p].rank;
|
||||
int p2p;
|
||||
NCCLCHECK(ncclTransports[TRANSPORT_P2P].canConnect(&p2p, system, NULL, srcInfo, dstInfo));
|
||||
if (p2p == 0) {
|
||||
int shm;
|
||||
NCCLCHECK(ncclTransports[TRANSPORT_SHM].canConnect(&shm, system, NULL, srcInfo, dstInfo));
|
||||
if (shm == 1) {
|
||||
// We cannot use GPU Direct, so we need all traffic to go through a CPU
|
||||
int cpu;
|
||||
NCCLCHECK(getLocalCpu(system, g, &cpu));
|
||||
NCCLCHECK(addCpuStep(system, cpu, GPU, p, GPU, g));
|
||||
} else {
|
||||
// We cannot communicate with that peer.
|
||||
system->nodes[GPU].nodes[p].paths[GPU][g].count = 0;
|
||||
}
|
||||
struct ncclPeerInfo* srcInfo = peerInfos+system->nodes[GPU].nodes[p].gpu.rank;
|
||||
int shm;
|
||||
NCCLCHECK(ncclTransports[TRANSPORT_SHM].canConnect(&shm, system, NULL, srcInfo, dstInfo));
|
||||
if (shm == 0) {
|
||||
// Mark this peer as inaccessible. We'll trim it later.
|
||||
system->nodes[GPU].nodes[p].paths[GPU][g].count = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -239,11 +376,12 @@ ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclPeer
|
||||
struct ncclTopoNode* netNode = system->nodes[NET].nodes+n;
|
||||
NCCLCHECK(ncclTopoSetPaths(netNode, system));
|
||||
|
||||
if (peerInfos == NULL) continue;
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
if ((peerInfos[system->nodes[GPU].nodes[g].rank].gdrSupport & (1 << n)) == 0) {
|
||||
// We cannot use GPU Direct RDMA, so we need all NIC<->GPU paths
|
||||
// to go through a CPU
|
||||
// Update path when we dont want to / can't use GPU Direct RDMA.
|
||||
int gdr;
|
||||
NCCLCHECK(ncclTopoCheckGdr(system, system->nodes[GPU].nodes[g].id, netNode->id, 0, &gdr));
|
||||
if (gdr == 0) {
|
||||
// We cannot use GPU Direct RDMA, divert all traffic through the CPU local to the GPU
|
||||
int localCpu;
|
||||
NCCLCHECK(getLocalCpu(system, g, &localCpu));
|
||||
NCCLCHECK(addCpuStep(system, localCpu, NET, n, GPU, g));
|
||||
@@ -251,7 +389,6 @@ ncclResult_t ncclTopoComputePaths(struct ncclTopoSystem* system, struct ncclPeer
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -270,7 +407,7 @@ ncclResult_t ncclTopoTrimSystem(struct ncclTopoSystem* system, struct ncclComm*
|
||||
domains[g] = std::min(domains[g], domains[p]);
|
||||
}
|
||||
}
|
||||
if (gpu->rank == comm->rank) myDomain = domains[g];
|
||||
if (gpu->gpu.rank == comm->rank) myDomain = domains[g];
|
||||
}
|
||||
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
@@ -288,98 +425,19 @@ ncclResult_t ncclTopoTrimSystem(struct ncclTopoSystem* system, struct ncclComm*
|
||||
free(ids);
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
// Remove GPUs I can't access (even indirectly) from my view of the node
|
||||
for (int t=0; t<NCCL_TOPO_NODE_TYPES; t++) {
|
||||
for (int n=0; n<system->nodes[t].count; n++) {
|
||||
struct ncclTopoNode* node = system->nodes[t].nodes+n;
|
||||
if (node == gpu) continue;
|
||||
for (int l=0; l<node->nlinks; l++) {
|
||||
while (l<node->nlinks && node->links[l].remNode == gpu) {
|
||||
if (l<node->nlinks-1)
|
||||
memmove(node->links+l, node->links+l+1, (node->nlinks-l-1)*sizeof(struct ncclTopoLink));
|
||||
node->nlinks--;
|
||||
}
|
||||
if (l<node->nlinks && node->links[l].remNode->type == GPU && node->links[l].remNode >= gpu) {
|
||||
node->links[l].remNode--;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (g != system->nodes[GPU].count-1)
|
||||
memmove(gpu, gpu+1, (system->nodes[GPU].count-g-1)*sizeof(struct ncclTopoNode));
|
||||
system->nodes[GPU].count--;
|
||||
NCCLCHECK(ncclTopoRemoveNode(system, GPU, g));
|
||||
}
|
||||
|
||||
comm->localRanks = system->nodes[GPU].count;
|
||||
if (system->nodes[GPU].count == comm->nRanks) {
|
||||
// Trim network
|
||||
ncclTopoRemovePathType(system, NET);
|
||||
system->nodes[NET].count = 0;
|
||||
for (int t=0; t<NCCL_TOPO_NODE_TYPES; t++) {
|
||||
for (int n=0; n<system->nodes[t].count; n++) {
|
||||
struct ncclTopoNode* node = system->nodes[t].nodes+n;
|
||||
for (int l=0; l<node->nlinks; l++) {
|
||||
struct ncclTopoLink* link = &(node->links[l]);
|
||||
if (link->remNode->type == NET) {
|
||||
// Remove the link
|
||||
for (int i=l; i<(node->nlinks-1); i++) {
|
||||
memcpy(&(node->links[i]), &(node->links[i+1]), sizeof(ncclTopoLink));
|
||||
}
|
||||
node->nlinks--;
|
||||
l--; // revisit the same value of "l" for the next iteration, since we edited the list in the middle of the loop
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int n=system->nodes[NET].count-1; n>=0; n--)
|
||||
NCCLCHECK(ncclTopoRemoveNode(system, NET, n));
|
||||
}
|
||||
free(domains);
|
||||
free(ids);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t getGpuSpeed(struct ncclTopoNode* node, int* speed) {
|
||||
int nvlSpeed = 0;
|
||||
int nvlPeers = 0;
|
||||
int pciSpeed = 0;
|
||||
for (int l=0; l<node->nlinks; l++) {
|
||||
if (node->links[l].type == LINK_NVL) nvlSpeed += node->links[l].width;
|
||||
if (node->links[l].remNode->type == GPU) nvlPeers++; else nvlPeers = 2;
|
||||
if (node->links[l].type == LINK_PCI) pciSpeed = node->links[l].width;
|
||||
}
|
||||
*speed = std::min(*speed, std::max(nvlSpeed, pciSpeed));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetMaxSpeed(struct ncclTopoSystem* system) {
|
||||
// Compute max speed to try to accelerate the search.
|
||||
system->maxSpeed = LOC_WIDTH;
|
||||
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
NCCLCHECK(getGpuSpeed(system->nodes[GPU].nodes+g, &system->maxSpeed));
|
||||
}
|
||||
if (system->nodes[NET].count) {
|
||||
// Try to assign one NIC per GPU
|
||||
int netMaxSpeed = 0;
|
||||
int netMaxSpeedCount = 0;
|
||||
for (int n=0; n<system->nodes[NET].count; n++) {
|
||||
int maxSpeed = 0;
|
||||
struct ncclTopoNode* net = system->nodes[NET].nodes+n;
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
maxSpeed = std::max(maxSpeed, net->paths[GPU][g].width);
|
||||
}
|
||||
if (maxSpeed > netMaxSpeed) {
|
||||
netMaxSpeed = maxSpeed;
|
||||
netMaxSpeedCount = 1;
|
||||
} else if (maxSpeed == netMaxSpeed) {
|
||||
netMaxSpeedCount++;
|
||||
}
|
||||
}
|
||||
system->maxSpeed = std::min(system->maxSpeed, netMaxSpeedCount*NET_WIDTH);
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
void ncclTopoFree(struct ncclTopoSystem* system) {
|
||||
for (int t=0; t<NCCL_TOPO_NODE_TYPES; t++) ncclTopoRemovePathType(system, t);
|
||||
free(system);
|
||||
|
||||
+470
-167
@@ -1,5 +1,5 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2016-2019, NVIDIA CORPORATION. All rights reserved.
|
||||
* Copyright (c) 2016-2020, NVIDIA CORPORATION. All rights reserved.
|
||||
* Modifications Copyright (c) 2019-2020 Advanced Micro Devices, Inc. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
@@ -8,29 +8,125 @@
|
||||
#include "core.h"
|
||||
#include "graph.h"
|
||||
#include "topo.h"
|
||||
#include "xml.h"
|
||||
#include <math.h>
|
||||
|
||||
static ncclResult_t ncclTopoFollowPath(struct ncclTopoGraph* graph, struct ncclTopoLinkList* path, struct ncclTopoNode** node, int width, int typeSave) {
|
||||
if (path->count == 0) return ncclSuccess;
|
||||
|
||||
*node = NULL;
|
||||
if (width > 0) {
|
||||
if (path->type > graph->type) return ncclSuccess;
|
||||
graph->type = std::max(graph->type, path->type);
|
||||
graph->nHops += path->count;
|
||||
} else {
|
||||
graph->type = typeSave;
|
||||
graph->nHops -= path->count;
|
||||
// Initialize system->maxWidth. This is the per-channel (i.e. per-SM)
|
||||
// max speed.
|
||||
static float getMaxWidth(struct ncclTopoSystem* system, struct ncclTopoNode* gpu, int type) {
|
||||
#if defined(__HIP_PLATFORM_HCC__) || defined(__HCC__) || defined(__HIPCC__)
|
||||
float nvLinkWidth = VEGA_XGMI_WIDTH;
|
||||
#else
|
||||
float nvLinkWidth = gpu->gpu.cudaCompCap > 60 ? VOLTA_NVLINK_WIDTH : PASCAL_NVLINK_WIDTH;
|
||||
#endif
|
||||
float maxWidth = 0.0;
|
||||
for (int i=0; i<system->nodes[type].count; i++) {
|
||||
struct ncclTopoLinkList* path = gpu->paths[type]+i;
|
||||
float width = path->width;
|
||||
if (path->count == 0) continue;
|
||||
if (path->type == PATH_NVL) width = std::min(nvLinkWidth, width);
|
||||
maxWidth = std::max(maxWidth, width);
|
||||
}
|
||||
return maxWidth;
|
||||
}
|
||||
ncclResult_t ncclTopoSearchInit(struct ncclTopoSystem* system) {
|
||||
system->maxWidth = 0.0;
|
||||
int inter = system->nodes[NET].count;
|
||||
if (inter == 0 && system->nodes[GPU].count == 1) {
|
||||
system->maxWidth = LOC_WIDTH;
|
||||
return ncclSuccess;
|
||||
}
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
system->maxWidth = std::max(system->maxWidth, getMaxWidth(system, gpu, inter ? NET : GPU));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
for (int i=0; i<path->count; i++) {
|
||||
if (path->list[i]->width < width) {
|
||||
// Can't follow this path, rewind and exit
|
||||
for (int j=0; j<i; j++) path->list[j]->width += width;
|
||||
static ncclResult_t findRevLink(struct ncclTopoNode* node1, struct ncclTopoNode* node2, struct ncclTopoLink** revLink) {
|
||||
for (int l=0; l<node2->nlinks; l++) {
|
||||
struct ncclTopoLink* link = node2->links+l;
|
||||
if (link->remNode == node1) {
|
||||
*revLink = link;
|
||||
return ncclSuccess;
|
||||
}
|
||||
path->list[i]->width -= width;
|
||||
}
|
||||
*node = path->list[path->count-1]->remNode;
|
||||
WARN("Could not find rev link for %d/%d -> %d/%d\n", node1->type, node1->id, node2->type, node2->id);
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
// This is unfortunately needed since manipulating floats often results in rounding errors.
|
||||
#define SUB_ROUND(a, b) (a = roundf((a-b)*1000)/1000)
|
||||
|
||||
static ncclResult_t followPath(struct ncclTopoLinkList* path, struct ncclTopoNode* start, int maxSteps, float speed, int* steps) {
|
||||
float pciSpeed = speed;
|
||||
for (int step=0; step<path->count; step++) {
|
||||
struct ncclTopoNode* node = path->list[step]->remNode;
|
||||
if (node->type == CPU) {
|
||||
// Account for P2P inefficiency through Intel CPU RC
|
||||
if (path->type == PATH_PHB && start->type == GPU &&
|
||||
node->cpu.arch == NCCL_TOPO_CPU_ARCH_X86 &&
|
||||
node->cpu.vendor == NCCL_TOPO_CPU_VENDOR_INTEL) {
|
||||
pciSpeed = INTEL_P2P_OVERHEAD(speed);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct ncclTopoNode* node = start;
|
||||
for (int step=0; step<maxSteps; step++) {
|
||||
struct ncclTopoLink* link = path->list[step];
|
||||
struct ncclTopoLink* revLink = NULL;
|
||||
float fwSpeed = link->type == LINK_PCI ? pciSpeed : speed;
|
||||
float revSpeed = 0;
|
||||
if (link->remNode->type == GPU && start->type != GPU) {
|
||||
if (revLink == NULL) NCCLCHECK(findRevLink(node, link->remNode, &revLink));
|
||||
revSpeed += fwSpeed/8;
|
||||
}
|
||||
if (link->remNode->type == CPU && link->type == LINK_NVL) {
|
||||
if (revLink == NULL) NCCLCHECK(findRevLink(node, link->remNode, &revLink));
|
||||
revSpeed += fwSpeed;
|
||||
}
|
||||
if (link->width < fwSpeed || (revSpeed && revLink->width < revSpeed)) { *steps = step; return ncclSuccess; }
|
||||
SUB_ROUND(link->width, fwSpeed);
|
||||
if (revSpeed) SUB_ROUND(revLink->width, revSpeed);
|
||||
node = link->remNode;
|
||||
}
|
||||
*steps = maxSteps;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Try to go from node type1/index1 to no type2/index2. mult indicates whether we are counting the bandwidth (1) or undoing (-1).
|
||||
static ncclResult_t ncclTopoFollowPath(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, int type1, int index1, int type2, int index2, int mult, struct ncclTopoNode** node) {
|
||||
// First handle easy cases
|
||||
*node = system->nodes[type2].nodes+index2;
|
||||
if (type1 == -1) return ncclSuccess;
|
||||
struct ncclTopoNode* node1 = system->nodes[type1].nodes+index1;
|
||||
struct ncclTopoLinkList* path = node1->paths[type2]+index2;
|
||||
if (path->count == 0 ) return ncclSuccess;
|
||||
|
||||
// Now check link type
|
||||
*node = NULL;
|
||||
int intra = type1 == GPU && type2 == GPU;
|
||||
float speed = intra ? graph->speedIntra : graph->speedInter;
|
||||
int type = intra ? graph->typeIntra : graph->typeInter;
|
||||
|
||||
if (mult == 1 && (path->type > type)) return ncclSuccess;
|
||||
|
||||
speed *= mult;
|
||||
|
||||
// Check there is enough bandwidth on paths.
|
||||
int step = 0;
|
||||
NCCLCHECK(followPath(path, node1, path->count, speed, &step));
|
||||
if (step < path->count) goto rewind;
|
||||
|
||||
// Enough bandwidth : return destination node.
|
||||
graph->nHops += mult*path->count;
|
||||
*node = system->nodes[type2].nodes+index2;
|
||||
return ncclSuccess;
|
||||
|
||||
rewind:
|
||||
// Not enough bandwidth : rewind and exit.
|
||||
NCCLCHECK(followPath(path, node1, step, -speed, &step));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -81,22 +177,42 @@ static int cmpIntraScores(struct ncclGpuScore* scores, int count) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
static ncclResult_t getNetPaths(struct ncclTopoSystem* system, const uint64_t flag, struct ncclTopoLinkList** netPaths) {
|
||||
for (int n=0; n<system->nodes[NET].count; n++) {
|
||||
if (system->nodes[NET].nodes[n].used & flag) {
|
||||
*netPaths=system->nodes[NET].nodes[n].paths[GPU];
|
||||
static ncclResult_t getGpuIndex(struct ncclTopoSystem* system, int rank, int* index) {
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
if (system->nodes[GPU].nodes[g].gpu.rank == rank) {
|
||||
*index = g;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
WARN("Could not find gpu rank %d\n", rank);
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
static ncclResult_t getNetIndex(struct ncclTopoSystem* system, int64_t id, int* index) {
|
||||
for (int n=0; n<system->nodes[NET].count; n++) {
|
||||
if (system->nodes[NET].nodes[n].id == id) {
|
||||
*index = n;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
WARN("Could not find net id %lx\n", id);
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
static ncclResult_t getNetPaths(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoLinkList** netPaths) {
|
||||
int netId = graph->inter[graph->nChannels*2];
|
||||
int n;
|
||||
NCCLCHECK(getNetIndex(system, netId, &n));
|
||||
*netPaths=system->nodes[NET].nodes[n].paths[GPU];
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoSearchNextGpuSort(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoNode* gpu, int* next, int* countPtr, int sortNet) {
|
||||
const uint64_t flag = 1ULL<<(graph->nChannels);
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
struct ncclTopoLinkList* paths = gpu->paths[GPU];
|
||||
struct ncclTopoLinkList* netPaths = NULL;
|
||||
if (sortNet) NCCLCHECK(getNetPaths(system, flag, &netPaths));
|
||||
if (sortNet) NCCLCHECK(getNetPaths(system, graph, &netPaths));
|
||||
|
||||
struct ncclGpuScore scores[NCCL_TOPO_MAX_NODES];
|
||||
memset(scores, 0, ngpus*sizeof(struct ncclGpuScore));
|
||||
@@ -131,9 +247,13 @@ ncclResult_t ncclTopoSearchNextGpuSort(struct ncclTopoSystem* system, struct ncc
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoSearchRec(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, int maxSpeed, int* time);
|
||||
ncclResult_t ncclTopoSearchRec(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, int* time);
|
||||
|
||||
#define NCCL_SEARCH_TIMEOUT (1ULL<<20) // This should get contain all search within a second or so.
|
||||
// Try to keep all searchs within one second
|
||||
#define NCCL_SEARCH_GLOBAL_TIMEOUT (3ULL<<19)
|
||||
#define NCCL_SEARCH_TIMEOUT (1<<18)
|
||||
#define NCCL_SEARCH_TIMEOUT_TREE (1<<17)
|
||||
#define NCCL_SEARCH_TIMEOUT_SAMECHANNELS (1<<10)
|
||||
|
||||
#define FORCED_ORDER_PCI 1
|
||||
#define FORCED_ORDER_REPLAY 2
|
||||
@@ -143,7 +263,7 @@ ncclResult_t ncclTopoReplayGetGpu(struct ncclTopoSystem* system, struct ncclTopo
|
||||
if (graph->nChannels == 0) return ncclInternalError;
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
int nextRank = graph->intra[(graph->nChannels-1)*ngpus+step+1];
|
||||
for (int i=0; i<ngpus; i++) if (system->nodes[GPU].nodes[i].rank == nextRank) {
|
||||
for (int i=0; i<ngpus; i++) if (system->nodes[GPU].nodes[i].gpu.rank == nextRank) {
|
||||
*g = i;
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -151,44 +271,37 @@ ncclResult_t ncclTopoReplayGetGpu(struct ncclTopoSystem* system, struct ncclTopo
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoSearchRecGpu(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, struct ncclTopoNode* gpu, int step, int backToNet, int backToFirstRank, int forcedOrder, int maxSpeed, int *time);
|
||||
ncclResult_t ncclTopoSearchRecGpu(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, struct ncclTopoNode* gpu, int step, int backToNet, int backToFirstRank, int forcedOrder, int *time);
|
||||
|
||||
ncclResult_t ncclTopoSearchTryGpu(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, struct ncclTopoLinkList* paths, int step, int backToNet, int backToFirstRank, int forcedOrder, int maxSpeed, int *time, int g, int speed) {
|
||||
int typeSave = graph->type;
|
||||
ncclResult_t ncclTopoSearchTryGpu(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, int step, int backToNet, int backToFirstRank, int forcedOrder, int *time, int type, int index, int g) {
|
||||
const uint64_t flag = 1ULL<<(graph->nChannels);
|
||||
struct ncclTopoNode* gpu = system->nodes[GPU].nodes+g;
|
||||
if (paths) NCCLCHECK(ncclTopoFollowPath(graph, paths+g, &gpu, speed, typeSave));
|
||||
struct ncclTopoNode* gpu;
|
||||
NCCLCHECK(ncclTopoFollowPath(system, graph, type, index, GPU, g, 1, &gpu));
|
||||
if (gpu) {
|
||||
gpu->used ^= flag;
|
||||
NCCLCHECK(ncclTopoSearchRecGpu(system, graph, saveGraph, gpu, step, backToNet, backToFirstRank, forcedOrder, maxSpeed, time));
|
||||
NCCLCHECK(ncclTopoSearchRecGpu(system, graph, saveGraph, gpu, step, backToNet, backToFirstRank, forcedOrder, time));
|
||||
gpu->used ^= flag;
|
||||
if (paths) NCCLCHECK(ncclTopoFollowPath(graph, paths+g, &gpu, -speed, typeSave));
|
||||
NCCLCHECK(ncclTopoFollowPath(system, graph, type, index, GPU, g, -1, &gpu));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoCompareGraphs(struct ncclTopoGraph* graph, struct ncclTopoGraph* refGraph, int* copy) {
|
||||
// 0. When we are trying to increase speedIntra, do not copy if the solution has less channels
|
||||
// since it would likely impact the rings algorithms too.
|
||||
if (graph->speedIntra > graph->speedInter && graph->nChannels < refGraph->nChannels) return ncclSuccess;
|
||||
// 1. Constraint to get the same nChannels between Rings and Trees
|
||||
if (graph->nChannels < graph->minChannels) return ncclSuccess;
|
||||
|
||||
// 1. Try to get better bandwidth
|
||||
// 2. Try to get better bandwidth
|
||||
if (graph->nChannels*graph->speedIntra < refGraph->nChannels*refGraph->speedIntra) return ncclSuccess;
|
||||
if (graph->nChannels*graph->speedIntra > refGraph->nChannels*refGraph->speedIntra) {
|
||||
*copy = 1;
|
||||
return ncclSuccess;
|
||||
}
|
||||
// 2. Give an advantage when all channels are the same
|
||||
if (graph->nChannels > 1 && graph->sameChannels && refGraph->sameChannels == 0) {
|
||||
*copy = 1;
|
||||
return ncclSuccess;
|
||||
}
|
||||
// 3. Less hops
|
||||
if (graph->nHops < refGraph->nHops) *copy = 1;
|
||||
// 3. Less hops (but not at the price of going cross NICs)
|
||||
if (graph->crossNic == refGraph->crossNic && graph->nHops < refGraph->nHops) *copy = 1;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoSearchRecGpu(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, struct ncclTopoNode* gpu, int step, int backToNet, int backToFirstRank, int forcedOrder, int maxSpeed, int *time) {
|
||||
ncclResult_t ncclTopoSearchRecGpu(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, struct ncclTopoNode* gpu, int step, int backToNet, int backToFirstRank, int forcedOrder, int *time) {
|
||||
if ((*time) <= 0) return ncclSuccess;
|
||||
(*time)--;
|
||||
|
||||
@@ -196,55 +309,43 @@ ncclResult_t ncclTopoSearchRecGpu(struct ncclTopoSystem* system, struct ncclTopo
|
||||
if (step == ngpus) {
|
||||
// Determine whether we found a better solution or not
|
||||
int copy = 0;
|
||||
int sameChannels = graph->sameChannels;
|
||||
if (graph->nChannels > 0) {
|
||||
int* intra = graph->intra+graph->nChannels*ngpus;
|
||||
for (int g=0; g<ngpus; g++) if (intra[g] != intra[g-ngpus]) graph->sameChannels = 0;
|
||||
}
|
||||
graph->nChannels++;
|
||||
NCCLCHECK(ncclTopoCompareGraphs(graph, saveGraph, ©));
|
||||
if (copy) {
|
||||
memcpy(saveGraph, graph, sizeof(struct ncclTopoGraph));
|
||||
if (graph->nChannels*graph->speedIntra == maxSpeed) *time = -1;
|
||||
if (graph->nChannels == graph->maxChannels) *time = -1;
|
||||
}
|
||||
if (graph->nChannels < MAXCHANNELS/2) {
|
||||
NCCLCHECK(ncclTopoSearchRec(system, graph, saveGraph, maxSpeed, time));
|
||||
if (graph->nChannels < graph->maxChannels) {
|
||||
NCCLCHECK(ncclTopoSearchRec(system, graph, saveGraph, time));
|
||||
}
|
||||
graph->nChannels--;
|
||||
graph->sameChannels = sameChannels;
|
||||
return ncclSuccess;
|
||||
}
|
||||
graph->intra[graph->nChannels*ngpus+step] = gpu->rank;
|
||||
graph->intra[graph->nChannels*ngpus+step] = gpu->gpu.rank;
|
||||
int g = gpu - system->nodes[GPU].nodes;
|
||||
if (step == backToNet) {
|
||||
// first get back to NIC
|
||||
if (system->nodes[NET].count) {
|
||||
int maxWidth = 0;
|
||||
struct ncclTopoLinkList* paths = gpu->paths[NET];
|
||||
int startNetIndex;
|
||||
NCCLCHECK(getNetIndex(system, graph->inter[graph->nChannels*2], &startNetIndex));
|
||||
struct ncclTopoNode* startNet = system->nodes[NET].nodes+startNetIndex;
|
||||
for (int n=0; n<system->nodes[NET].count; n++) {
|
||||
if (graph->crossNic != 1 && (system->nodes[NET].nodes[n].id != graph->inter[graph->nChannels*2])) continue;
|
||||
maxWidth = std::max(paths[n].width, maxWidth);
|
||||
}
|
||||
for (int n=0; n<system->nodes[NET].count; n++) {
|
||||
if (graph->crossNic != 1 && (system->nodes[NET].nodes[n].id != graph->inter[graph->nChannels*2])) continue;
|
||||
if (paths[n].width == maxWidth) {
|
||||
struct ncclTopoNode* net = system->nodes[NET].nodes+n;
|
||||
int typeSave = graph->type;
|
||||
NCCLCHECK(ncclTopoFollowPath(graph, paths+n, &net, graph->speedInter, typeSave));
|
||||
if (net) {
|
||||
graph->inter[graph->nChannels*2+1] = net->id;
|
||||
NCCLCHECK(ncclTopoSearchRecGpu(system, graph, saveGraph, gpu, step, -1, backToFirstRank, forcedOrder, maxSpeed, time));
|
||||
NCCLCHECK(ncclTopoFollowPath(graph, paths+n, &net, -graph->speedInter, typeSave));
|
||||
}
|
||||
struct ncclTopoNode* net = system->nodes[NET].nodes+n;
|
||||
if (graph->crossNic != 1 && (net->net.asic != startNet->net.asic || net->net.port != startNet->net.port)) continue;
|
||||
NCCLCHECK(ncclTopoFollowPath(system, graph, GPU, g, NET, n, 1, &net));
|
||||
if (net) {
|
||||
graph->inter[graph->nChannels*2+1] = net->id;
|
||||
NCCLCHECK(ncclTopoSearchRecGpu(system, graph, saveGraph, gpu, step, -1, backToFirstRank, forcedOrder, time));
|
||||
NCCLCHECK(ncclTopoFollowPath(system, graph, GPU, g, NET, n, -1, &net));
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (step < system->nodes[GPU].count-1) {
|
||||
// Go to next GPU
|
||||
struct ncclTopoLinkList* paths = gpu->paths[GPU];
|
||||
int next[NCCL_TOPO_MAX_NODES];
|
||||
int count;
|
||||
if (forcedOrder == FORCED_ORDER_PCI) { // Try the PCI order
|
||||
next[0] = step+1;
|
||||
next[0] = (busIdToCudaDev(gpu->id)+1)%system->nodes[GPU].count;
|
||||
count = 1;
|
||||
} else if (forcedOrder == FORCED_ORDER_REPLAY) { // Try last channel order
|
||||
NCCLCHECK(ncclTopoReplayGetGpu(system, graph, step, next));
|
||||
@@ -253,64 +354,64 @@ ncclResult_t ncclTopoSearchRecGpu(struct ncclTopoSystem* system, struct ncclTopo
|
||||
NCCLCHECK(ncclTopoSearchNextGpuSort(system, graph, gpu, next, &count, backToNet == -1 ? 0 : backToNet == step+1 ? 1 : -1 ));
|
||||
}
|
||||
for (int i=0; i<count; i++) {
|
||||
int g = next[i];
|
||||
int nvlink = graph->nvlink;
|
||||
graph->nvlink &= paths[g].type <= LINK_NVL ? 1 : 0;
|
||||
int speed = graph->speedIntra;
|
||||
if (paths[g].type == LINK_QPI) speed = INTEL_P2P_OVERHEAD(speed);
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, paths, step+1, backToNet, backToFirstRank, forcedOrder, maxSpeed, time, g, speed));
|
||||
graph->nvlink = nvlink;
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, step+1, backToNet, backToFirstRank, forcedOrder, time, GPU, g, next[i]));
|
||||
}
|
||||
} else if (step == backToFirstRank) {
|
||||
// Find first GPU and loop back to it
|
||||
int g;
|
||||
int rank = graph->intra[graph->nChannels*ngpus];
|
||||
for (g=0; g<ngpus; g++) {
|
||||
if (system->nodes[GPU].nodes[g].rank == rank) break;
|
||||
}
|
||||
if (g == ngpus) {
|
||||
WARN("Could not find GPU with rank %d\n", rank);
|
||||
return ncclInternalError;
|
||||
}
|
||||
struct ncclTopoLinkList* paths = gpu->paths[GPU];
|
||||
struct ncclTopoNode* firstGpu = system->nodes[GPU].nodes+g;
|
||||
int typeSave = graph->type;
|
||||
NCCLCHECK(ncclTopoFollowPath(graph, paths+g, &firstGpu, graph->speedIntra, typeSave));
|
||||
int p;
|
||||
NCCLCHECK(getGpuIndex(system, graph->intra[graph->nChannels*ngpus], &p));
|
||||
struct ncclTopoNode* firstGpu;
|
||||
NCCLCHECK(ncclTopoFollowPath(system, graph, GPU, g, GPU, p, 1, &firstGpu));
|
||||
if (firstGpu) {
|
||||
NCCLCHECK(ncclTopoSearchRecGpu(system, graph, saveGraph, firstGpu, step+1, backToNet, -1, forcedOrder, maxSpeed, time));
|
||||
NCCLCHECK(ncclTopoFollowPath(graph, paths+g, &firstGpu, -graph->speedIntra, typeSave));
|
||||
NCCLCHECK(ncclTopoSearchRecGpu(system, graph, saveGraph, firstGpu, step+1, backToNet, -1, forcedOrder, time));
|
||||
NCCLCHECK(ncclTopoFollowPath(system, graph, GPU, g, GPU, p, -1, &firstGpu));
|
||||
}
|
||||
} else {
|
||||
// Next path
|
||||
NCCLCHECK(ncclTopoSearchRecGpu(system, graph, saveGraph, gpu, ngpus, -1, -1, forcedOrder, maxSpeed, time));
|
||||
NCCLCHECK(ncclTopoSearchRecGpu(system, graph, saveGraph, gpu, ngpus, -1, -1, forcedOrder, time));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoSearchRecNet(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, int backToNet, int backToFirstRank, int maxSpeed, int* time) {
|
||||
const uint64_t flag = 1ULL<<(graph->nChannels);
|
||||
ncclResult_t ncclTopoSearchRecNet(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, int backToNet, int backToFirstRank, int* time) {
|
||||
const int speed = graph->speedInter;
|
||||
for (int n=0; n<system->nodes[NET].count; n++) {
|
||||
struct ncclTopoNode* net = system->nodes[NET].nodes+n;
|
||||
struct ncclTopoNode* gpu;
|
||||
if (net->used == 0) {
|
||||
graph->inter[graph->nChannels*2] = net->id;
|
||||
for (int i=0; i<system->nodes[NET].count; i++) {
|
||||
if (system->nodes[NET].nodes[i].rank == net->rank) system->nodes[NET].nodes[i].used ^= flag;
|
||||
}
|
||||
struct ncclTopoLinkList* paths = net->paths[GPU];
|
||||
if (graph->collNet && net->net.collSupport == 0) continue;
|
||||
if (net->net.width < speed) continue;
|
||||
if (net->net.maxChannels == 0) continue;
|
||||
|
||||
// First try the PCI order to set a reference
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, paths, 0, backToNet, backToFirstRank, FORCED_ORDER_PCI, maxSpeed, time, 0, speed));
|
||||
// Then try to replay the last channel
|
||||
if (graph->nChannels > 0) {
|
||||
int g;
|
||||
NCCLCHECK(ncclTopoReplayGetGpu(system, graph, -1, &g));
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, paths, 0, backToNet, backToFirstRank, FORCED_ORDER_REPLAY, maxSpeed, time, g, speed));
|
||||
graph->inter[graph->nChannels*2] = net->id;
|
||||
for (int i=0; i<system->nodes[NET].count; i++) {
|
||||
if ((system->nodes[NET].nodes[i].net.asic == net->net.asic) &&
|
||||
(system->nodes[NET].nodes[i].net.port == net->net.port)) {
|
||||
system->nodes[NET].nodes[i].net.width -= speed;
|
||||
}
|
||||
}
|
||||
net->net.maxChannels--;
|
||||
|
||||
// First try to replay the last channel
|
||||
if (graph->nChannels > 0) {
|
||||
int g;
|
||||
NCCLCHECK(ncclTopoReplayGetGpu(system, graph, -1, &g));
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, 0, backToNet, backToFirstRank, FORCED_ORDER_REPLAY, time, NET, n, g));
|
||||
}
|
||||
if (graph->nChannels == 0 || graph->sameChannels == 0) {
|
||||
if (graph->nChannels == 0) {
|
||||
// Always try the PCI order first to set a reference
|
||||
struct ncclTopoLinkList* paths = net->paths[GPU];
|
||||
// find the first GPU that is closest to NIC
|
||||
int f = 0;
|
||||
for (int i = 0; i<system->nodes[GPU].count; i++)
|
||||
if (paths[i].count < paths[f].count) f = i;
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, 0, backToNet, backToFirstRank, FORCED_ORDER_PCI, time, NET, n, f));
|
||||
}
|
||||
|
||||
// Then try the most local GPUs
|
||||
int maxWidth = 0, minHops = 0xfffffff;
|
||||
float maxWidth = 0;
|
||||
int minHops = 0xfffffff;
|
||||
struct ncclTopoLinkList* paths = net->paths[GPU];
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
if (paths[g].width > maxWidth) {
|
||||
maxWidth = paths[g].width;
|
||||
@@ -329,14 +430,19 @@ ncclResult_t ncclTopoSearchRecNet(struct ncclTopoSystem* system, struct ncclTopo
|
||||
gpu = system->nodes[GPU].nodes+g;
|
||||
int gpuUsed = gpuPciWidth(gpu) > 0 ? 0 : 1;
|
||||
if (tryGpuBidir == gpuUsed) {
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, paths, 0, backToNet, backToFirstRank, 0, maxSpeed, time, g, speed));
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, 0, backToNet, backToFirstRank, 0, time, NET, n, g));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (int i=0; i<system->nodes[NET].count; i++) {
|
||||
if (system->nodes[NET].nodes[i].rank == net->rank) system->nodes[NET].nodes[i].used ^= flag;
|
||||
}
|
||||
|
||||
net->net.maxChannels++;
|
||||
for (int i=0; i<system->nodes[NET].count; i++) {
|
||||
if ((system->nodes[NET].nodes[i].net.asic == net->net.asic) &&
|
||||
(system->nodes[NET].nodes[i].net.port == net->net.port)) {
|
||||
system->nodes[NET].nodes[i].net.width += speed;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -375,17 +481,152 @@ ncclResult_t ncclTopoSearchParams(struct ncclTopoSystem* system, int pattern, in
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoSearchRec(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, int maxSpeed, int* time) {
|
||||
ncclResult_t ncclTopoSearchRec(struct ncclTopoSystem* system, struct ncclTopoGraph* graph, struct ncclTopoGraph* saveGraph, int* time) {
|
||||
int backToNet, backToFirstRank;
|
||||
NCCLCHECK(ncclTopoSearchParams(system, graph->pattern, &backToNet, &backToFirstRank));
|
||||
if (system->nodes[NET].count) {
|
||||
// Start from NET
|
||||
ncclTopoSearchRecNet(system, graph, saveGraph, backToNet, backToFirstRank, maxSpeed, time);
|
||||
ncclTopoSearchRecNet(system, graph, saveGraph, backToNet, backToFirstRank, time);
|
||||
} else {
|
||||
// Start from GPU 0
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, NULL, 0, backToNet, backToFirstRank, FORCED_ORDER_PCI, maxSpeed, time, 0, graph->speedIntra));
|
||||
if (graph->nChannels > 0) NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, NULL, 0, backToNet, backToFirstRank, FORCED_ORDER_REPLAY, maxSpeed, time, 0, graph->speedIntra));
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, NULL, 0, backToNet, backToFirstRank, 0, maxSpeed, time, 0, graph->speedIntra));
|
||||
// Intra-node only.
|
||||
if (graph->nChannels == 0) {
|
||||
// Try PCI order first
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, 0, backToNet, backToFirstRank, FORCED_ORDER_PCI, time, -1, -1, 0));
|
||||
} else {
|
||||
// Also try to replay previous channel
|
||||
int g;
|
||||
NCCLCHECK(ncclTopoReplayGetGpu(system, graph, -1, &g));
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, 0, backToNet, backToFirstRank, FORCED_ORDER_REPLAY, time, -1, -1, g));
|
||||
}
|
||||
if (graph->sameChannels == 0 || graph->nChannels == 0) {
|
||||
// Finally, try all other possibilities unless we are forced to use the same channels
|
||||
for (int g=0; g<system->nodes[GPU].count; g++) {
|
||||
NCCLCHECK(ncclTopoSearchTryGpu(system, graph, saveGraph, 0, backToNet, backToFirstRank, 0, time, -1, -1, g));
|
||||
}
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
/************************************/
|
||||
/* User defined graph from XML file */
|
||||
/************************************/
|
||||
|
||||
struct kvDict kvDictLinkType[] = { { "SYS", PATH_SYS }, { "PHB", PATH_PHB }, { "PIX", PATH_PIX }, { "PXB", PATH_PXB }, { "NVL", PATH_NVL }, { "LOC", PATH_LOC }, { NULL, 0 } };
|
||||
ncclResult_t ncclTopoGetChannelFromXml(struct ncclXmlNode *xmlChannel, int c, struct ncclTopoSystem* system, struct ncclTopoGraph* graph) {
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
int* inter = graph->inter+2*c;
|
||||
int* intra = graph->intra+ngpus*c;
|
||||
int n=0, g=0;
|
||||
for (int s=0; s<xmlChannel->nSubs; s++) {
|
||||
struct ncclXmlNode* sub = xmlChannel->subs[s];
|
||||
int dev;
|
||||
NCCLCHECK(xmlGetAttrInt(sub, "dev", &dev));
|
||||
if (strcmp(sub->name, "net") == 0) {
|
||||
inter[n++] = dev;
|
||||
} else if (strcmp(sub->name, "gpu") == 0) {
|
||||
int rank = -1;
|
||||
for (int g=0; g<ngpus; g++) {
|
||||
if (system->nodes[GPU].nodes[g].gpu.dev == dev) rank = system->nodes[GPU].nodes[g].gpu.rank;
|
||||
}
|
||||
if (rank == -1) {
|
||||
WARN("XML Import Channel : dev %d not found.", dev);
|
||||
return ncclSystemError;
|
||||
}
|
||||
intra[g++] = rank;
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
ncclResult_t ncclTopoGetGraphFromXmlSub(struct ncclXmlNode *xmlGraph, struct ncclTopoSystem* system, struct ncclTopoGraph* graph) {
|
||||
int id;
|
||||
NCCLCHECK(xmlGetAttrInt(xmlGraph, "id", &id));
|
||||
if (graph->id != id) return ncclSuccess;
|
||||
|
||||
int crossNic;
|
||||
NCCLCHECK(xmlGetAttrInt(xmlGraph, "crossnic", &crossNic));
|
||||
if (graph->crossNic == 0 && crossNic == 1) return ncclSuccess;
|
||||
graph->crossNic = crossNic;
|
||||
|
||||
NCCLCHECK(xmlGetAttrInt(xmlGraph, "pattern", &graph->pattern));
|
||||
NCCLCHECK(xmlGetAttrInt(xmlGraph, "nchannels", &graph->nChannels));
|
||||
NCCLCHECK(xmlGetAttrFloat(xmlGraph, "speedintra", &graph->speedIntra));
|
||||
NCCLCHECK(xmlGetAttrFloat(xmlGraph, "speedinter", &graph->speedInter));
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttr(xmlGraph, "typeintra", &str));
|
||||
NCCLCHECK(kvConvertToInt(str, &graph->typeIntra, kvDictLinkType));
|
||||
NCCLCHECK(xmlGetAttr(xmlGraph, "typeinter", &str));
|
||||
NCCLCHECK(kvConvertToInt(str, &graph->typeInter, kvDictLinkType));
|
||||
NCCLCHECK(xmlGetAttrInt(xmlGraph, "samechannels", &graph->sameChannels));
|
||||
for (int s=0; s<xmlGraph->nSubs; s++) {
|
||||
NCCLCHECK(ncclTopoGetChannelFromXml(xmlGraph->subs[s], s, system, graph));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
ncclResult_t ncclTopoGetGraphFromXml(struct ncclXmlNode *xmlGraphs, struct ncclTopoSystem* system, struct ncclTopoGraph* graph) {
|
||||
for (int s=0; s<xmlGraphs->nSubs; s++) {
|
||||
NCCLCHECK(ncclTopoGetGraphFromXmlSub(xmlGraphs->subs[s], system, graph));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
/* And the reverse : graph->xml */
|
||||
ncclResult_t ncclTopoGetXmlFromChannel(struct ncclTopoGraph* graph, int c, struct ncclTopoSystem* system, struct ncclXml *xml, struct ncclXmlNode* parent) {
|
||||
struct ncclXmlNode* xmlChannel;
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
int* inter = graph->inter+2*c;
|
||||
int* intra = graph->intra+ngpus*c;
|
||||
NCCLCHECK(xmlAddNode(xml, parent, "channel", &xmlChannel));
|
||||
struct ncclXmlNode* node;
|
||||
if (system->nodes[NET].count) {
|
||||
NCCLCHECK(xmlAddNode(xml, xmlChannel, "net", &node));
|
||||
NCCLCHECK(xmlSetAttrInt(node, "dev", inter[0]));
|
||||
}
|
||||
for (int g=0; g<ngpus; g++) {
|
||||
NCCLCHECK(xmlAddNode(xml, xmlChannel, "gpu", &node));
|
||||
int dev = -1;
|
||||
for (int i=0; i<ngpus; i++) {
|
||||
if (system->nodes[GPU].nodes[i].gpu.rank == intra[g]) dev = system->nodes[GPU].nodes[i].gpu.dev;
|
||||
}
|
||||
if (dev == -1) {
|
||||
WARN("XML Export Channel : rank %d not found.", intra[g]);
|
||||
return ncclInternalError;
|
||||
}
|
||||
NCCLCHECK(xmlSetAttrInt(node, "dev", dev));
|
||||
}
|
||||
if (system->nodes[NET].count) {
|
||||
NCCLCHECK(xmlAddNode(xml, xmlChannel, "net", &node));
|
||||
NCCLCHECK(xmlSetAttrInt(node, "dev", inter[1]));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
ncclResult_t ncclTopoGetXmlFromGraph(struct ncclTopoGraph* graph, struct ncclTopoSystem* system, struct ncclXml *xml, struct ncclXmlNode* parent) {
|
||||
struct ncclXmlNode* xmlGraph;
|
||||
NCCLCHECK(xmlAddNode(xml, parent, "graph", &xmlGraph));
|
||||
NCCLCHECK(xmlSetAttrInt(xmlGraph, "id", graph->id));
|
||||
NCCLCHECK(xmlSetAttrInt(xmlGraph, "pattern", graph->pattern));
|
||||
NCCLCHECK(xmlSetAttrInt(xmlGraph, "crossnic", graph->crossNic));
|
||||
NCCLCHECK(xmlSetAttrInt(xmlGraph, "nchannels", graph->nChannels));
|
||||
NCCLCHECK(xmlSetAttrFloat(xmlGraph, "speedintra", graph->speedIntra));
|
||||
NCCLCHECK(xmlSetAttrFloat(xmlGraph, "speedinter", graph->speedInter));
|
||||
const char* str;
|
||||
NCCLCHECK(kvConvertToStr(graph->typeIntra, &str, kvDictLinkType));
|
||||
NCCLCHECK(xmlSetAttr(xmlGraph, "typeintra", str));
|
||||
NCCLCHECK(kvConvertToStr(graph->typeInter, &str, kvDictLinkType));
|
||||
NCCLCHECK(xmlSetAttr(xmlGraph, "typeinter", str));
|
||||
NCCLCHECK(xmlSetAttrInt(xmlGraph, "samechannels", graph->sameChannels));
|
||||
for (int c=0; c<graph->nChannels; c++) {
|
||||
NCCLCHECK(ncclTopoGetXmlFromChannel(graph, c, system, xml, xmlGraph));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
ncclResult_t ncclTopoGetXmlFromGraphs(int ngraphs, struct ncclTopoGraph** graphs, struct ncclTopoSystem* system, struct ncclXml *xml) {
|
||||
xml->maxIndex = 0;
|
||||
struct ncclXmlNode* xmlGraphs;
|
||||
NCCLCHECK(xmlAddNode(xml, NULL, "graphs", &xmlGraphs));
|
||||
NCCLCHECK(xmlSetAttrInt(xmlGraphs, "version", NCCL_GRAPH_XML_VERSION));
|
||||
for (int g=0; g<ngraphs; g++) {
|
||||
NCCLCHECK(ncclTopoGetXmlFromGraph(graphs[g], system, xml, xmlGraphs));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -456,11 +697,16 @@ static void parseChordalRing(struct ncclTopoSystem* system, char **str) {
|
||||
for (i=0; i<ngpus; i++) {
|
||||
struct ncclTopoNode* node = system->nodes[GPU].nodes+i;
|
||||
if (node->paths[GPU] == NULL) continue;
|
||||
int sum = ngpus*(ngpus-1)/2 - node->rank;
|
||||
int sum = ngpus*(ngpus-1)/2 - node->gpu.rank;
|
||||
int count = 0;
|
||||
for (int n = 0; n<ngpus; n++) {
|
||||
if (node->paths[GPU][n].type != LINK_NVL) continue;
|
||||
sum -= system->nodes[GPU].nodes[n].rank;
|
||||
struct ncclTopoLink* link;
|
||||
for (link = node->links; link->remNode; link++) {
|
||||
if (link->remNode->gpu.rank == n) break;
|
||||
}
|
||||
if (!link->remNode) continue;
|
||||
if (link->type != LINK_NVL) continue;
|
||||
sum -= system->nodes[GPU].nodes[n].gpu.rank;
|
||||
count ++;
|
||||
}
|
||||
if(count != ngpus-2 || sum < 0 || sum > ngpus-1) {
|
||||
@@ -492,28 +738,39 @@ static void parseChordalRing(struct ncclTopoSystem* system, char **str) {
|
||||
return;
|
||||
}
|
||||
|
||||
float speedArray[] = { 24.0, 20.0, 18.0, 15.0, 12.0, 10.0, 9.0, 7.0, 6.0, 5.0, 4.0, 3.0, 2.4, 1.2, 0.24, 0.12 };
|
||||
#define NSPEEDS (sizeof(speedArray)/sizeof(float))
|
||||
|
||||
ncclResult_t ncclTopoCompute(ncclTopoSystem* system, struct ncclTopoGraph* graph) {
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
int crossNic = (system->nodes[NET].count > 1) && graph->crossNic ? 1 : 0;
|
||||
graph->speedIntra = graph->speedInter = 0;
|
||||
if (graph->crossNic == 2) graph->crossNic = 0;
|
||||
graph->nvlink = 0;
|
||||
graph->type = LINK_LOC;
|
||||
graph->typeIntra = ngpus == 1 ? PATH_LOC : PATH_NVL;
|
||||
graph->typeInter = PATH_PIX;
|
||||
graph->nChannels = 0;
|
||||
graph->sameChannels = 1;
|
||||
|
||||
char* str = getenv("NCCL_GRAPH");
|
||||
char* str = getenv("NCCL_GRAPH_FILE");
|
||||
if (str) {
|
||||
struct ncclXml* xml;
|
||||
NCCLCHECK(ncclCalloc(&xml, 1));
|
||||
NCCLCHECK(ncclTopoGetXmlGraphFromFile(str, xml));
|
||||
NCCLCHECK(ncclTopoGetGraphFromXml(xml->nodes, system, graph));
|
||||
free(xml);
|
||||
if (graph->nChannels > 0) return ncclSuccess;
|
||||
}
|
||||
|
||||
if (!str) parseChordalRing(system, &str);
|
||||
if (str) {
|
||||
NCCLCHECK(parseGraph(str, &graph->nChannels, ngpus, graph->intra));
|
||||
for (int i=0; i<graph->nChannels*ngpus; i++) {
|
||||
// Translate gpu numbers into ranks
|
||||
graph->intra[i] = system->nodes[GPU].nodes[graph->intra[i]].rank;
|
||||
graph->intra[i] = system->nodes[GPU].nodes[graph->intra[i]].gpu.rank;
|
||||
}
|
||||
// TODO : let user specify NICs
|
||||
graph->inter[0] = graph->inter[1] = 0;
|
||||
graph->speedIntra = graph->speedInter = system->maxWidth;
|
||||
graph->nvlink = 0;
|
||||
if (graph->pattern == NCCL_TOPO_PATTERN_RING) {
|
||||
// Reverse the loop
|
||||
for (int c=0; c<graph->nChannels; c++) {
|
||||
@@ -531,22 +788,24 @@ ncclResult_t ncclTopoCompute(ncclTopoSystem* system, struct ncclTopoGraph* graph
|
||||
|
||||
struct ncclTopoGraph tmpGraph;
|
||||
memcpy(&tmpGraph, graph, sizeof(struct ncclTopoGraph));
|
||||
int bestSpeed = 0;
|
||||
|
||||
// First try crossnic, then decrease speed and finally increase speedIntra.
|
||||
tmpGraph.speedIntra = tmpGraph.speedInter = system->maxWidth;
|
||||
int maxSpeed = system->maxSpeed;
|
||||
tmpGraph.pattern = graph->pattern;
|
||||
int pass = 1;
|
||||
int speedIndex = 0;
|
||||
while (speedArray[speedIndex] > system->maxWidth && speedIndex < NSPEEDS-1) speedIndex++;
|
||||
tmpGraph.speedIntra = tmpGraph.speedInter = speedArray[speedIndex];
|
||||
int64_t globalTimeout = NCCL_SEARCH_GLOBAL_TIMEOUT;
|
||||
|
||||
search:
|
||||
int time = NCCL_SEARCH_TIMEOUT;
|
||||
int stepSpeed = system->maxWidth/4;
|
||||
tmpGraph.nvlink = 1;
|
||||
int time = tmpGraph.sameChannels ? NCCL_SEARCH_TIMEOUT_SAMECHANNELS :
|
||||
tmpGraph.pattern == NCCL_TOPO_PATTERN_TREE ? NCCL_SEARCH_TIMEOUT_TREE : NCCL_SEARCH_TIMEOUT;
|
||||
tmpGraph.nChannels = 0;
|
||||
tmpGraph.sameChannels = 1;
|
||||
NCCLCHECK(ncclTopoSearchRec(system, &tmpGraph, graph, maxSpeed, &time));
|
||||
globalTimeout -= time;
|
||||
|
||||
NCCLCHECK(ncclTopoSearchRec(system, &tmpGraph, graph, &time));
|
||||
#if 0
|
||||
printf("Pattern %d, crossNic %d, Speed %d/%d, type %d -> nChannels %dx%d/%d %s\n", tmpGraph.pattern, tmpGraph.crossNic, tmpGraph.speedInter, tmpGraph.speedIntra, tmpGraph.type, graph->nChannels, graph->speedInter, graph->speedIntra, time == 0 ? "TIMEOUT" : "");
|
||||
printf("Pattern %d, crossNic %d, Speed %g/%g, type %d/%d, channels %d-%d sameChannels %d -> nChannels %dx%g/%g %s\n", tmpGraph.pattern, tmpGraph.crossNic, tmpGraph.speedInter, tmpGraph.speedIntra, tmpGraph.typeInter, tmpGraph.typeIntra, tmpGraph.minChannels, tmpGraph.maxChannels, tmpGraph.sameChannels, graph->nChannels, graph->speedInter, graph->speedIntra, time == 0 ? "TIMEOUT" : "");
|
||||
for (int c=0; c<graph->nChannels; c++) {
|
||||
printf("%2d : ", c);
|
||||
for (int g=0; g<ngpus; g++) {
|
||||
@@ -555,13 +814,34 @@ search:
|
||||
printf("\n");
|
||||
}
|
||||
#endif
|
||||
if (time == -1) goto done;
|
||||
// We already have a solution and we timed out so lower speed will just timeout as well
|
||||
if (time == 0 && graph->nChannels > 0) goto done;
|
||||
if ((graph->nChannels > 0) && (bestSpeed == 0)) bestSpeed = graph->speedIntra;
|
||||
// Optimal solution, stop here
|
||||
if (graph->nChannels == graph->maxChannels && graph->speedInter == system->maxWidth) goto done;
|
||||
|
||||
if (tmpGraph.speedIntra == tmpGraph.speedInter) {
|
||||
// First pass, we don't have a solution yet ; try to go slower.
|
||||
if (pass == 1) {
|
||||
// First pass, we don't have a solution yet ; try other options
|
||||
|
||||
// Try having different channels
|
||||
if (tmpGraph.sameChannels == 1) {
|
||||
tmpGraph.sameChannels = 0;
|
||||
goto search;
|
||||
}
|
||||
tmpGraph.sameChannels = 1;
|
||||
|
||||
if (time != -1) globalTimeout += time;
|
||||
else globalTimeout = NCCL_SEARCH_GLOBAL_TIMEOUT;
|
||||
if (globalTimeout < 0) goto done;
|
||||
|
||||
int maxTypeIntra = system->nodes[NET].count > 0 ? tmpGraph.typeInter : PATH_SYS;
|
||||
if (tmpGraph.typeIntra < maxTypeIntra && (graph->nChannels == 0 || tmpGraph.typeIntra < graph->typeIntra)) {
|
||||
tmpGraph.typeIntra += 1;
|
||||
goto search;
|
||||
}
|
||||
tmpGraph.typeIntra = ngpus == 1 ? PATH_LOC : PATH_NVL;
|
||||
if (system->nodes[NET].count > 0 && tmpGraph.typeInter < PATH_SYS && (graph->nChannels == 0 || tmpGraph.typeInter < graph->typeInter || tmpGraph.typeInter < PATH_PXB)) {
|
||||
tmpGraph.typeInter += 1;
|
||||
goto search;
|
||||
}
|
||||
tmpGraph.typeInter = PATH_PIX;
|
||||
|
||||
// Try a simpler tree
|
||||
if (tmpGraph.pattern == NCCL_TOPO_PATTERN_SPLIT_TREE_LOOP) {
|
||||
@@ -574,50 +854,61 @@ search:
|
||||
}
|
||||
tmpGraph.pattern = graph->pattern;
|
||||
|
||||
if (tmpGraph.type < LINK_QPI) {
|
||||
tmpGraph.type += 1;
|
||||
goto search;
|
||||
}
|
||||
tmpGraph.type = graph->type;
|
||||
|
||||
if (crossNic && tmpGraph.crossNic == 0) {
|
||||
// Try again with crossNic if permitted
|
||||
tmpGraph.crossNic = crossNic;
|
||||
goto search;
|
||||
}
|
||||
tmpGraph.crossNic = graph->crossNic;
|
||||
tmpGraph.crossNic = 0;
|
||||
|
||||
// Decrease speed until we find a solution
|
||||
if ((speedIndex < NSPEEDS-1) && (graph->nChannels == 0 || (speedArray[speedIndex+1]/graph->speedInter > .49))) {
|
||||
tmpGraph.speedInter = tmpGraph.speedIntra = speedArray[++speedIndex];
|
||||
goto search;
|
||||
}
|
||||
speedIndex = 0;
|
||||
while (speedArray[speedIndex] > system->maxWidth && speedIndex < NSPEEDS-1) speedIndex++;
|
||||
tmpGraph.speedIntra = tmpGraph.speedInter = speedArray[speedIndex];
|
||||
|
||||
// Try to reduce speed per channel
|
||||
tmpGraph.speedIntra = tmpGraph.speedInter -= stepSpeed;
|
||||
if (tmpGraph.speedIntra >= bestSpeed/2 && tmpGraph.speedIntra >= stepSpeed) goto search;
|
||||
}
|
||||
|
||||
done:
|
||||
// We have a solution now. See if we can increase speedIntra
|
||||
if (tmpGraph.speedIntra == tmpGraph.speedInter) {
|
||||
// We have a solution. Start from that solution and move to pass 2.
|
||||
if (pass == 1) {
|
||||
time = -1;
|
||||
memcpy(&tmpGraph, graph, sizeof(tmpGraph));
|
||||
speedIndex = 0;
|
||||
while (speedArray[speedIndex] > graph->speedInter && speedIndex < NSPEEDS-1) speedIndex++;
|
||||
tmpGraph.speedIntra = tmpGraph.speedInter = speedArray[speedIndex];
|
||||
tmpGraph.minChannels = graph->nChannels;
|
||||
pass = 2;
|
||||
}
|
||||
|
||||
// 3. See if we can increase speedIntra for trees (2 nodes or collnet)
|
||||
if (pass == 2) {
|
||||
if (time != 0 && graph->pattern != NCCL_TOPO_PATTERN_RING &&
|
||||
tmpGraph.speedIntra == graph->speedIntra && tmpGraph.speedIntra < tmpGraph.speedInter*2 &&
|
||||
speedIndex > 0) {
|
||||
tmpGraph.speedIntra = speedArray[--speedIndex];
|
||||
goto search;
|
||||
}
|
||||
time = -1;
|
||||
memcpy(&tmpGraph, graph, sizeof(tmpGraph));
|
||||
}
|
||||
if (time != 0 && tmpGraph.pattern != NCCL_TOPO_PATTERN_RING && tmpGraph.speedIntra == graph->speedIntra) {
|
||||
// Try to increase the intra speed only but keeping nChannels the same
|
||||
tmpGraph.speedIntra += stepSpeed;
|
||||
maxSpeed = tmpGraph.speedIntra * graph->nChannels;
|
||||
if (tmpGraph.speedIntra <= tmpGraph.speedInter*2) goto search;
|
||||
}
|
||||
|
||||
if (graph->nChannels == 0) {
|
||||
if (graph->nChannels == 0 && graph->collNet == 0) {
|
||||
WARN("Could not find a path for pattern %d, falling back to simple order\n", graph->pattern);
|
||||
for (int i=0; i<ngpus; i++) graph->intra[i] = system->nodes[GPU].nodes[i].rank;
|
||||
for (int i=0; i<ngpus; i++) graph->intra[i] = system->nodes[GPU].nodes[i].gpu.rank;
|
||||
graph->inter[0] = graph->inter[1] = 0;
|
||||
graph->speedIntra = graph->speedInter = stepSpeed;
|
||||
graph->nvlink = 0;
|
||||
graph->speedIntra = graph->speedInter = 0.1;
|
||||
graph->typeIntra = graph->typeInter = PATH_SYS;
|
||||
graph->nChannels = 1;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoPrintGraph(struct ncclTopoSystem* system, struct ncclTopoGraph* graph) {
|
||||
INFO(NCCL_GRAPH, "Pattern %d, crossNic %d, nChannels %d, speed %d/%d, nvlink %d, type %d, sameChannels %d", graph->pattern, graph->crossNic, graph->nChannels, graph->speedIntra, graph->speedInter, graph->nvlink, graph->type, graph->sameChannels);
|
||||
INFO(NCCL_GRAPH, "Pattern %d, crossNic %d, nChannels %d, speed %f/%f, type %s/%s, sameChannels %d", graph->pattern, graph->crossNic, graph->nChannels, graph->speedIntra, graph->speedInter, topoPathTypeStr[graph->typeIntra], topoPathTypeStr[graph->typeInter], graph->sameChannels);
|
||||
int ngpus = system->nodes[GPU].count;
|
||||
|
||||
char line[1024];
|
||||
@@ -641,6 +932,18 @@ ncclResult_t ncclTopoPrintGraph(struct ncclTopoSystem* system, struct ncclTopoGr
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoDumpGraphs(struct ncclTopoSystem* system, int ngraphs, struct ncclTopoGraph** graphs) {
|
||||
char* str = getenv("NCCL_GRAPH_DUMP_FILE");
|
||||
if (str) {
|
||||
struct ncclXml* xml;
|
||||
NCCLCHECK(ncclCalloc(&xml, 1));
|
||||
NCCLCHECK(ncclTopoGetXmlFromGraphs(ngraphs, graphs, system, xml));
|
||||
NCCLCHECK(ncclTopoDumpXmlToFile(str, xml));
|
||||
free(xml);
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetNetDev(struct ncclTopoGraph* graph, int dir, int channelId, int* dev) {
|
||||
*dev = graph->inter[(channelId%graph->nChannels)*2+dir];
|
||||
return ncclSuccess;
|
||||
|
||||
+539
-575
File diff suppressed because it is too large
Load Diff
@@ -1,5 +1,5 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2016-2019, NVIDIA CORPORATION. All rights reserved.
|
||||
* Copyright (c) 2016-2020, NVIDIA CORPORATION. All rights reserved.
|
||||
* Modifications Copyright (c) 2019-2020 Advanced Micro Devices, Inc. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
@@ -10,27 +10,28 @@
|
||||
|
||||
#include "graph.h"
|
||||
#include "core.h"
|
||||
#include <sched.h>
|
||||
|
||||
#define LOC_WIDTH 5000
|
||||
#define PASCAL_NVLINK_WIDTH 18
|
||||
#define VOLTA_NVLINK_WIDTH 21
|
||||
#define PCI_WIDTH 12 // PCI Gen3 x16
|
||||
#define QPI_WIDTH 8
|
||||
#define SKL_QPI_WIDTH 12
|
||||
#define SKL_PCI_WIDTH 12
|
||||
#define SKL_CPUPCI_WIDTH 12
|
||||
#define P9_WIDTH 32
|
||||
#define NET_WIDTH 12 // 100Gbit
|
||||
#define ROME_QPI_WIDTH 18
|
||||
#define ROME_PCI_WIDTH 18
|
||||
#define ROME_CPUPCI_WIDTH 18
|
||||
#define LOC_WIDTH 5000.0
|
||||
#define PASCAL_NVLINK_WIDTH 18.0
|
||||
#define VOLTA_NVLINK_WIDTH 21.0
|
||||
#define PCI_WIDTH 12.0 // PCI Gen3 x16
|
||||
#define QPI_WIDTH 6.0
|
||||
#define SKL_QPI_WIDTH 9.0
|
||||
#define P9_WIDTH 32.0
|
||||
#define ARM_WIDTH 6.0
|
||||
#define NET_WIDTH 12.0 // 100Gbit
|
||||
#define VEGA_XGMI_WIDTH 20.0
|
||||
#define ROME_QPI_WIDTH 18.0
|
||||
#define ROME_PCI_WIDTH 18.0
|
||||
#define ROME_CPUPCI_WIDTH 18.0
|
||||
|
||||
// Intel CPU convert GPU P2P traffic into 64B PCI TLPs, to GPU
|
||||
// to GPU traffic consumed more PCI bandwidth.
|
||||
// Intel CPU convert GPU P2P traffic into 64B PCI TLPs, so GPU
|
||||
// to GPU traffic consumes more PCI bandwidth.
|
||||
#define INTEL_P2P(speed) (speed*9/12)
|
||||
#define INTEL_P2P_OVERHEAD(speed) (speed*12/9)
|
||||
|
||||
#define NCCL_TOPO_NODE_TYPES 6
|
||||
#define NCCL_TOPO_NODE_TYPES 7
|
||||
#define GPU 0
|
||||
#define PCI 1
|
||||
#define NVS 2
|
||||
@@ -39,37 +40,72 @@
|
||||
#define NET 5
|
||||
extern const char* topoNodeTypeStr[];
|
||||
|
||||
// We want link types and path types to match as much as possible
|
||||
#define LINK_LOC 0
|
||||
#define LINK_NVL 1
|
||||
#define LINK_PCI 2
|
||||
#define LINK_QPI 3
|
||||
#define LINK_NET 4
|
||||
// Skipping 3 for PATH_PXB
|
||||
// Skipping 4 for PATH_PHB
|
||||
#define LINK_SYS 5
|
||||
#define LINK_NET 6
|
||||
extern const char* topoLinkTypeStr[];
|
||||
|
||||
#define PATH_LOC 0
|
||||
#define PATH_NVL 1
|
||||
#define PATH_PIX 2
|
||||
#define PATH_PXB 3
|
||||
#define PATH_PHB 4
|
||||
#define PATH_SYS 5
|
||||
#define PATH_NET 6
|
||||
extern const char* topoPathTypeStr[];
|
||||
|
||||
struct ncclTopoNode;
|
||||
struct ncclTopoLink {
|
||||
int type;
|
||||
int width;
|
||||
float width;
|
||||
struct ncclTopoNode* remNode;
|
||||
};
|
||||
#define NCCL_TOPO_MAX_LINKS 32
|
||||
#define NCCL_TOPO_MAX_HOPS (NCCL_TOPO_MAX_NODES*NCCL_TOPO_NODE_TYPES)
|
||||
#define SELECT_PATH 1
|
||||
#define SELECT_LAST 2
|
||||
|
||||
#define NET_GDR_MASK 0x70000000
|
||||
|
||||
struct ncclTopoLinkList {
|
||||
struct ncclTopoLink* list[NCCL_TOPO_MAX_HOPS];
|
||||
int count;
|
||||
int width;
|
||||
float width;
|
||||
int type;
|
||||
};
|
||||
|
||||
#define NCCL_TOPO_CPU_INTEL_BDW 1
|
||||
#define NCCL_TOPO_CPU_INTEL_SKL 2
|
||||
|
||||
#define NCCL_TOPO_UNDEF (-1)
|
||||
|
||||
struct ncclTopoNode {
|
||||
int type;
|
||||
int64_t id;
|
||||
int rank;
|
||||
// Type specific data
|
||||
union {
|
||||
struct {
|
||||
int dev; // NVML dev number
|
||||
int rank;
|
||||
int cudaCompCap;
|
||||
int gdrSupport;
|
||||
}gpu;
|
||||
struct {
|
||||
uint64_t asic;
|
||||
int port;
|
||||
float width;
|
||||
int gdrSupport;
|
||||
int collSupport;
|
||||
int maxChannels;
|
||||
}net;
|
||||
struct {
|
||||
int arch;
|
||||
int vendor;
|
||||
int model;
|
||||
cpu_set_t affinity;
|
||||
}cpu;
|
||||
};
|
||||
int nlinks;
|
||||
struct ncclTopoLink links[NCCL_TOPO_MAX_LINKS];
|
||||
// Pre-computed paths to GPUs and NICs
|
||||
@@ -85,60 +121,29 @@ struct ncclTopoNodeSet {
|
||||
|
||||
struct ncclTopoSystem {
|
||||
struct ncclTopoNodeSet nodes[NCCL_TOPO_NODE_TYPES];
|
||||
int maxSpeed;
|
||||
int maxWidth;
|
||||
int searchInitDone;
|
||||
float maxWidth;
|
||||
};
|
||||
|
||||
static ncclResult_t ncclTopoCreateNode(struct ncclTopoSystem* system, struct ncclTopoNode** node, int type, uint64_t id) {
|
||||
ncclResult_t ncclTopoGetNode(struct ncclTopoSystem* system, struct ncclTopoNode** node, int type, uint64_t id);
|
||||
ncclResult_t ncclTopoCreateNode(struct ncclTopoSystem* system, struct ncclTopoNode** node, int type, uint64_t id);
|
||||
ncclResult_t ncclTopoRemoveNode(struct ncclTopoSystem* system, int type, int id);
|
||||
ncclResult_t ncclTopoConnectNodes(struct ncclTopoNode* node, struct ncclTopoNode* remNode, int type, float width);
|
||||
ncclResult_t ncclTopoPrintPaths(struct ncclTopoSystem* system);
|
||||
ncclResult_t ncclTopoLoadSystem(const char* xmlTopoFile, struct ncclTopoSystem* system);
|
||||
|
||||
ncclResult_t ncclTopoGetSystemFromXml(struct ncclXml* xml, struct ncclTopoSystem** topoSystem);
|
||||
ncclResult_t ncclTopoGetGraphFromXml(struct ncclXmlNode *xmlGraphs, struct ncclTopoSystem* system, struct ncclTopoGraph* graph);
|
||||
ncclResult_t ncclTopoGetXmlFromGraphs(int ngraphs, struct ncclTopoGraph** graphs, struct ncclTopoSystem* system, struct ncclXml *xml);
|
||||
|
||||
static ncclResult_t ncclTopoIdToIndex(struct ncclTopoSystem* system, int type, int64_t id, int* index) {
|
||||
*index = -1;
|
||||
for (int i=0; i<system->nodes[type].count; i++) {
|
||||
if (system->nodes[type].nodes[i].id == id) {
|
||||
*node = system->nodes[type].nodes+i;
|
||||
*index = i;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
if (system->nodes[type].count == NCCL_TOPO_MAX_NODES) {
|
||||
WARN("Error : tried to create too many nodes of type %d\n", type);
|
||||
return ncclInternalError;
|
||||
}
|
||||
struct ncclTopoNode* n = system->nodes[type].nodes+system->nodes[type].count;
|
||||
system->nodes[type].count++;
|
||||
n->type = type;
|
||||
n->id = id;
|
||||
if (type == GPU) {
|
||||
// Create link to itself (used in some corner cases)
|
||||
n->nlinks=1;
|
||||
n->links[0].type = LINK_LOC;
|
||||
n->links[0].remNode = n;
|
||||
n->links[0].width = LOC_WIDTH;
|
||||
}
|
||||
*node = n;
|
||||
return ncclSuccess;
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
static ncclResult_t ncclTopoConnectNodes(struct ncclTopoNode* node, struct ncclTopoNode* remNode, int type, int width) {
|
||||
// Aggregate links into higher width for NVLink
|
||||
struct ncclTopoLink* link;
|
||||
for (link = node->links; link->remNode; link++) {
|
||||
if (link->remNode == remNode && link->type == type) break;
|
||||
}
|
||||
if (link->remNode == NULL) node->nlinks++;
|
||||
link->type = type;
|
||||
link->remNode = remNode;
|
||||
link->width += width;
|
||||
|
||||
// Sort links in BW descending order
|
||||
struct ncclTopoLink linkSave;
|
||||
memcpy(&linkSave, link, sizeof(struct ncclTopoLink));
|
||||
while (link != node->links) {
|
||||
if ((link-1)->width >= linkSave.width) break;
|
||||
memcpy(link, link-1, sizeof(struct ncclTopoLink));
|
||||
link--;
|
||||
}
|
||||
memcpy(link, &linkSave, sizeof(struct ncclTopoLink));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoPrintPaths(struct ncclTopoSystem* system);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2016-2019, NVIDIA CORPORATION. All rights reserved.
|
||||
* Copyright (c) 2016-2020, NVIDIA CORPORATION. All rights reserved.
|
||||
* Modifications Copyright (c) 2019-2020 Advanced Micro Devices, Inc. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
@@ -53,12 +53,12 @@ ncclResult_t parseList(const char* str, const char* elems[], int nelems, int* li
|
||||
}
|
||||
|
||||
static const char* ncclFuncStr[] = { "Broadcast", "Reduce", "AllGather", "ReduceScatter", "AllReduce" };
|
||||
static const char* ncclAlgoStr[] = { "Tree", "Ring" };
|
||||
static const char* ncclAlgoStr[] = { "Tree", "Ring", "CollNet" };
|
||||
static const char* ncclProtoStr[] = { "LL", "LL128", "Simple" };
|
||||
|
||||
// Latencies in us, Bandwidths in GB/s
|
||||
// Tree { LL, LL128, Simple } , Ring { LL, LL128, Simple }
|
||||
static const float baseLat [NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] = { { 37.9, 37.9, 40.4 }, { 20.5, 20.5, 27.9 } };
|
||||
static const float baseLat [NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] = { { 37.9, 37.9, 40.4 }, { 20.5, 20.5, 27.9 }, { 37.9, 37.9, 40.4 } };
|
||||
|
||||
// NVLink, PCI, Network
|
||||
#define NCCL_HW_NVLINK 0
|
||||
@@ -67,29 +67,32 @@ static const float baseLat [NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] = { { 37.9
|
||||
// Tree/Simple is the latency a 256kB chunk, which is ~ base lat + 256k/12GB/s (+ 256k/12GB/s for the network).
|
||||
static const float hwLat [3][NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] =
|
||||
{ /* NVLINK */
|
||||
{ /* Tree (LL/LL128/Simple)*/ { 1.2, 1.2, 3.8 }, /* Ring (LL/LL128/Simple)*/ { 2.3, 2.3, 2.7 } },
|
||||
{ /* Tree (LL/LL128/Simple)*/ { 1.2, 1.2, 3.8 }, /* Ring (LL/LL128/Simple)*/ { 2.3, 2.3, 2.7 }, /* CollNet (LL/LL128/Simple)*/ { 1.2, 1.2, 3.8 } },
|
||||
/* PCI */
|
||||
{ /* Tree (LL/LL128/Simple)*/ { 2.2, 2.2, 5.7 }, /* Ring (LL/LL128/Simple)*/ { 1.3, 1.3, 1.9 } },
|
||||
{ /* Tree (LL/LL128/Simple)*/ { 2.2, 2.2, 5.7 }, /* Ring (LL/LL128/Simple)*/ { 1.3, 1.3, 1.9 }, /* CollNet (LL/LL128/Simple)*/ { 2.2, 2.2, 5.7 } },
|
||||
/* NET */
|
||||
{ /* Tree (LL/LL128/Simple)*/ { 9.8, 9.8, 19.5 }, /* Ring (LL/LL128/Simple)*/ { 2.0, 2.0, 4.5 } }
|
||||
{ /* Tree (LL/LL128/Simple)*/ { 9.8, 9.8, 19.5 }, /* Ring (LL/LL128/Simple)*/ { 2.0, 2.0, 4.5 }, /* CollNet (LL/LL128/Simple)*/ { 9.8, 9.8, 19.5 } }
|
||||
};
|
||||
|
||||
// LL128 max BW for the different collectives
|
||||
static const double ll128MaxBw[NCCL_NUM_FUNCTIONS] = { 113.0, 72.0, 110.0, 91.0, 100.0 };
|
||||
|
||||
ncclResult_t ncclSetThresholds(struct ncclComm* comm, int minCompCap, int maxCompCap, struct ncclTopoGraph* treeGraph, struct ncclTopoGraph* ringGraph) {
|
||||
int simpleDefaultThreads = (treeGraph->speedIntra*treeGraph->nChannels <= 12) ? 256 : NCCL_MAX_NTHREADS;
|
||||
comm->maxThreads[NCCL_PROTO_SIMPLE] = getNthreads("NCCL_NTHREADS", ncclParamNthreads(), 4*WARP_SIZE, NCCL_MAX_NTHREADS, simpleDefaultThreads);
|
||||
comm->maxThreads[NCCL_PROTO_LL] = getNthreads("NCCL_NTHREADS", ncclParamNthreads(), 4*WARP_SIZE, NCCL_MAX_NTHREADS, NCCL_MAX_NTHREADS);
|
||||
comm->maxThreads[NCCL_PROTO_LL128] = getNthreads("NCCL_LL128_NTHREADS", ncclParamLl128Nthreads(), NCCL_LL128_MAX_NTHREADS/4, NCCL_LL128_MAX_NTHREADS, NCCL_LL128_MAX_NTHREADS);
|
||||
|
||||
INFO(NCCL_INIT, "Threads per block : %d/%d/%d", comm->maxThreads[NCCL_PROTO_LL], comm->maxThreads[NCCL_PROTO_LL128], comm->maxThreads[NCCL_PROTO_SIMPLE]);
|
||||
ncclResult_t ncclTopoSetThresholds(struct ncclComm* comm, int minCompCap, int maxCompCap, struct ncclTopoGraph* treeGraph, struct ncclTopoGraph* ringGraph, struct ncclTopoGraph* collNetGraph) {
|
||||
int simpleDefaultThreads = (ringGraph->speedIntra*ringGraph->nChannels <= PCI_WIDTH) ? 256 : NCCL_MAX_NTHREADS;
|
||||
comm->maxThreads[NCCL_ALGO_RING][NCCL_PROTO_SIMPLE] =
|
||||
getNthreads("NCCL_NTHREADS", ncclParamNthreads(), 4*WARP_SIZE, NCCL_MAX_NTHREADS, simpleDefaultThreads);
|
||||
comm->maxThreads[NCCL_ALGO_TREE][NCCL_PROTO_SIMPLE] = comm->maxThreads[NCCL_ALGO_COLLNET][NCCL_PROTO_SIMPLE] =
|
||||
getNthreads("NCCL_NTHREADS", ncclParamNthreads(), 4*WARP_SIZE, NCCL_MAX_NTHREADS, NCCL_MAX_NTHREADS);
|
||||
comm->maxThreads[NCCL_ALGO_RING][NCCL_PROTO_LL] = comm->maxThreads[NCCL_ALGO_TREE][NCCL_PROTO_LL] = comm->maxThreads[NCCL_ALGO_COLLNET][NCCL_PROTO_LL] =
|
||||
getNthreads("NCCL_NTHREADS", ncclParamNthreads(), 4*WARP_SIZE, NCCL_MAX_NTHREADS, NCCL_MAX_NTHREADS);
|
||||
comm->maxThreads[NCCL_ALGO_RING][NCCL_PROTO_LL128] = comm->maxThreads[NCCL_ALGO_TREE][NCCL_PROTO_LL128] = comm->maxThreads[NCCL_ALGO_COLLNET][NCCL_PROTO_LL128] =
|
||||
getNthreads("NCCL_LL128_NTHREADS", ncclParamLl128Nthreads(), NCCL_LL128_MAX_NTHREADS/4, NCCL_LL128_MAX_NTHREADS, NCCL_LL128_MAX_NTHREADS);
|
||||
|
||||
if (comm->nRanks <= 1) return ncclSuccess;
|
||||
|
||||
struct ncclTopoGraph* graphs[2] = { treeGraph, ringGraph };
|
||||
int intraHw[2], hw[2];
|
||||
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) intraHw[a] = graphs[a]->nvlink ? NCCL_HW_NVLINK : NCCL_HW_PCI;
|
||||
struct ncclTopoGraph* graphs[NCCL_NUM_ALGORITHMS] = { treeGraph, ringGraph, collNetGraph };
|
||||
int intraHw[NCCL_NUM_ALGORITHMS], hw[NCCL_NUM_ALGORITHMS];
|
||||
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) intraHw[a] = graphs[a]->typeIntra == LINK_NVL ? NCCL_HW_NVLINK : NCCL_HW_PCI;
|
||||
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) hw[a] = comm->nNodes == 1 ? intraHw[a] : NCCL_HW_NET;
|
||||
|
||||
for (int coll=0; coll<NCCL_NUM_FUNCTIONS; coll++) {
|
||||
@@ -98,11 +101,11 @@ ncclResult_t ncclSetThresholds(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
comm->nRanks;
|
||||
|
||||
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) {
|
||||
if (coll != ncclCollAllReduce && a == NCCL_ALGO_TREE) continue;
|
||||
if (coll != ncclCollAllReduce && a != NCCL_ALGO_RING) continue;
|
||||
|
||||
for (int p=0; p<NCCL_NUM_PROTOCOLS; p++) {
|
||||
int speed = comm->nNodes <= 2 ? graphs[a]->speedIntra : graphs[a]->speedInter;
|
||||
float busBw = graphs[a]->nChannels * speed * 0.6;
|
||||
float speed = comm->nNodes <= 2 || a == NCCL_ALGO_COLLNET ? graphs[a]->speedIntra : graphs[a]->speedInter;
|
||||
float busBw = graphs[a]->nChannels * speed;
|
||||
|
||||
// Various model refinements
|
||||
if (a == NCCL_ALGO_RING && p == NCCL_PROTO_LL) busBw *= 1.0/5.0;
|
||||
@@ -110,9 +113,12 @@ ncclResult_t ncclSetThresholds(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
if (a == NCCL_ALGO_TREE) busBw = std::min(busBw*.27, comm->nNodes > 1 ? 70.0 : 90.0);
|
||||
if (a == NCCL_ALGO_TREE && p == NCCL_PROTO_LL) busBw *= 1.0/2.3;
|
||||
if (a == NCCL_ALGO_TREE && p == NCCL_PROTO_LL128) busBw *= 7.0/9.0;
|
||||
if (a == NCCL_ALGO_COLLNET) busBw *= .9;
|
||||
if (a == NCCL_ALGO_COLLNET && p == NCCL_PROTO_LL) busBw *= 1.0/6.0; // Take into account that GDR read is disabled on both sides
|
||||
if (a == NCCL_ALGO_COLLNET && p == NCCL_PROTO_LL128) busBw = 0; // CollNet does not support LL128
|
||||
|
||||
// Convert bus BW to algorithm BW
|
||||
float ratio = a == NCCL_ALGO_TREE ? .5 : (1.0 * comm->nRanks) / nsteps;
|
||||
float ratio = (a != NCCL_ALGO_RING) ? .5 : (1.0 * comm->nRanks) / nsteps;
|
||||
comm->bandwidths[coll][a][p] = busBw * ratio;
|
||||
|
||||
comm->latencies[coll][a][p] = baseLat[a][p];
|
||||
@@ -128,11 +134,16 @@ ncclResult_t ncclSetThresholds(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
} else {
|
||||
comm->latencies[coll][a][p] += nsteps*lat;
|
||||
}
|
||||
} else {
|
||||
} else if (a == NCCL_ALGO_TREE) {
|
||||
float intraLat = hwLat[intraHw[a]][a][p];
|
||||
float interLat = hwLat[NCCL_HW_NET][a][p];
|
||||
comm->latencies[coll][a][p] +=
|
||||
2 * ((comm->nRanks/comm->nNodes-1) * intraLat + log2i(comm->nNodes) * interLat);
|
||||
} else {
|
||||
float intraLat = hwLat[intraHw[a]][a][p];
|
||||
float interLat = hwLat[NCCL_HW_NET][a][p];
|
||||
comm->latencies[coll][a][p] +=
|
||||
2 * (comm->nRanks/comm->nNodes-1) * intraLat + interLat;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -141,7 +152,7 @@ ncclResult_t ncclSetThresholds(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
// Protocols/Algorithms enable/disable, and user overrides.
|
||||
// All are enabled except ll128 which is enabled by default only in certain cases.
|
||||
int protoEnable[NCCL_NUM_PROTOCOLS] = { 1, 2, 1 };
|
||||
int algoEnable[NCCL_NUM_ALGORITHMS] = { 1, 1 };
|
||||
int algoEnable[NCCL_NUM_ALGORITHMS] = { 1, 1, 1 };
|
||||
|
||||
const char *protoStr = getenv("NCCL_PROTO");
|
||||
if (protoStr) NCCLCHECK(parseList(protoStr, ncclProtoStr, NCCL_NUM_PROTOCOLS, protoEnable));
|
||||
@@ -152,30 +163,32 @@ ncclResult_t ncclSetThresholds(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
int pEnable = protoEnable[p];
|
||||
if (pEnable == 2 && p == NCCL_PROTO_LL128) {
|
||||
// Enable LL128 by default only on Volta+NVLink. Other cases are not tested and may cause silent data corruption.
|
||||
pEnable = (graphs[a]->type <= LINK_PCI) && graphs[a]->nvlink && minCompCap == 70 && maxCompCap == 70 ? 1 : 0;
|
||||
pEnable = (graphs[a]->typeInter <= LINK_PCI) && graphs[a]->typeIntra == LINK_NVL && minCompCap == 70 && maxCompCap == 70 ? 1 : 0;
|
||||
}
|
||||
if (pEnable == 0 || algoEnable[a] == 0) comm->bandwidths[c][a][p] = 0;
|
||||
}
|
||||
|
||||
if (comm->rank == 0) {
|
||||
char line[1024];
|
||||
int offset = 0;
|
||||
sprintf(line, "Latency/AlgBw |");
|
||||
offset = strlen(line);
|
||||
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) {
|
||||
for (int p=0; p<NCCL_NUM_PROTOCOLS; p++) {
|
||||
sprintf(line+offset, " %4s/%6s |", ncclAlgoStr[a], ncclProtoStr[p]);
|
||||
offset = strlen(line);
|
||||
sprintf(line+strlen(line), " %7s/%6s |", ncclAlgoStr[a], ncclProtoStr[p]);
|
||||
}
|
||||
}
|
||||
INFO(NCCL_TUNING, "%s", line);
|
||||
sprintf(line, " Max NThreads |");
|
||||
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) {
|
||||
for (int p=0; p<NCCL_NUM_PROTOCOLS; p++) {
|
||||
sprintf(line+strlen(line), " %14d |", comm->maxThreads[a][p]);
|
||||
}
|
||||
}
|
||||
INFO(NCCL_TUNING, "%s", line);
|
||||
for (int c=0; c<NCCL_NUM_FUNCTIONS; c++) {
|
||||
sprintf(line, "%13s |", ncclFuncStr[c]);
|
||||
offset = strlen(line);
|
||||
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) {
|
||||
for (int p=0; p<NCCL_NUM_PROTOCOLS; p++) {
|
||||
sprintf(line+offset, "%7.1f/%5.1f|", comm->latencies[c][a][p], comm->bandwidths[c][a][p]);
|
||||
offset = strlen(line);
|
||||
sprintf(line+strlen(line), "%8.1f/%6.1f |", comm->latencies[c][a][p], comm->bandwidths[c][a][p]);
|
||||
}
|
||||
}
|
||||
INFO(NCCL_TUNING, "%s", line);
|
||||
@@ -202,12 +215,41 @@ ncclResult_t ncclSetThresholds(struct ncclComm* comm, int minCompCap, int maxCom
|
||||
}
|
||||
}
|
||||
|
||||
INFO(NCCL_INIT, "threadThresholds %ld/%ld/%ld | %ld/%ld/%ld",
|
||||
INFO(NCCL_INIT, "threadThresholds %ld/%ld/%ld | %ld/%ld/%ld | %ld/%ld/%ld",
|
||||
comm->threadThresholds[NCCL_ALGO_TREE][NCCL_PROTO_LL],
|
||||
comm->threadThresholds[NCCL_ALGO_TREE][NCCL_PROTO_LL128],
|
||||
comm->threadThresholds[NCCL_ALGO_TREE][NCCL_PROTO_SIMPLE],
|
||||
comm->threadThresholds[NCCL_ALGO_RING][NCCL_PROTO_LL],
|
||||
comm->threadThresholds[NCCL_ALGO_RING][NCCL_PROTO_LL128],
|
||||
comm->threadThresholds[NCCL_ALGO_RING][NCCL_PROTO_SIMPLE]);
|
||||
comm->threadThresholds[NCCL_ALGO_RING][NCCL_PROTO_SIMPLE],
|
||||
comm->threadThresholds[NCCL_ALGO_COLLNET][NCCL_PROTO_LL],
|
||||
comm->threadThresholds[NCCL_ALGO_COLLNET][NCCL_PROTO_LL128],
|
||||
comm->threadThresholds[NCCL_ALGO_COLLNET][NCCL_PROTO_SIMPLE]);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Trees are not perfectly sticking to the model for medium sizes. Applying a static correction
|
||||
// factor is not ideal but works quite well. Powers of two, 64 B to 1 GB.
|
||||
static float treeCorrectionFactor[NCCL_NUM_PROTOCOLS][22] = {
|
||||
{ 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, .84, .49, .42, .60, .75, .87, .94, .94, .99, 1.0, 1.0 , 1.0 , 1.0 , 1.0 , 1.0 },
|
||||
{ 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, .84, .49, .42, .60, .75, .87, .94, .94, .99, 1.0, 1.0 , 1.0 , 1.0 , 1.0 , 1.0 },
|
||||
{ 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, .41, .27, .25, .39, .46, .72, .76, .87, .92, .97, 1.0, 1.0 , 1.0 , 1.0 , 1.0 , 1.0 }
|
||||
};
|
||||
|
||||
static float ringCorrectionFactor[NCCL_NUM_PROTOCOLS][22] = {
|
||||
{ 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, .25, .41, .55, .56, .78, .94, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0 , 1.0 , 1.0 , 1.0 , 1.0 },
|
||||
{ 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, .25, .41, .55, .56, .78, .94, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0 , 1.0 , 1.0 , 1.0 , 1.0 },
|
||||
{ 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, .04, .08, .09, .09, .11, .13, .25, .40, .59, .76, .86, 1.0 , 1.0 , 1.0 , 1.0 , 1.0 }
|
||||
};
|
||||
|
||||
ncclResult_t ncclTopoGetAlgoTime(struct ncclInfo* info, int algorithm, int protocol, float* time) {
|
||||
float bw = info->comm->bandwidths[info->coll][algorithm][protocol];
|
||||
if (bw == 0) {
|
||||
*time = -1.0; return ncclSuccess;
|
||||
}
|
||||
int logSize = log2i(info->nBytes>>6);
|
||||
if (algorithm == NCCL_ALGO_TREE && logSize < 22) bw *= treeCorrectionFactor[protocol][logSize];
|
||||
else if (algorithm == NCCL_ALGO_RING && logSize < 22) bw *= ringCorrectionFactor[protocol][logSize];
|
||||
*time = info->comm->latencies[info->coll][algorithm][protocol] + (info->nBytes) / (1000 * bw);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,819 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2019-2020, NVIDIA CORPORATION. All rights reserved.
|
||||
* Modifications Copyright (c) 2019-2020 Advanced Micro Devices, Inc. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include <sys/types.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#include <fcntl.h>
|
||||
#include <ctype.h>
|
||||
#include "core.h"
|
||||
#include "nvmlwrap.h"
|
||||
#include "xml.h"
|
||||
|
||||
/*******************/
|
||||
/* XML File Parser */
|
||||
/*******************/
|
||||
|
||||
ncclResult_t xmlGetChar(FILE* file, char* c) {
|
||||
if (fread(c, 1, 1, file) == 0) {
|
||||
WARN("XML Parse : Unexpected EOF");
|
||||
return ncclInternalError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t xmlGetValue(FILE* file, char* value, char* last) {
|
||||
char c;
|
||||
NCCLCHECK(xmlGetChar(file, &c));
|
||||
if (c != '"' && c != '\'') {
|
||||
#if INT_OK
|
||||
int o = 0;
|
||||
do {
|
||||
value[o++] = c;
|
||||
NCCLCHECK(xmlGetChar(file, &c));
|
||||
} while (c >= '0' && c <= '9');
|
||||
value[o] = '\0';
|
||||
*last = c;
|
||||
return ncclSuccess;
|
||||
#else
|
||||
WARN("XML Parse : Expected (double) quote.");
|
||||
return ncclInternalError;
|
||||
#endif
|
||||
}
|
||||
int o = 0;
|
||||
do {
|
||||
NCCLCHECK(xmlGetChar(file, &c));
|
||||
value[o++] = c;
|
||||
} while (c != '"');
|
||||
value[o-1] = '\0';
|
||||
NCCLCHECK(xmlGetChar(file, last));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t xmlGetToken(FILE* file, char* name, char* value, char* last) {
|
||||
char c;
|
||||
char* ptr = name;
|
||||
int o = 0;
|
||||
do {
|
||||
NCCLCHECK(xmlGetChar(file, &c));
|
||||
if (c == '=') {
|
||||
ptr[o] = '\0';
|
||||
if (value == NULL) {
|
||||
WARN("XML Parse : Unexpected value with name %s\n", ptr);
|
||||
return ncclInternalError;
|
||||
}
|
||||
return xmlGetValue(file, value, last);
|
||||
}
|
||||
ptr[o] = c;
|
||||
if (o == MAX_STR_LEN-1) {
|
||||
ptr[o] = '\0';
|
||||
WARN("Error : name %s too long (max %d)", ptr, MAX_STR_LEN);
|
||||
return ncclInternalError;
|
||||
}
|
||||
o++;
|
||||
} while (c != ' ' && c != '>' && c != '/' && c != '\n' && c != '\r');
|
||||
ptr[o-1] = '\0';
|
||||
*last = c;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Shift the 3-chars string by one char and append c at the end
|
||||
#define SHIFT_APPEND(s, c) do { s[0]=s[1]; s[1]=s[2]; s[2]=c; } while(0)
|
||||
ncclResult_t xmlSkipComment(FILE* file, char* start, char next) {
|
||||
// Start from something neutral with \0 at the end.
|
||||
char end[4] = "...";
|
||||
|
||||
// Inject all trailing chars from previous reads. We don't need
|
||||
// to check for --> here because there cannot be a > in the name.
|
||||
for (int i=0; i<strlen(start); i++) SHIFT_APPEND(end, start[i]);
|
||||
SHIFT_APPEND(end, next);
|
||||
|
||||
// Stop when we find "-->"
|
||||
while (strcmp(end, "-->") != 0) {
|
||||
int c;
|
||||
if (fread(&c, 1, 1, file) != 1) {
|
||||
WARN("XML Parse error : unterminated comment");
|
||||
return ncclInternalError;
|
||||
}
|
||||
SHIFT_APPEND(end, c);
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t xmlGetNode(FILE* file, struct ncclXmlNode* node) {
|
||||
node->type = NODE_TYPE_NONE;
|
||||
char c = ' ';
|
||||
while (c == ' ' || c == '\n' || c == '\r') {
|
||||
if (fread(&c, 1, 1, file) == 0) return ncclSuccess;
|
||||
}
|
||||
if (c != '<') {
|
||||
WARN("XML Parse error : expecting '<', got '%c'", c);
|
||||
return ncclInternalError;
|
||||
}
|
||||
// Read XML element name
|
||||
NCCLCHECK(xmlGetToken(file, node->name, NULL, &c));
|
||||
|
||||
// Check for comments
|
||||
if (strncmp(node->name, "!--", 3) == 0) {
|
||||
NCCLCHECK(xmlSkipComment(file, node->name+3, c));
|
||||
return xmlGetNode(file, node);
|
||||
}
|
||||
|
||||
// Check for closing tag
|
||||
if (node->name[0] == '\0' && c == '/') {
|
||||
node->type = NODE_TYPE_CLOSE;
|
||||
// Re-read the name, we got '/' in the first call
|
||||
NCCLCHECK(xmlGetToken(file, node->name, NULL, &c));
|
||||
if (c != '>') {
|
||||
WARN("XML Parse error : unexpected trailing %c in closing tag %s\n", c, node->name);
|
||||
return ncclInternalError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
node->type = NODE_TYPE_OPEN;
|
||||
|
||||
// Get Attributes
|
||||
int a = 0;
|
||||
while (c == ' ') {
|
||||
NCCLCHECK(xmlGetToken(file, node->attrs[a].key, node->attrs[a].value, &c));
|
||||
if (a == MAX_ATTR_COUNT) {
|
||||
INFO(NCCL_GRAPH, "XML Parse : Ignoring extra attributes (max %d)\n", MAX_ATTR_COUNT);
|
||||
// Actually we need to still consume the extra attributes so we have an extra one.
|
||||
} else a++;
|
||||
}
|
||||
node->nAttrs = a;
|
||||
if (c == '/') {
|
||||
node->type = NODE_TYPE_SINGLE;
|
||||
char str[MAX_STR_LEN];
|
||||
NCCLCHECK(xmlGetToken(file, str, NULL, &c));
|
||||
}
|
||||
if (c != '>') {
|
||||
WARN("XML Parse : expected >, got '%c'", c);
|
||||
return ncclInternalError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
typedef ncclResult_t (*xmlHandlerFunc_t)(FILE*, struct ncclXml*, struct ncclXmlNode*);
|
||||
|
||||
struct xmlHandler {
|
||||
const char * name;
|
||||
xmlHandlerFunc_t func;
|
||||
};
|
||||
|
||||
ncclResult_t xmlLoadSub(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head, struct xmlHandler handlers[], int nHandlers) {
|
||||
if (head && head->type == NODE_TYPE_SINGLE) return ncclSuccess;
|
||||
while (1) {
|
||||
if (xml->maxIndex == MAX_NODES) {
|
||||
WARN("Error : XML parser is limited to 1024 nodes\n");
|
||||
return ncclInternalError;
|
||||
}
|
||||
struct ncclXmlNode* node = xml->nodes+xml->maxIndex;
|
||||
memset(node, 0, sizeof(struct ncclXmlNode));
|
||||
NCCLCHECK(xmlGetNode(file, node));
|
||||
if (node->type == NODE_TYPE_NONE) {
|
||||
if (head) {
|
||||
WARN("XML Parse : unterminated %s", head->name);
|
||||
return ncclInternalError;
|
||||
} else {
|
||||
// All done
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
if (head && node->type == NODE_TYPE_CLOSE) {
|
||||
if (strcmp(node->name, head->name) != 0) {
|
||||
WARN("XML Mismatch : %s / %s", head->name, node->name);
|
||||
return ncclInternalError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
int found = 0;
|
||||
for (int h=0; h<nHandlers; h++) {
|
||||
if (strcmp(node->name, handlers[h].name) == 0) {
|
||||
if (head) head->subs[head->nSubs++] = node;
|
||||
node->parent = head;
|
||||
node->nSubs = 0;
|
||||
xml->maxIndex++;
|
||||
NCCLCHECK(handlers[h].func(file, xml, node));
|
||||
found = 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!found) {
|
||||
if (nHandlers) INFO(NCCL_GRAPH, "Ignoring element %s", node->name);
|
||||
NCCLCHECK(xmlLoadSub(file, xml, node, NULL, 0));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**************/
|
||||
/* XML Writer */
|
||||
/**************/
|
||||
|
||||
ncclResult_t ncclTopoDumpXmlRec(int indent, FILE* file, struct ncclXmlNode* node) {
|
||||
for (int i=0; i<indent; i++) fprintf(file, " ");
|
||||
fprintf(file, "<%s", node->name);
|
||||
|
||||
for (int a=0; a<node->nAttrs; a++) {
|
||||
fprintf(file, " %s=\"%s\"", node->attrs[a].key, node->attrs[a].value);
|
||||
}
|
||||
if (node->nSubs == 0) {
|
||||
fprintf(file, "/>\n");
|
||||
} else {
|
||||
fprintf(file, ">\n");
|
||||
for (int s=0; s<node->nSubs; s++) {
|
||||
NCCLCHECK(ncclTopoDumpXmlRec(indent+2, file, node->subs[s]));
|
||||
}
|
||||
for (int i=0; i<indent; i++) fprintf(file, " ");
|
||||
fprintf(file, "</%s>\n", node->name);
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoDumpXmlToFile(const char* xmlTopoFile, struct ncclXml* xml) {
|
||||
FILE* file = fopen(xmlTopoFile, "w");
|
||||
if (file == NULL) {
|
||||
WARN("Unable to open %s, not dumping topology.", xmlTopoFile);
|
||||
return ncclSuccess;
|
||||
}
|
||||
NCCLCHECK(ncclTopoDumpXmlRec(0, file, xml->nodes));
|
||||
fclose(file);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
/****************************************/
|
||||
/* Parser rules for our specific format */
|
||||
/****************************************/
|
||||
|
||||
ncclResult_t ncclTopoXmlLoadNvlink(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, NULL, 0));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlLoadGpu(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
#if defined(__HIP_PLATFORM_HCC__) || defined(__HCC__) || defined(__HIPCC__)
|
||||
struct xmlHandler handlers[] = { { "xgmi", ncclTopoXmlLoadNvlink } };
|
||||
#else
|
||||
struct xmlHandler handlers[] = { { "nvlink", ncclTopoXmlLoadNvlink } };
|
||||
#endif
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, handlers, 1));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlLoadNet(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, NULL, 0));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlLoadNic(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
struct xmlHandler handlers[] = { { "net", ncclTopoXmlLoadNet } };
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, handlers, 1));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlLoadPci(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
struct xmlHandler handlers[] = { { "pci", ncclTopoXmlLoadPci }, { "gpu", ncclTopoXmlLoadGpu }, { "nic", ncclTopoXmlLoadNic} };
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, handlers, 3));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlLoadCpu(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
struct xmlHandler handlers[] = { { "pci", ncclTopoXmlLoadPci }, { "nic", ncclTopoXmlLoadNic } };
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, handlers, 2));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlLoadSystem(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
int version;
|
||||
NCCLCHECK(xmlGetAttrInt(head, "version", &version));
|
||||
if (version != NCCL_TOPO_XML_VERSION) {
|
||||
WARN("XML Topology has wrong version %d, %d needed", version, NCCL_TOPO_XML_VERSION);
|
||||
return ncclInvalidUsage;
|
||||
}
|
||||
const char* name;
|
||||
NCCLCHECK(xmlGetAttr(head, "name", &name));
|
||||
if (name != NULL) INFO(NCCL_GRAPH, "Loading topology %s", name);
|
||||
else INFO(NCCL_GRAPH, "Loading unnamed topology");
|
||||
|
||||
struct xmlHandler handlers[] = { { "cpu", ncclTopoXmlLoadCpu } };
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, handlers, 1));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetXmlFromFile(const char* xmlTopoFile, struct ncclXml* xml) {
|
||||
FILE* file = fopen(xmlTopoFile, "r");
|
||||
if (file == NULL) {
|
||||
WARN("Could not open XML topology file %s : %s", xmlTopoFile, strerror(errno));
|
||||
return ncclSuccess;
|
||||
}
|
||||
struct xmlHandler handlers[] = { { "system", ncclTopoXmlLoadSystem } };
|
||||
xml->maxIndex = 0;
|
||||
NCCLCHECK(xmlLoadSub(file, xml, NULL, handlers, 1));
|
||||
fclose(file);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
/**********************/
|
||||
/* XML creation */
|
||||
/* from autodetection */
|
||||
/**********************/
|
||||
|
||||
#define BUSID_SIZE (sizeof("0000:00:00.0"))
|
||||
#define BUSID_REDUCED_SIZE (sizeof("0000:00"))
|
||||
static void memcpylower(char* dst, const char* src, const size_t size) {
|
||||
for (int i=0; i<size; i++) dst[i] = tolower(src[i]);
|
||||
}
|
||||
static ncclResult_t getPciPath(const char* busId, char** path) {
|
||||
char busPath[] = "/sys/class/pci_bus/0000:00/../../0000:00:00.0";
|
||||
memcpylower(busPath+sizeof("/sys/class/pci_bus/")-1, busId, BUSID_REDUCED_SIZE-1);
|
||||
memcpylower(busPath+sizeof("/sys/class/pci_bus/0000:00/../../")-1, busId, BUSID_SIZE-1);
|
||||
*path = realpath(busPath, NULL);
|
||||
if (*path == NULL) {
|
||||
WARN("Could not find real path of %s", busPath);
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetStrFromSys(const char* path, const char* fileName, char* strValue) {
|
||||
char filePath[PATH_MAX];
|
||||
sprintf(filePath, "%s/%s", path, fileName);
|
||||
int offset = 0;
|
||||
FILE* file;
|
||||
if ((file = fopen(filePath, "r")) != NULL) {
|
||||
while (feof(file) == 0 && ferror(file) == 0 && offset < MAX_STR_LEN) {
|
||||
int len = fread(strValue+offset, 1, MAX_STR_LEN-offset, file);
|
||||
offset += len;
|
||||
}
|
||||
fclose(file);
|
||||
}
|
||||
if (offset == 0) {
|
||||
strValue[0] = '\0';
|
||||
INFO(NCCL_GRAPH, "Topology detection : could not read %s, ignoring", filePath);
|
||||
} else {
|
||||
strValue[offset-1] = '\0';
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoSetAttrFromSys(struct ncclXmlNode* pciNode, const char* path, const char* fileName, const char* attrName) {
|
||||
char strValue[MAX_STR_LEN];
|
||||
NCCLCHECK(ncclTopoGetStrFromSys(path, fileName, strValue));
|
||||
if (strValue[0] != '\0') { NCCLCHECK(xmlSetAttr(pciNode, attrName, strValue)); }
|
||||
TRACE(NCCL_GRAPH, "Read from sys %s/%s -> %s=%s\n", path, fileName, attrName, strValue);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetXmlFromCpu(struct ncclXmlNode* cpuNode, struct ncclXml* xml) {
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(cpuNode, "affinity", &index));
|
||||
if (index == -1) {
|
||||
const char* numaId;
|
||||
NCCLCHECK(xmlGetAttr(cpuNode, "numaid", &numaId));
|
||||
if (numaId == NULL) {
|
||||
WARN("GetXmlFromCpu : could not find CPU numa ID.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
// Set affinity
|
||||
char cpumaskPath[] = "/sys/devices/system/node/node0000";
|
||||
sprintf(cpumaskPath, "/sys/devices/system/node/node%s", numaId);
|
||||
NCCLCHECK(ncclTopoSetAttrFromSys(cpuNode, cpumaskPath, "cpumap", "affinity"));
|
||||
}
|
||||
|
||||
NCCLCHECK(xmlGetAttrIndex(cpuNode, "arch", &index));
|
||||
if (index == -1) {
|
||||
// Fill CPU type / vendor / model
|
||||
#if defined(__PPC__)
|
||||
NCCLCHECK(xmlSetAttr(cpuNode, "arch", "ppc64"));
|
||||
#elif defined(__aarch64__)
|
||||
NCCLCHECK(xmlSetAttr(cpuNode, "arch", "arm64"));
|
||||
#elif defined(__x86_64__)
|
||||
NCCLCHECK(xmlSetAttr(cpuNode, "arch", "x86_64"));
|
||||
#endif
|
||||
}
|
||||
|
||||
#if defined(__x86_64__)
|
||||
NCCLCHECK(xmlGetAttrIndex(cpuNode, "vendor", &index));
|
||||
if (index == -1) {
|
||||
union {
|
||||
struct {
|
||||
// CPUID 0 String register order
|
||||
uint32_t ebx;
|
||||
uint32_t edx;
|
||||
uint32_t ecx;
|
||||
};
|
||||
char vendor[12];
|
||||
} cpuid0;
|
||||
|
||||
asm volatile("cpuid" : "=b" (cpuid0.ebx), "=c" (cpuid0.ecx), "=d" (cpuid0.edx) : "a" (0) : "memory");
|
||||
char vendor[13];
|
||||
strncpy(vendor, cpuid0.vendor, 12);
|
||||
vendor[12] = '\0';
|
||||
NCCLCHECK(xmlSetAttr(cpuNode, "vendor", vendor));
|
||||
}
|
||||
|
||||
NCCLCHECK(xmlGetAttrIndex(cpuNode, "familyid", &index));
|
||||
if (index == -1) {
|
||||
union {
|
||||
struct {
|
||||
unsigned steppingId:4;
|
||||
unsigned modelId:4;
|
||||
unsigned familyId:4;
|
||||
unsigned processorType:2;
|
||||
unsigned resv0:2;
|
||||
unsigned extModelId:4;
|
||||
unsigned extFamilyId:8;
|
||||
unsigned resv1:4;
|
||||
};
|
||||
uint32_t val;
|
||||
} cpuid1;
|
||||
asm volatile("cpuid" : "=a" (cpuid1.val) : "a" (1) : "memory");
|
||||
int familyId = cpuid1.familyId + (cpuid1.extFamilyId << 4);
|
||||
int modelId = cpuid1.modelId + (cpuid1.extModelId << 4);
|
||||
NCCLCHECK(xmlSetAttrInt(cpuNode, "familyid", familyId));
|
||||
NCCLCHECK(xmlSetAttrInt(cpuNode, "modelid", modelId));
|
||||
}
|
||||
#endif
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetPciNode(struct ncclXml* xml, const char* busId, struct ncclXmlNode** pciNode) {
|
||||
NCCLCHECK(xmlFindTagKv(xml, "pci", pciNode, "busid", busId));
|
||||
if (*pciNode == NULL) {
|
||||
NCCLCHECK(xmlAddNode(xml, NULL, "pci", pciNode));
|
||||
}
|
||||
NCCLCHECK(xmlSetAttr(*pciNode, "busid", busId));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Check whether a string is in BDF format or not.
|
||||
// BDF (Bus-Device-Function) is "BBBB:BB:DD.F" where B, D and F are hex digits.
|
||||
// There can be trailing chars.
|
||||
int isHex(char c) { return ((c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F')); }
|
||||
int checkBDFFormat(char* bdf) {
|
||||
if (bdf[4] != ':' || bdf[7] != ':' || bdf[10] != '.') return 0;
|
||||
if (isHex(bdf[0]) == 0 || isHex(bdf[1] == 0) || isHex(bdf[2] == 0) || isHex(bdf[3] == 0) ||
|
||||
isHex(bdf[5] == 0) || isHex(bdf[6] == 0) || isHex(bdf[8] == 0) || isHex(bdf[9] == 0) ||
|
||||
isHex(bdf[11] == 0)) return 0;
|
||||
return 1;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetXmlFromSys(struct ncclXmlNode* pciNode, struct ncclXml* xml) {
|
||||
// Fill info, then parent
|
||||
const char* busId;
|
||||
NCCLCHECK(xmlGetAttr(pciNode, "busid", &busId));
|
||||
char* path = NULL;
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(pciNode, "class", &index));
|
||||
if (index == -1) {
|
||||
if (path == NULL) NCCLCHECK(getPciPath(busId, &path));
|
||||
NCCLCHECK(ncclTopoSetAttrFromSys(pciNode, path, "class", "class"));
|
||||
}
|
||||
NCCLCHECK(xmlGetAttrIndex(pciNode, "link_speed", &index));
|
||||
if (index == -1) {
|
||||
if (path == NULL) NCCLCHECK(getPciPath(busId, &path));
|
||||
char deviceSpeedStr[MAX_STR_LEN];
|
||||
float deviceSpeed;
|
||||
NCCLCHECK(ncclTopoGetStrFromSys(path, "max_link_speed", deviceSpeedStr));
|
||||
sscanf(deviceSpeedStr, "%f GT/s", &deviceSpeed);
|
||||
char portSpeedStr[MAX_STR_LEN];
|
||||
float portSpeed;
|
||||
NCCLCHECK(ncclTopoGetStrFromSys(path, "../max_link_speed", portSpeedStr));
|
||||
sscanf(portSpeedStr, "%f GT/s", &portSpeed);
|
||||
NCCLCHECK(xmlSetAttr(pciNode, "link_speed", portSpeed < deviceSpeed ? portSpeedStr : deviceSpeedStr));
|
||||
}
|
||||
NCCLCHECK(xmlGetAttrIndex(pciNode, "link_width", &index));
|
||||
if (index == -1) {
|
||||
if (path == NULL) NCCLCHECK(getPciPath(busId, &path));
|
||||
char strValue[MAX_STR_LEN];
|
||||
NCCLCHECK(ncclTopoGetStrFromSys(path, "max_link_width", strValue));
|
||||
int deviceWidth = strtol(strValue, NULL, 0);
|
||||
NCCLCHECK(ncclTopoGetStrFromSys(path, "../max_link_width", strValue));
|
||||
int portWidth = strtol(strValue, NULL, 0);
|
||||
NCCLCHECK(xmlSetAttrInt(pciNode, "link_width", std::min(deviceWidth,portWidth)));
|
||||
}
|
||||
struct ncclXmlNode* parent = pciNode->parent;
|
||||
if (parent == NULL) {
|
||||
if (path == NULL) NCCLCHECK(getPciPath(busId, &path));
|
||||
|
||||
// Save that for later in case next step is a CPU
|
||||
char numaIdStr[MAX_STR_LEN];
|
||||
NCCLCHECK(ncclTopoGetStrFromSys(path, "numa_node", numaIdStr));
|
||||
|
||||
// Go up one level in the PCI tree. Rewind two "/" and follow the upper PCI
|
||||
// switch, or stop if we reach a CPU root complex.
|
||||
int slashCount = 0;
|
||||
int parentOffset;
|
||||
for (parentOffset = strlen(path)-1; parentOffset>0; parentOffset--) {
|
||||
if (path[parentOffset] == '/') {
|
||||
slashCount++;
|
||||
path[parentOffset] = '\0';
|
||||
int start = parentOffset - 1;
|
||||
while (start>0 && path[start] != '/') start--;
|
||||
// Check whether the parent path looks like "BBBB:BB:DD.F" or not.
|
||||
if (checkBDFFormat(path+start+1) == 0) {
|
||||
// This a CPU root complex. Create a CPU tag and stop there.
|
||||
struct ncclXmlNode* topNode;
|
||||
NCCLCHECK(xmlFindTag(xml, "system", &topNode));
|
||||
NCCLCHECK(xmlGetSubKv(topNode, "cpu", &parent, "numaid", numaIdStr));
|
||||
if (parent == NULL) {
|
||||
NCCLCHECK(xmlAddNode(xml, topNode, "cpu", &parent));
|
||||
NCCLCHECK(xmlSetAttr(parent, "numaid", numaIdStr));
|
||||
}
|
||||
} else if (slashCount == 2) {
|
||||
// Continue on the upper PCI switch
|
||||
for (int i = strlen(path)-1; i>0; i--) {
|
||||
if (path[i] == '/') {
|
||||
NCCLCHECK(xmlFindTagKv(xml, "pci", &parent, "busid", path+i+1));
|
||||
if (parent == NULL) {
|
||||
NCCLCHECK(xmlAddNode(xml, NULL, "pci", &parent));
|
||||
NCCLCHECK(xmlSetAttr(parent, "busid", path+i+1));
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (parent) break;
|
||||
}
|
||||
pciNode->parent = parent;
|
||||
parent->subs[parent->nSubs++] = pciNode;
|
||||
}
|
||||
if (strcmp(parent->name, "pci") == 0) {
|
||||
NCCLCHECK(ncclTopoGetXmlFromSys(parent, xml));
|
||||
} else if (strcmp(parent->name, "cpu") == 0) {
|
||||
NCCLCHECK(ncclTopoGetXmlFromCpu(parent, xml));
|
||||
}
|
||||
free(path);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetXmlFromGpu(struct ncclXmlNode* pciNode, nvmlDevice_t nvmlDev, struct ncclXml* xml, struct ncclXmlNode** gpuNodeRet) {
|
||||
struct ncclXmlNode* gpuNode = NULL;
|
||||
NCCLCHECK(xmlGetSub(pciNode, "gpu", &gpuNode));
|
||||
if (gpuNode == NULL) NCCLCHECK(xmlAddNode(xml, pciNode, "gpu", &gpuNode));
|
||||
|
||||
int index = -1;
|
||||
|
||||
int dev = -1;
|
||||
NCCLCHECK(xmlGetAttrIndex(gpuNode, "dev", &index));
|
||||
if (index == -1) {
|
||||
if (nvmlDev == NULL) {
|
||||
//WARN("No NVML, trying to use CUDA instead");
|
||||
const char* busId;
|
||||
NCCLCHECK(xmlGetAttr(pciNode, "busid", &busId));
|
||||
if (busId == NULL || hipDeviceGetByPCIBusId(&dev, busId) != hipSuccess) dev = -1;
|
||||
} else {
|
||||
NCCLCHECK(wrapNvmlDeviceGetIndex(nvmlDev, (unsigned int*)&dev));
|
||||
}
|
||||
NCCLCHECK(xmlSetAttrInt(gpuNode, "dev", dev));
|
||||
}
|
||||
NCCLCHECK(xmlGetAttrInt(gpuNode, "dev", &dev));
|
||||
if (dev == -1) return ncclSuccess;
|
||||
|
||||
NCCLCHECK(xmlGetAttrIndex(gpuNode, "sm", &index));
|
||||
if (index == -1) {
|
||||
int cudaMajor, cudaMinor;
|
||||
if (nvmlDev == NULL) {
|
||||
hipDeviceProp_t devProp;
|
||||
CUDACHECK(hipGetDeviceProperties(&devProp, dev));
|
||||
cudaMajor = devProp.major; cudaMinor = devProp.minor;
|
||||
} else {
|
||||
NCCLCHECK(wrapNvmlDeviceGetCudaComputeCapability(nvmlDev, &cudaMajor, &cudaMinor));
|
||||
}
|
||||
NCCLCHECK(xmlSetAttrInt(gpuNode, "sm", cudaMajor*10+cudaMinor));
|
||||
}
|
||||
int sm;
|
||||
NCCLCHECK(xmlGetAttrInt(gpuNode, "sm", &sm));
|
||||
|
||||
struct ncclXmlNode* nvlNode = NULL;
|
||||
NCCLCHECK(xmlGetSub(pciNode, "nvlink", &nvlNode));
|
||||
if (nvlNode == NULL) {
|
||||
#if defined(__HIP_PLATFORM_HCC__) || defined(__HCC__) || defined(__HIPCC__)
|
||||
const char* busId;
|
||||
NCCLCHECK(xmlGetAttr(pciNode, "busid", &busId));
|
||||
if (busId == NULL || hipDeviceGetByPCIBusId(&dev, busId) != hipSuccess) return ncclInternalError;
|
||||
int deviceCnt;
|
||||
CUDACHECK(hipGetDeviceCount(&deviceCnt));
|
||||
for (int i=0; i<deviceCnt; i++) {
|
||||
if (i != dev) {
|
||||
uint32_t link_type, hops;
|
||||
if (hipExtGetLinkTypeAndHopCount(dev, i, &link_type, &hops) == hipSuccess) {
|
||||
if (link_type == HSA_AMD_LINK_INFO_TYPE_XGMI && hops == 1) {
|
||||
char busIdStr[] = "00000000:00:00.0";
|
||||
CUDACHECK(hipDeviceGetPCIBusId(busIdStr, sizeof(busIdStr), i));
|
||||
char lowerId[NVML_DEVICE_PCI_BUS_ID_BUFFER_SIZE];
|
||||
for (int c=0; c<NVML_DEVICE_PCI_BUS_ID_BUFFER_SIZE; c++) {
|
||||
lowerId[c] = tolower(busIdStr[c]);
|
||||
if (busIdStr[c] == 0) break;
|
||||
}
|
||||
NCCLCHECK(xmlGetSubKv(gpuNode, "xgmi", &nvlNode, "target", lowerId));
|
||||
if (nvlNode == NULL) {
|
||||
NCCLCHECK(xmlAddNode(xml, gpuNode, "xgmi", &nvlNode));
|
||||
NCCLCHECK(xmlSetAttr(nvlNode, "target", lowerId));
|
||||
NCCLCHECK(xmlSetAttrInt(nvlNode, "count", 1));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
// NVML NVLink detection
|
||||
int maxNvLinks = (sm < 60) ? 0 : (sm < 70) ? 4 : 6;
|
||||
|
||||
if (maxNvLinks > 0 && nvmlDev == NULL) {
|
||||
WARN("No NVML device handle. Skipping nvlink detection.\n");
|
||||
maxNvLinks = 0;
|
||||
}
|
||||
|
||||
for (int l=0; l<maxNvLinks; ++l) {
|
||||
// Check whether we can use this NVLink for P2P
|
||||
unsigned canP2P;
|
||||
if ((wrapNvmlDeviceGetNvLinkCapability(nvmlDev, l, NVML_NVLINK_CAP_P2P_SUPPORTED, &canP2P) != ncclSuccess) || !canP2P) continue;
|
||||
|
||||
// Make sure the Nvlink is up. The previous call should have trained the link.
|
||||
nvmlEnableState_t isActive;
|
||||
if ((wrapNvmlDeviceGetNvLinkState(nvmlDev, l, &isActive) != ncclSuccess) || (isActive != NVML_FEATURE_ENABLED)) continue;
|
||||
|
||||
// Try to figure out what's on the other side of the NVLink
|
||||
nvmlPciInfo_t remoteProc;
|
||||
if (wrapNvmlDeviceGetNvLinkRemotePciInfo(nvmlDev, l, &remoteProc) != ncclSuccess) continue;
|
||||
|
||||
// Make a lower case copy of the bus ID for calling ncclDeviceType
|
||||
// PCI system path is in lower case
|
||||
char* p = remoteProc.busId;
|
||||
char lowerId[NVML_DEVICE_PCI_BUS_ID_BUFFER_SIZE];
|
||||
for (int c=0; c<NVML_DEVICE_PCI_BUS_ID_BUFFER_SIZE; c++) {
|
||||
lowerId[c] = tolower(p[c]);
|
||||
if (p[c] == 0) break;
|
||||
}
|
||||
|
||||
NCCLCHECK(xmlGetSubKv(gpuNode, "nvlink", &nvlNode, "target", lowerId));
|
||||
if (nvlNode == NULL) {
|
||||
NCCLCHECK(xmlAddNode(xml, gpuNode, "nvlink", &nvlNode));
|
||||
NCCLCHECK(xmlSetAttr(nvlNode, "target", lowerId));
|
||||
NCCLCHECK(xmlSetAttrInt(nvlNode, "count", 1));
|
||||
} else {
|
||||
int count;
|
||||
NCCLCHECK(xmlGetAttrInt(nvlNode, "count", &count));
|
||||
NCCLCHECK(xmlSetAttrInt(nvlNode, "count", count+1));
|
||||
}
|
||||
}
|
||||
#endif
|
||||
}
|
||||
// Fill target classes
|
||||
for (int s=0; s<gpuNode->nSubs; s++) {
|
||||
struct ncclXmlNode* sub = gpuNode->subs[s];
|
||||
#if defined(__HIP_PLATFORM_HCC__) || defined(__HCC__) || defined(__HIPCC__)
|
||||
if (strcmp(sub->name, "xgmi") != 0) continue;
|
||||
#else
|
||||
if (strcmp(sub->name, "nvlink") != 0) continue;
|
||||
#endif
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(sub, "tclass", &index));
|
||||
if (index == -1) {
|
||||
const char* busId;
|
||||
NCCLCHECK(xmlGetAttr(sub, "target", &busId));
|
||||
char* path;
|
||||
NCCLCHECK(getPciPath(busId, &path));
|
||||
NCCLCHECK(ncclTopoSetAttrFromSys(sub, path, "class", "tclass"));
|
||||
}
|
||||
}
|
||||
*gpuNodeRet = gpuNode;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoFillGpu(struct ncclXml* xml, const char* busId, struct ncclXmlNode** gpuNode) {
|
||||
struct ncclXmlNode* node;
|
||||
NCCLCHECK(ncclTopoGetPciNode(xml, busId, &node));
|
||||
NCCLCHECK(ncclTopoGetXmlFromSys(node, xml));
|
||||
NCCLCHECK(wrapNvmlSymbols());
|
||||
NCCLCHECK(wrapNvmlInit());
|
||||
nvmlDevice_t nvmlDev;
|
||||
if (wrapNvmlDeviceGetHandleByPciBusId(busId, &nvmlDev) != ncclSuccess) nvmlDev = NULL;
|
||||
NCCLCHECK(ncclTopoGetXmlFromGpu(node, nvmlDev, xml, gpuNode));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Returns the subsystem name of a path, i.e. the end of the path
|
||||
// where sysPath/subsystem points to.
|
||||
ncclResult_t ncclTopoGetSubsystem(const char* sysPath, char* subSys) {
|
||||
char subSysPath[PATH_MAX];
|
||||
sprintf(subSysPath, "%s/subsystem", sysPath);
|
||||
char* path = realpath(subSysPath, NULL);
|
||||
if (path == NULL) {
|
||||
subSys[0] = '\0';
|
||||
} else {
|
||||
int offset;
|
||||
for (offset = strlen(path); offset > 0 && path[offset] != '/'; offset--);
|
||||
strcpy(subSys, path+offset+1);
|
||||
free(path);
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoFillNet(struct ncclXml* xml, const char* pciPath, const char* netName, struct ncclXmlNode** netNode) {
|
||||
NCCLCHECK(xmlFindTagKv(xml, "net", netNode, "name", netName));
|
||||
if (*netNode != NULL) return ncclSuccess;
|
||||
|
||||
const char* pciSysPath = pciPath;
|
||||
if (pciSysPath) {
|
||||
char subSystem[PATH_MAX];
|
||||
NCCLCHECK(ncclTopoGetSubsystem(pciSysPath, subSystem));
|
||||
// This is not a PCI device (virtual, usb, ...).
|
||||
if (strcmp(subSystem, "pci") != 0) {
|
||||
INFO(NCCL_GRAPH, "Topology detection: network path %s is not a PCI device (%s). Attaching to first CPU", pciSysPath, subSystem);
|
||||
pciSysPath = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
struct ncclXmlNode* parent = NULL;
|
||||
if (pciSysPath) {
|
||||
int offset;
|
||||
for (offset=strlen(pciSysPath)-1; pciSysPath[offset] != '/'; offset--);
|
||||
char busId[NVML_DEVICE_PCI_BUS_ID_BUFFER_SIZE];
|
||||
strcpy(busId, pciSysPath+offset+1);
|
||||
NCCLCHECK(xmlFindTagKv(xml, "pci", &parent, "busid", busId));
|
||||
if (parent == NULL) {
|
||||
NCCLCHECK(xmlAddNode(xml, NULL, "pci", &parent));
|
||||
NCCLCHECK(xmlSetAttr(parent, "busid", busId));
|
||||
NCCLCHECK(ncclTopoGetXmlFromSys(parent, xml));
|
||||
}
|
||||
} else {
|
||||
// Virtual NIC, no PCI device, attach to first CPU
|
||||
NCCLCHECK(xmlFindTag(xml, "cpu", &parent));
|
||||
}
|
||||
|
||||
struct ncclXmlNode* nicNode = NULL;
|
||||
NCCLCHECK(xmlGetSub(parent, "nic", &nicNode));
|
||||
if (nicNode == NULL) {
|
||||
NCCLCHECK(xmlAddNode(xml, parent, "nic", &nicNode));
|
||||
}
|
||||
|
||||
// We know that this net does not exist yet (we searched for it at the
|
||||
// beginning of this function), so we can add it.
|
||||
NCCLCHECK(xmlAddNode(xml, nicNode, "net", netNode));
|
||||
NCCLCHECK(xmlSetAttr(*netNode, "name", netName));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
/**************************************************/
|
||||
/* Parser rules for the user-defined graph search */
|
||||
/**************************************************/
|
||||
|
||||
ncclResult_t ncclTopoXmlGraphLoadGpu(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, NULL, 0));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlGraphLoadNet(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, NULL, 0));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlGraphLoadChannel(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
struct xmlHandler handlers[] = { { "net", ncclTopoXmlGraphLoadNet }, { "gpu", ncclTopoXmlGraphLoadGpu } };
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, handlers, 2));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlGraphLoadGraph(FILE* file, struct ncclXml* xml, struct ncclXmlNode* head) {
|
||||
struct xmlHandler handlers[] = { { "channel", ncclTopoXmlGraphLoadChannel } };
|
||||
NCCLCHECK(xmlLoadSub(file, xml, head, handlers, 1));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoXmlGraphLoadGraphs(FILE* file, struct ncclXml* xmlGraph, struct ncclXmlNode* head) {
|
||||
int version;
|
||||
NCCLCHECK(xmlGetAttrInt(head, "version", &version));
|
||||
if (version != NCCL_GRAPH_XML_VERSION) {
|
||||
WARN("XML Graph has wrong version %d, %d needed", version, NCCL_GRAPH_XML_VERSION);
|
||||
return ncclInvalidUsage;
|
||||
}
|
||||
const char* name;
|
||||
NCCLCHECK(xmlGetAttr(head, "name", &name));
|
||||
if (name != NULL) INFO(NCCL_GRAPH, "Loading graphs for topology %s", name);
|
||||
else INFO(NCCL_GRAPH, "Loading graphs");
|
||||
|
||||
struct xmlHandler handlers[] = { { "graph", ncclTopoXmlGraphLoadGraph } };
|
||||
NCCLCHECK(xmlLoadSub(file, xmlGraph, head, handlers, 1));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclTopoGetXmlGraphFromFile(const char* xmlGraphFile, struct ncclXml* xml) {
|
||||
FILE* file = fopen(xmlGraphFile, "r");
|
||||
if (file == NULL) {
|
||||
WARN("Could not open XML graph file %s : %s", xmlGraphFile, strerror(errno));
|
||||
return ncclSystemError;
|
||||
}
|
||||
struct xmlHandler handlers[] = { { "graphs", ncclTopoXmlGraphLoadGraphs } };
|
||||
xml->maxIndex = 0;
|
||||
NCCLCHECK(xmlLoadSub(file, xml, NULL, handlers, 1));
|
||||
fclose(file);
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -0,0 +1,237 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2019-2020, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#ifndef XML_H_
|
||||
#define XML_H_
|
||||
|
||||
// A few constraints to make the implementation easy
|
||||
#define MAX_STR_LEN 256
|
||||
#define MAX_ATTR_COUNT 16
|
||||
#define MAX_SUBS 32
|
||||
#define MAX_NODES 1024
|
||||
|
||||
#define NODE_TYPE_NONE 0
|
||||
#define NODE_TYPE_OPEN 1
|
||||
#define NODE_TYPE_CLOSE 2
|
||||
#define NODE_TYPE_SINGLE 3
|
||||
|
||||
struct ncclXmlNode {
|
||||
char name[MAX_STR_LEN];
|
||||
struct {
|
||||
char key[MAX_STR_LEN];
|
||||
char value[MAX_STR_LEN];
|
||||
} attrs[MAX_ATTR_COUNT+1]; // Need an extra one to consume extra params
|
||||
int nAttrs;
|
||||
int type;
|
||||
struct ncclXmlNode* parent;
|
||||
struct ncclXmlNode* subs[MAX_SUBS];
|
||||
int nSubs;
|
||||
};
|
||||
|
||||
struct ncclXml {
|
||||
struct ncclXmlNode nodes[MAX_NODES];
|
||||
int maxIndex;
|
||||
};
|
||||
|
||||
/* File functions */
|
||||
#define NCCL_TOPO_XML_VERSION 1
|
||||
ncclResult_t ncclTopoGetXmlFromFile(const char* xmlTopoFile, struct ncclXml* xml);
|
||||
ncclResult_t ncclTopoDumpXmlToFile(const char* xmlTopoFile, struct ncclXml* xml);
|
||||
#define NCCL_GRAPH_XML_VERSION 1
|
||||
ncclResult_t ncclTopoGetXmlGraphFromFile(const char* xmlGraphFile, struct ncclXml* xml);
|
||||
|
||||
/* Auto-detect functions */
|
||||
ncclResult_t ncclTopoFillGpu(struct ncclXml* xml, const char* busId, struct ncclXmlNode** gpuNode);
|
||||
ncclResult_t ncclTopoFillNet(struct ncclXml* xml, const char* pciPath, const char* netName, struct ncclXmlNode** netNode);
|
||||
|
||||
/**************/
|
||||
/* XML Struct */
|
||||
/* Functions */
|
||||
/**************/
|
||||
|
||||
static ncclResult_t xmlGetAttrIndex(struct ncclXmlNode* node, const char* attrName, int* index) {
|
||||
*index = -1;
|
||||
const int nAttrs = node->nAttrs;
|
||||
for (int a=0; a<nAttrs; a++) {
|
||||
if (strncmp(node->attrs[a].key, attrName, MAX_STR_LEN-1) == 0) {
|
||||
*index = a;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetAttr(struct ncclXmlNode* node, const char* attrName, const char** value) {
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(node, attrName, &index));
|
||||
*value = index == -1 ? NULL : node->attrs[index].value;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetAttrStr(struct ncclXmlNode* node, const char* attrName, const char** value) {
|
||||
NCCLCHECK(xmlGetAttr(node, attrName, value));
|
||||
if (*value == NULL) {
|
||||
WARN("Attribute %s of node %s not found", attrName, node->name);
|
||||
return ncclInternalError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
static ncclResult_t xmlGetAttrInt(struct ncclXmlNode* node, const char* attrName, int* value) {
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttrStr(node, attrName, &str));
|
||||
*value = strtol(str, NULL, 0);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetAttrFloat(struct ncclXmlNode* node, const char* attrName, float* value) {
|
||||
const char* str;
|
||||
NCCLCHECK(xmlGetAttrStr(node, attrName, &str));
|
||||
*value = strtof(str, NULL);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlFindTag(struct ncclXml* xml, const char* tagName, struct ncclXmlNode** node) {
|
||||
*node = NULL;
|
||||
for (int i=0; i<xml->maxIndex; i++) {
|
||||
struct ncclXmlNode* n = xml->nodes+i;
|
||||
if (strcmp(n->name, tagName) == 0) {
|
||||
*node = n;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlFindTagKv(struct ncclXml* xml, const char* tagName, struct ncclXmlNode** node, const char* attrName, const char* attrValue) {
|
||||
*node = NULL;
|
||||
for (int i=0; i<xml->maxIndex; i++) {
|
||||
struct ncclXmlNode* n = xml->nodes+i;
|
||||
if (strcmp(n->name, tagName) == 0) {
|
||||
const char* value;
|
||||
NCCLCHECK(xmlGetAttr(n, attrName, &value));
|
||||
if (value && strcmp(value, attrValue) == 0) {
|
||||
*node = n;
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlSetAttr(struct ncclXmlNode* node, const char* attrName, const char* value) {
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(node, attrName, &index));
|
||||
if (index == -1) {
|
||||
index = node->nAttrs++;
|
||||
strncpy(node->attrs[index].key, attrName, MAX_STR_LEN);
|
||||
}
|
||||
strncpy(node->attrs[index].value, value, MAX_STR_LEN);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlSetAttrInt(struct ncclXmlNode* node, const char* attrName, const int value) {
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(node, attrName, &index));
|
||||
if (index == -1) {
|
||||
index = node->nAttrs++;
|
||||
strncpy(node->attrs[index].key, attrName, MAX_STR_LEN);
|
||||
}
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "%d", value);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlSetAttrFloat(struct ncclXmlNode* node, const char* attrName, const float value) {
|
||||
int index;
|
||||
NCCLCHECK(xmlGetAttrIndex(node, attrName, &index));
|
||||
if (index == -1) {
|
||||
index = node->nAttrs++;
|
||||
strncpy(node->attrs[index].key, attrName, MAX_STR_LEN);
|
||||
}
|
||||
snprintf(node->attrs[index].value, MAX_STR_LEN, "%g", value);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetSub(struct ncclXmlNode* node, const char* subName, struct ncclXmlNode** sub) {
|
||||
*sub = NULL;
|
||||
for (int s=0; s<node->nSubs; s++) {
|
||||
if (strcmp(node->subs[s]->name, subName) == 0) {
|
||||
*sub = node->subs[s];
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlGetSubKv(struct ncclXmlNode* node, const char* subName, struct ncclXmlNode** sub, const char* attrName, const char* attrValue) {
|
||||
*sub = NULL;
|
||||
for (int s=0; s<node->nSubs; s++) {
|
||||
struct ncclXmlNode* subNode = node->subs[s];
|
||||
if (strcmp(subNode->name, subName) == 0) {
|
||||
const char* value;
|
||||
NCCLCHECK(xmlGetAttr(subNode, attrName, &value));
|
||||
if (value && strcmp(value, attrValue) == 0) {
|
||||
*sub = node->subs[s];
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
static ncclResult_t xmlGetSubKvInt(struct ncclXmlNode* node, const char* subName, struct ncclXmlNode** sub, const char* attrName, const int attrValue) {
|
||||
char strValue[10];
|
||||
snprintf(strValue, 10, "%d", attrValue);
|
||||
NCCLCHECK(xmlGetSubKv(node, subName, sub, attrName, strValue));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
static ncclResult_t xmlAddNode(struct ncclXml* xml, struct ncclXmlNode* parent, const char* subName, struct ncclXmlNode** sub) {
|
||||
if (xml->maxIndex == MAX_NODES) {
|
||||
WARN("Error : too many XML nodes (max %d)", MAX_NODES);
|
||||
return ncclInternalError;
|
||||
}
|
||||
struct ncclXmlNode* s = xml->nodes+xml->maxIndex++;
|
||||
s->nSubs = 0;
|
||||
s->nAttrs = 0;
|
||||
*sub = s;
|
||||
s->parent = parent;
|
||||
if (parent) parent->subs[parent->nSubs++] = s;
|
||||
strncpy(s->name, subName, MAX_STR_LEN);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Dictionary for STR -> INT conversions. No dictionary size information,
|
||||
// there needs to be a last element with str == NULL.
|
||||
struct kvDict {
|
||||
const char* str;
|
||||
int value;
|
||||
};
|
||||
|
||||
static ncclResult_t kvConvertToInt(const char* str, int* value, struct kvDict* dict) {
|
||||
struct kvDict* d = dict;
|
||||
while (d->str) {
|
||||
if (strncmp(str, d->str, strlen(d->str)) == 0) {
|
||||
*value = d->value;
|
||||
return ncclSuccess;
|
||||
}
|
||||
d++;
|
||||
}
|
||||
WARN("KV Convert to int : could not find value of '%s' in dictionary", str);
|
||||
return ncclInternalError;
|
||||
}
|
||||
static ncclResult_t kvConvertToStr(int value, const char** str, struct kvDict* dict) {
|
||||
struct kvDict* d = dict;
|
||||
while (d->str) {
|
||||
if (value == d->value) {
|
||||
*str = d->str;
|
||||
return ncclSuccess;
|
||||
}
|
||||
d++;
|
||||
}
|
||||
WARN("KV Convert to str : could not find value %d in dictionary", value);
|
||||
return ncclInternalError;
|
||||
}
|
||||
|
||||
#endif
|
||||
Reference in New Issue
Block a user