|
|
|
@@ -53,7 +53,7 @@ ncclResult_t parseList(const char* str, const char* elems[], int nelems, int* li
|
|
|
|
|
|
|
|
|
|
// Latencies in us, Bandwidths in GB/s
|
|
|
|
|
// Tree { LL, LL128, Simple } , Ring { LL, LL128, Simple }
|
|
|
|
|
static const float baseLat [NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] = { { 4.4, 4.4, 0 }, { 3.6, 10.0, 8.4 }, { 4.4, 4.4, 0 }, { 4.4, 4.4, 0 }};
|
|
|
|
|
static const float baseLat [NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] = { { 4.4, 4.4, 0 }, { 3.6, 10.0, 8.4 }, { 4.4, 4.4, 0 }, { 4.4, 4.4, 0 }, { 0, 0, 40.0 }};
|
|
|
|
|
|
|
|
|
|
// NVLink, PCI, Network
|
|
|
|
|
#define NCCL_HW_NVLINK 0
|
|
|
|
@@ -63,13 +63,16 @@ static const float baseLat [NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] = { { 4.4,
|
|
|
|
|
static float hwLat [3][NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] =
|
|
|
|
|
{ /* NVLINK */
|
|
|
|
|
{ /* Tree (LL/LL128/Simple)*/ { .52, 1.25, 28 }, /* Ring (LL/LL128/Simple)*/ { .47, 1.9, 3.4 },
|
|
|
|
|
/* CollNetDirect (Simple)*/ { 0, 0, 8.0 }, /* CollNetChain (Simple)*/ { 0, 0, 8.0 } },
|
|
|
|
|
/* CollNetDirect (Simple)*/ { 0, 0, 8.0 }, /* CollNetChain (Simple)*/ { 0, 0, 8.0 },
|
|
|
|
|
/* NVLS */ { 0, 0, 0 } },
|
|
|
|
|
/* PCI */
|
|
|
|
|
{ /* Tree (LL/LL128/Simple)*/ { 1.0, 1.9, 28 }, /* Ring (LL/LL128/Simple)*/ { 1.0, 2.5, 5.7 },
|
|
|
|
|
/* CollNetDirect (Simple)*/ { 0, 0, 8.0 }, /* CollNetChain (Simple)*/ { 0, 0, 8.0 } },
|
|
|
|
|
/* CollNetDirect (Simple)*/ { 0, 0, 8.0 }, /* CollNetChain (Simple)*/ { 0, 0, 8.0 },
|
|
|
|
|
/* NVLS */ { 0, 0, 0 } },
|
|
|
|
|
/* NET */
|
|
|
|
|
{ /* Tree (LL/LL128/Simple)*/ { 5.0, 8.5, 28 }, /* Ring (LL/LL128/Simple)*/ { 2.7, 4.0, 9.6 },
|
|
|
|
|
/* CollNetDirect (Simple)*/ { 0, 0, 10.7 }, /* CollNetChain (Simple)*/ { 0, 0, 10.7 } }
|
|
|
|
|
/* CollNetDirect (Simple)*/ { 0, 0, 10.7 }, /* CollNetChain (Simple)*/ { 0, 0, 10.7 },
|
|
|
|
|
/* NVLS */ { 0, 0, 0 } }
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
/* Array indexes used below */
|
|
|
|
@@ -78,7 +81,7 @@ static float hwLat [3][NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] =
|
|
|
|
|
#define HOPPER_COMPCAP_IDX 2
|
|
|
|
|
|
|
|
|
|
// LL128 max BW per channel
|
|
|
|
|
static const double ll128MaxBwPerCh = 20.0;
|
|
|
|
|
static const double ll128MaxBwPerCh[3] = { 20.0, 20.0, 36.7 };
|
|
|
|
|
static const double llMaxBws[3][3] = {
|
|
|
|
|
/* Volta-N1/Intel-N2/Intel-N4) */ {39.0, 39.0, 20.4},
|
|
|
|
|
/* Ampere-N1/AMD-N2/AMD-N4) */ {87.7, 22.5 /*avg of ring & tree*/, 19.0},
|
|
|
|
@@ -88,7 +91,7 @@ static const double llMaxBws[3][3] = {
|
|
|
|
|
static const double perChMaxTreeBws[3][3] = {
|
|
|
|
|
/* Volta (N1/N2/N4) */ {26.5, 18.5, 10.0},
|
|
|
|
|
/* Ampere (N1/N2/N4) */ {24.0, 23.6, 17.8},
|
|
|
|
|
/* Hopper (N1/N2/N4) */ {24.0, 23.6, 17.8},
|
|
|
|
|
/* Hopper (N1/N2/N4) */ {38.7, 41.4, 33.0},
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCompCap, struct ncclTopoGraph* treeGraph, struct ncclTopoGraph* ringGraph, struct ncclTopoGraph* collNetGraph) {
|
|
|
|
@@ -98,7 +101,8 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
comm->maxThreads[NCCL_ALGO_TREE][NCCL_PROTO_SIMPLE] =
|
|
|
|
|
getNthreads("NCCL_NTHREADS", ncclParamNthreads(), 2*WARP_SIZE, NCCL_SIMPLE_MAX_NTHREADS, NCCL_SIMPLE_MAX_NTHREADS);
|
|
|
|
|
comm->maxThreads[NCCL_ALGO_COLLNET_DIRECT][NCCL_PROTO_SIMPLE] =
|
|
|
|
|
comm->maxThreads[NCCL_ALGO_COLLNET_CHAIN][NCCL_PROTO_SIMPLE] = NCCL_SIMPLE_MAX_NTHREADS;
|
|
|
|
|
comm->maxThreads[NCCL_ALGO_COLLNET_CHAIN][NCCL_PROTO_SIMPLE] =
|
|
|
|
|
comm->maxThreads[NCCL_ALGO_NVLS][NCCL_PROTO_SIMPLE] = NCCL_SIMPLE_MAX_NTHREADS;
|
|
|
|
|
comm->maxThreads[NCCL_ALGO_RING][NCCL_PROTO_LL] = comm->maxThreads[NCCL_ALGO_TREE][NCCL_PROTO_LL] =
|
|
|
|
|
getNthreads("NCCL_NTHREADS", ncclParamNthreads(), 2*WARP_SIZE, NCCL_LL_MAX_NTHREADS, NCCL_LL_MAX_NTHREADS);
|
|
|
|
|
comm->maxThreads[NCCL_ALGO_RING][NCCL_PROTO_LL128] = comm->maxThreads[NCCL_ALGO_TREE][NCCL_PROTO_LL128] =
|
|
|
|
@@ -108,7 +112,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
int nRanks = comm->nRanks;
|
|
|
|
|
if (nRanks <= 1) return ncclSuccess;
|
|
|
|
|
|
|
|
|
|
int compCapIndex = (minCompCap == 80 && maxCompCap == 80) ? AMPERE_COMPCAP_IDX : ((minCompCap == 90 && maxCompCap == 90) ? HOPPER_COMPCAP_IDX : VOLTA_COMPCAP_IDX);
|
|
|
|
|
int compCapIndex = minCompCap >= 90 ? HOPPER_COMPCAP_IDX : minCompCap >= 80 ? AMPERE_COMPCAP_IDX : VOLTA_COMPCAP_IDX;
|
|
|
|
|
int cpuArch, cpuVendor, cpuModel;
|
|
|
|
|
NCCLCHECK(ncclTopoCpuType(comm->topo, &cpuArch, &cpuVendor, &cpuModel));
|
|
|
|
|
int index2 = nNodes <= 2 ? nNodes-1 : 2;
|
|
|
|
@@ -120,7 +124,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
if (cpuArch == NCCL_TOPO_CPU_ARCH_POWER) hwLat[NCCL_HW_PCI][NCCL_ALGO_TREE][NCCL_PROTO_SIMPLE] = hwLat[NCCL_HW_PCI][NCCL_ALGO_RING][NCCL_PROTO_SIMPLE];
|
|
|
|
|
float ppn = (float)nRanks / nNodes; // if ppn < 2, then we are sending/receiving at the same GPU through the NIC, apply some bw discount
|
|
|
|
|
|
|
|
|
|
struct ncclTopoGraph* graphs[NCCL_NUM_ALGORITHMS] = { treeGraph, ringGraph, collNetGraph, collNetGraph };
|
|
|
|
|
struct ncclTopoGraph* graphs[NCCL_NUM_ALGORITHMS] = { treeGraph, ringGraph, collNetGraph, collNetGraph, ringGraph/* we only need the NVSwitch speed for NVLS*/ };
|
|
|
|
|
int intraHw[NCCL_NUM_ALGORITHMS], hw[NCCL_NUM_ALGORITHMS];
|
|
|
|
|
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) intraHw[a] = graphs[a]->typeIntra == LINK_NVL ? NCCL_HW_NVLINK : NCCL_HW_PCI;
|
|
|
|
|
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) hw[a] = nNodes == 1 ? intraHw[a] : NCCL_HW_NET;
|
|
|
|
@@ -134,20 +138,25 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
nNodes;
|
|
|
|
|
|
|
|
|
|
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) {
|
|
|
|
|
if (coll != ncclFuncAllReduce && a != NCCL_ALGO_RING) continue;
|
|
|
|
|
if (coll == ncclFuncBroadcast && a != NCCL_ALGO_RING) continue;
|
|
|
|
|
if (coll == ncclFuncReduce && a != NCCL_ALGO_RING) continue;
|
|
|
|
|
if (coll == ncclFuncReduceScatter && a != NCCL_ALGO_RING && a != NCCL_ALGO_NVLS) continue;
|
|
|
|
|
if (coll == ncclFuncAllGather && a != NCCL_ALGO_RING && a != NCCL_ALGO_NVLS) continue;
|
|
|
|
|
|
|
|
|
|
for (int p=0; p<NCCL_NUM_PROTOCOLS; p++) {
|
|
|
|
|
if (a == NCCL_ALGO_NVLS && p != NCCL_PROTO_SIMPLE) continue;
|
|
|
|
|
int collnet = (a == NCCL_ALGO_COLLNET_DIRECT || a == NCCL_ALGO_COLLNET_CHAIN) ? 1 : 0;
|
|
|
|
|
float bw = nNodes <= 2 || collnet ? graphs[a]->bwIntra : graphs[a]->bwInter;
|
|
|
|
|
float busBw = graphs[a]->nChannels * bw;
|
|
|
|
|
|
|
|
|
|
// Various model refinements
|
|
|
|
|
if (compCapIndex == AMPERE_COMPCAP_IDX) busBw = std::min(busBw, 235.0f);
|
|
|
|
|
if (compCapIndex == HOPPER_COMPCAP_IDX) busBw = std::min(busBw, 370.0f);
|
|
|
|
|
if (a == NCCL_ALGO_RING && p == NCCL_PROTO_LL) { busBw = std::min(llMaxBw, busBw * ((nNodes > 1 || coll == ncclFuncAllReduce || coll == ncclFuncReduce) ? 1.0/4.0 : 1.0/3.0)); }
|
|
|
|
|
if (a == NCCL_ALGO_RING && p == NCCL_PROTO_LL128) busBw = std::min(busBw * (ppn < 2 ? 0.7 : 0.92 /*120.0/128.0*/), ll128MaxBwPerCh*graphs[a]->nChannels);
|
|
|
|
|
if (a == NCCL_ALGO_RING && p == NCCL_PROTO_LL128) busBw = std::min(busBw * (ppn < 2 ? 0.7 : 0.92 /*120.0/128.0*/), ll128MaxBwPerCh[compCapIndex]*graphs[a]->nChannels);
|
|
|
|
|
if (a == NCCL_ALGO_TREE) busBw = std::min(busBw*.92, graphs[a]->nChannels*perChMaxTreeBw);
|
|
|
|
|
if (a == NCCL_ALGO_TREE && p == NCCL_PROTO_LL) busBw = std::min(busBw*1.0/3.8, llMaxBw);
|
|
|
|
|
if (a == NCCL_ALGO_TREE && p == NCCL_PROTO_LL128) busBw = std::min(busBw * (nNodes == 1 ? 7.0/9.0 : 120.0/128.0), ll128MaxBwPerCh*graphs[a]->nChannels);
|
|
|
|
|
if (a == NCCL_ALGO_TREE && p == NCCL_PROTO_LL128) busBw = std::min(busBw * (nNodes == 1 ? 7.0/9.0 : 120.0/128.0), ll128MaxBwPerCh[compCapIndex]*graphs[a]->nChannels);
|
|
|
|
|
if (a == NCCL_ALGO_COLLNET_DIRECT && p != NCCL_PROTO_SIMPLE) busBw = 0; // Not used
|
|
|
|
|
if (a == NCCL_ALGO_COLLNET_CHAIN && p != NCCL_PROTO_SIMPLE) busBw = 0; // Not used
|
|
|
|
|
if (a == NCCL_ALGO_COLLNET_DIRECT && p == NCCL_PROTO_SIMPLE) {
|
|
|
|
@@ -159,7 +168,10 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
if (a == NCCL_ALGO_COLLNET_CHAIN && p == NCCL_PROTO_SIMPLE) busBw *= .75;
|
|
|
|
|
|
|
|
|
|
// Convert bus BW to algorithm BW
|
|
|
|
|
float ratio = (a != NCCL_ALGO_RING) ? .5 : (1.0 * nRanks) / nsteps;
|
|
|
|
|
float ratio;
|
|
|
|
|
if (a == NCCL_ALGO_RING) ratio = (1.0 * nRanks) / nsteps;
|
|
|
|
|
else if (a == NCCL_ALGO_NVLS) ratio = .75;
|
|
|
|
|
else ratio = .5;
|
|
|
|
|
comm->bandwidths[coll][a][p] = busBw * ratio;
|
|
|
|
|
|
|
|
|
|
comm->latencies[coll][a][p] = baseLat[a][p];
|
|
|
|
@@ -195,7 +207,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
// Protocols/Algorithms enable/disable, and user overrides.
|
|
|
|
|
// All are enabled except ll128 which is enabled by default only in certain cases.
|
|
|
|
|
int protoEnable[NCCL_NUM_PROTOCOLS] = { 1, 2, 1 };
|
|
|
|
|
int algoEnable[NCCL_NUM_ALGORITHMS] = { 1, 1, 1, 1 };
|
|
|
|
|
int algoEnable[NCCL_NUM_ALGORITHMS] = { 1, 1, 1, 1, 1 };
|
|
|
|
|
|
|
|
|
|
const char *protoStr = getenv("NCCL_PROTO");
|
|
|
|
|
if (protoStr) {
|
|
|
|
@@ -207,6 +219,10 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
INFO(NCCL_ENV, "NCCL_ALGO set by environment to %s", algoStr);
|
|
|
|
|
NCCLCHECK(parseList(algoStr, ncclAlgoStr, NCCL_NUM_ALGORITHMS, algoEnable));
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Disable NVLink SHARP if not supported
|
|
|
|
|
if (comm->nvlsSupport == 0 /* || comm->localRanks <= 2*/) algoEnable[NCCL_ALGO_NVLS] = 0;
|
|
|
|
|
|
|
|
|
|
// Disable CollNet if it is not supported
|
|
|
|
|
if (comm->collNetSupport == 0) {
|
|
|
|
|
algoEnable[NCCL_ALGO_COLLNET_DIRECT] = 0;
|
|
|
|
@@ -228,7 +244,7 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
if (pEnable == 2 && p == NCCL_PROTO_LL128) {
|
|
|
|
|
// Enable LL128 by default only on Volta/Ampere/Hopper+NVLink. Other cases are not tested and may cause silent data corruption.
|
|
|
|
|
pEnable = 1;
|
|
|
|
|
pEnable &= (graphs[a]->typeInter <= PATH_PXB);
|
|
|
|
|
pEnable &= (graphs[a]->typeInter <= PATH_PXB || (minCompCap >= 90 && graphs[a]->typeInter <= PATH_PXN));
|
|
|
|
|
pEnable &= (graphs[a]->typeIntra <= PATH_NVL);
|
|
|
|
|
pEnable &= (minCompCap == maxCompCap);
|
|
|
|
|
switch (minCompCap) {
|
|
|
|
@@ -239,8 +255,9 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
if (pEnable == 0) comm->bandwidths[c][a][p] = 0;
|
|
|
|
|
// Only disable algo for Allreduce since others only have one
|
|
|
|
|
if (c == ncclFuncAllReduce && algoEnable[a] == 0) comm->bandwidths[c][a][p] = 0;
|
|
|
|
|
// Never disable ring for non-allreduce operations. That allows to run real apps with NCCL_ALGO=TREE.
|
|
|
|
|
if (a == NCCL_ALGO_RING && c != ncclFuncAllReduce) continue;
|
|
|
|
|
if (algoEnable[a] == 0) comm->bandwidths[c][a][p] = 0;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (comm->rank == 0) {
|
|
|
|
@@ -284,9 +301,9 @@ ncclResult_t ncclTopoTuneModel(struct ncclComm* comm, int minCompCap, int maxCom
|
|
|
|
|
char* str = getenv("NCCL_THREAD_THRESHOLDS");
|
|
|
|
|
if (str) {
|
|
|
|
|
INFO(NCCL_ENV, "NCCL_THREAD_THRESHOLDS set by environment to %s", str);
|
|
|
|
|
ssize_t t[NCCL_NUM_ALGORITHMS][NCCL_NUM_PROTOCOLS] = {{ -2, -2, -2 }, { -2, -2, -2 }, { -2, -2, -2 }, { -2, -2, -2 }};
|
|
|
|
|
ssize_t t[2][NCCL_NUM_PROTOCOLS] = {{ -2, -2, -2 }, { -2, -2, -2 }};
|
|
|
|
|
sscanf(str, "%ld %ld %ld %ld %ld %ld", t[0], t[0]+1, t[0]+2, t[1], t[1]+1, t[1]+2);
|
|
|
|
|
for (int a=0; a<NCCL_NUM_ALGORITHMS; a++) {
|
|
|
|
|
for (int a=0; a<2; a++) {
|
|
|
|
|
for (int p=0; p<NCCL_NUM_PROTOCOLS; p++) {
|
|
|
|
|
if (t[a][p] >= 0) comm->threadThresholds[a][p] = t[a][p];
|
|
|
|
|
}
|
|
|
|
@@ -323,7 +340,9 @@ ncclResult_t ncclTopoGetAlgoTime(struct ncclInfo* info, int algorithm, int proto
|
|
|
|
|
if (algorithm == NCCL_ALGO_TREE && logSize < 23) bw *= treeCorrectionFactor[protocol][logSize];
|
|
|
|
|
if (info->nChannels != 0) bw = bw / info->comm->nChannels * info->nChannels;
|
|
|
|
|
if (algorithm == NCCL_ALGO_RING && protocol == NCCL_PROTO_SIMPLE && info->comm->nNodes > 1
|
|
|
|
|
&& info->coll == ncclFuncAllReduce && info->nBytes >= info->comm->nRanks/16.0*65536) lat *= 1.9; // Plateau effect of ring
|
|
|
|
|
&& info->coll == ncclFuncAllReduce && info->nBytes >= info->comm->nRanks/16.0*65536) {
|
|
|
|
|
lat *= info->comm->minCompCap < 90 ? 1.9 : 1.5; // Plateau effect of ring
|
|
|
|
|
}
|
|
|
|
|
// Tree pipelining saves latency in aggregation cases
|
|
|
|
|
int latCount = algorithm == NCCL_ALGO_RING ? numPipeOps : DIVUP(numPipeOps, NCCL_MAX_WORK_ELEMENTS);
|
|
|
|
|
*time = lat * latCount + (info->nBytes) / (1000 * bw);
|
|
|
|
|