Increase default number of channels for MI300A in multi-node scenario. (#1366)
This commit changed the default of channels of MI300A from 8 upto 24. This helps bring up multi-node performance to the expected level.
This commit is contained in:
committed by
GitHub
parent
b55b6be0cb
commit
133ea201cf
+15
-6
@@ -1485,12 +1485,21 @@ static ncclResult_t initTransportsRank(struct ncclComm* comm, struct ncclComm* p
|
|||||||
if (ringGraph.nChannels > MAXCHANNELS/2)
|
if (ringGraph.nChannels > MAXCHANNELS/2)
|
||||||
allGather3Data[rank].nc = 1;
|
allGather3Data[rank].nc = 1;
|
||||||
if (IsArchMatch(comm->topo->nodes[GPU].nodes[idx].gpu.gcn, "gfx94")) {
|
if (IsArchMatch(comm->topo->nodes[GPU].nodes[idx].gpu.gcn, "gfx94")) {
|
||||||
if (nranks == 2)
|
// Multi-node MI300A
|
||||||
// NCCL_MIN_NCHANNELS=32
|
int managed = 0;
|
||||||
allGather3Data[rank].nc = 16;
|
CUDACHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeDirectManagedMemAccessFromHost, 0));
|
||||||
else if (nranks == 4)
|
if (managed && nNodes > 1) {
|
||||||
// NCCL_MIN_NCHANNELS=24
|
// This forces the minimum channels to 24
|
||||||
allGather3Data[rank].nc = 4;
|
allGather3Data[rank].nc = 6;
|
||||||
|
} else {
|
||||||
|
// MI300X
|
||||||
|
if (nranks == 2)
|
||||||
|
// NCCL_MIN_NCHANNELS=32
|
||||||
|
allGather3Data[rank].nc = 16;
|
||||||
|
else if (nranks == 4)
|
||||||
|
// NCCL_MIN_NCHANNELS=24
|
||||||
|
allGather3Data[rank].nc = 4;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
allGather3Data[rank].pivotA2AEnabled = comm->topo->pivotA2AEnabled && rcclParamPivotAlltoallEnable();
|
allGather3Data[rank].pivotA2AEnabled = comm->topo->pivotA2AEnabled && rcclParamPivotAlltoallEnable();
|
||||||
|
|||||||
Reference in New Issue
Block a user