Increase default number of channels for MI300A in multi-node scenario. (#1366)

This commit changed the default of channels of MI300A from 8 upto 24.
This helps bring up multi-node performance to the expected level.
此提交包含在:
Arm Patinyasakdikul
2024-10-11 11:37:48 -05:00
提交者 GitHub
父節點 b55b6be0cb
當前提交 133ea201cf
+15 -6
查看文件
@@ -1485,12 +1485,21 @@ static ncclResult_t initTransportsRank(struct ncclComm* comm, struct ncclComm* p
if (ringGraph.nChannels > MAXCHANNELS/2) if (ringGraph.nChannels > MAXCHANNELS/2)
allGather3Data[rank].nc = 1; allGather3Data[rank].nc = 1;
if (IsArchMatch(comm->topo->nodes[GPU].nodes[idx].gpu.gcn, "gfx94")) { if (IsArchMatch(comm->topo->nodes[GPU].nodes[idx].gpu.gcn, "gfx94")) {
if (nranks == 2) // Multi-node MI300A
// NCCL_MIN_NCHANNELS=32 int managed = 0;
allGather3Data[rank].nc = 16; CUDACHECK(hipDeviceGetAttribute(&managed, hipDeviceAttributeDirectManagedMemAccessFromHost, 0));
else if (nranks == 4) if (managed && nNodes > 1) {
// NCCL_MIN_NCHANNELS=24 // This forces the minimum channels to 24
allGather3Data[rank].nc = 4; allGather3Data[rank].nc = 6;
} else {
// MI300X
if (nranks == 2)
// NCCL_MIN_NCHANNELS=32
allGather3Data[rank].nc = 16;
else if (nranks == 4)
// NCCL_MIN_NCHANNELS=24
allGather3Data[rank].nc = 4;
}
} }
allGather3Data[rank].pivotA2AEnabled = comm->topo->pivotA2AEnabled && rcclParamPivotAlltoallEnable(); allGather3Data[rank].pivotA2AEnabled = comm->topo->pivotA2AEnabled && rcclParamPivotAlltoallEnable();