2.12.12-1

Improve allreduce performance when we have more than one network interface per
GPU and we need to use PXN to close rings.
Add support for PCI Gen5 on 5.4 kernels.
Fix crash when setting NCCL_SET_THREAD_NAME.
Fix random crash in init due to uninitialized struct.
Fix hang on cubemesh topologies.
Add P2P_DIRECT_DISABLE parameter to disable direct access to pointers within a
process.
This commit is contained in:
Sylvain Jeaugey
2022-05-03 01:30:26 -07:00
parent 9bfc1c6e35
commit 7aa1c46fd5
12 changed files with 94 additions and 50 deletions
+7 -15
View File
@@ -9,6 +9,7 @@
#include "coll_net.h"
#include "gdrwrap.h"
#include "bootstrap.h"
#include "channel.h"
#include <cstring> // std::memcpy
@@ -861,20 +862,14 @@ static ncclResult_t ncclSaveP2p(struct ncclInfo* info) {
struct ncclComm* comm = info->comm;
int peer = info->root;
ssize_t nBytes = info->count*ncclTypeSize(info->datatype);
int p2pGroupSize = NCCL_MAX_WORK_ELEMENTS_P2P/2;
int peerNode = comm->rankToNode[peer];
int peerIndex = comm->rankToLocalRank[peer];
int nsteps = comm->maxLocalRanks;
int rankIndex = comm->rankToLocalRank[comm->rank];
int channelBaseId;
NCCLCHECK(ncclChannelComputeBase(comm, peer, info->coll, &channelBaseId));
if (info->coll == ncclFuncSend) {
if (peer != comm->rank) {
int step = (nsteps + peerIndex - rankIndex)%nsteps;
int delta = (comm->nNodes + peerNode - comm->node) % comm->nNodes;
if (comm->nNodes == 1) delta = (comm->nRanks + peer - comm->rank) % comm->nRanks;
// Mark channels that need pre-connect
for (int c=0; c<comm->p2pnChannelsPerPeer; c++) {
int shuffle = comm->nNodes > 1 ? delta+(step/p2pGroupSize) : step;
int channelId = (shuffle+comm->p2pChannels[c]) % comm->p2pnChannels;
int channelId;
NCCLCHECK(ncclChannelComputeFromBase(comm, channelBaseId, c, &channelId));
if (comm->channels[channelId].peers[peer].send[1].connected == 0) { // P2P uses only 1 connector
comm->connectSend[peer] |= (1<<channelId);
comm->connect = 1;
@@ -885,13 +880,10 @@ static ncclResult_t ncclSaveP2p(struct ncclInfo* info) {
comm->p2pSendCount++;
} else {
if (peer != comm->rank) {
int step = (nsteps + rankIndex - peerIndex)%nsteps;
int delta = (comm->nNodes + comm->node - peerNode) % comm->nNodes;
if (comm->nNodes == 1) delta = (comm->nRanks - peer + comm->rank) % comm->nRanks;
// Mark channels that need pre-connect
for (int c=0; c<comm->p2pnChannelsPerPeer; c++) {
int shuffle = comm->nNodes > 1 ? delta+(step/p2pGroupSize) : step;
int channelId = (shuffle+comm->p2pChannels[c]) % comm->p2pnChannels;
int channelId;
NCCLCHECK(ncclChannelComputeFromBase(comm, channelBaseId, c, &channelId));
if (comm->channels[channelId].peers[peer].recv[1].connected == 0) { // P2P uses only 1 connector
comm->connectRecv[peer] |= (1<<channelId);
comm->connect = 1;