178b6b7590
Rework core for NVIDIA Trusted Computing * Compress work structs so that they are shared between channels * Utilize the full amount of kernel argument space permitted (4k) before resorting to work fifo. * Rework the task preprocessing phase. * Use a separate abortDevFlag which is kept in sync with abortFlag using cudaMemcpy operations. * Rename src/include/align.h to src/include/bitops.h Add lazy connection establishment for collective operations * Move buffer allocation and connection establishment to the first collective operation using that algorithm. * Accelerate init time and reduce memory usage. * Avoid allocating NVLS buffers if all calls are registered. * Compute algo/proto in ncclLaunchCollTasksInfo early on. * Connect peers in ncclCollPreconnectFunc if not connected already. * Also move shared buffer creation to the first send/recv call. Accelerate intra-node NVLink detection * Make each rank only detect NVLinks attached to its GPU. * Fuse XMLs to reconstruct the full NVLink topology Add init profiling to report time spend in different init phases. * Report timings of bootstrap, allgather, search, connect, etc. * Add new "PROFILE" category for NCCL_DEBUG_SUBSYS. Add support for PCI p2p on split PCI switches * Detect split PCI switches through a kernel module exposing switch information. * Update the topology XML and graph to add those inter-switch connections. Add cost estimation API * Add a new ncclGroupEndSimulate primitive to return the estimated time a group would take. Net/IB: Add separate traffic class for fifo messages * Add NCCL_IB_FIFO_TC to control the traffic class of fifo messages independently from NCCL_IB_TC. Merges PR #1194 Net/IB: Add support for IB router * Use flid instead of lid if subnets do not match * Warn if flid is 0 Optimizations and fixes for device network offload (unpack) * Double the default number of channels * Cache netDeviceType * Fix save/increment head logic to enable Tree support. Support ncclGroupStart/End for ncclCommAbort/Destroy * Allow Abort/Destroy to be called within a group when managing multiple GPUs with a single process. Improve Tuner API * Provide to the plugin the original cost table so that the plugin can leave unknown or disabled algo/proto combinations untouched. * Remove nvlsSupport and collnetSupport. Do not print version to stdout when using a debug file * Also print version from all processes with INFO debug level. Fixes issue #1271 Fix clang warnings in NVTX headers * Update NVTX headers to the latest version Fixes issue #1270 Disable port fusion in heterogeneous systems * Do not fuse ports if a mix of multi-port and single port are detected. Fix NVLS graphs search for dual NICs. * Fix NVLS graph search when we have more than one NIC per GPU. Fix crash with collnetDirect * Add separate graph search for collnetDirect, testing alltoall paths and working similarly to the NVLS search. Fix hang when nodes have different CPU types * Add the CPU type to the rank peer info. * Align all ranks on the CPU type after the first allgather. * Only use the aligned CPU type for all tuning operations. Fixes issue #1136 Fixes issue #1184 Fix performance of registered send/recv operations * Allow for single full size operations * Add INFO to confirm the registration of send/recv buffers. Move all sync ops to finalize stage * Ensure ncclCommDestroy is non-blocking if ncclCommFinalize has been called. Improve error reporting during SHM segment creation Improve support of various compilers Merges PR #1177 Merges PR #1228 Allow net and tuner plugins to be statically linked * Search for ncclNet or ncclTuner symbols in the main binary. Merges PR #979 Plugin examples includes cleanup * Harmonize err.h and common.h usage. * Add mixed plugin with both net and tuner.
177 righe
7.9 KiB
C++
177 righe
7.9 KiB
C++
/*************************************************************************
|
|
* Copyright (c) 2015-2022, NVIDIA CORPORATION. All rights reserved.
|
|
*
|
|
* See LICENSE.txt for license information
|
|
************************************************************************/
|
|
|
|
#include "channel.h"
|
|
#include "param.h"
|
|
#include "gdrwrap.h"
|
|
#include "transport.h"
|
|
|
|
ncclResult_t initChannel(struct ncclComm* comm, int channelId) {
|
|
struct ncclChannel* channel = &comm->channels[channelId];
|
|
if (channel->id != -1) return ncclSuccess;
|
|
|
|
int nRanks = comm->nRanks;
|
|
int nvlsRanks = comm->localRanks;
|
|
int nPeers = nRanks + 1 /* Collnet */ + nvlsRanks /* NVLS */;
|
|
channel->id = channelId;
|
|
channel->workFifoProduced = 0;
|
|
|
|
struct ncclSharedResources* sharedRes = comm->sharedRes;
|
|
|
|
NCCLCHECK(ncclStrongStreamAcquireUncaptured(&sharedRes->deviceStream));
|
|
|
|
if (channel->peers == NULL) {
|
|
// The extra on nRanks+1 is for collnet root (i.e. network)
|
|
// Allocate everything related to sharedRes with ncclCalloc as this can be
|
|
// shared between communicators hence should not be tied to comm.
|
|
if (sharedRes->peers[channelId] == NULL) {
|
|
NCCLCHECK(ncclCalloc(sharedRes->peers + channelId, sharedRes->tpNRanks));
|
|
}
|
|
channel->peers = ncclMemoryStackAlloc<struct ncclChannelPeer*>(&comm->memPermanent, nPeers);
|
|
for (int r = 0; r < nRanks; r++) {
|
|
channel->peers[r] = comm->sharedRes->peers[channelId] + comm->topParentRanks[r];
|
|
ncclAtomicRefCountIncrement(&channel->peers[r]->refCount);
|
|
}
|
|
}
|
|
|
|
if (channel->devPeers == NULL) {
|
|
if (sharedRes->devPeers[channelId] == NULL) {
|
|
NCCLCHECK(ncclCudaCallocAsync(sharedRes->devPeers + channelId, sharedRes->tpNRanks, sharedRes->deviceStream.cudaStream));
|
|
}
|
|
/* channel->devPeers is not shared, so just free it when calling commFree() */
|
|
NCCLCHECK(ncclCudaCallocAsync(&channel->devPeers, nPeers, sharedRes->deviceStream.cudaStream));
|
|
ncclCommPushCudaFree(comm, channel->devPeers);
|
|
NCCLCHECK(ncclCalloc(&channel->devPeersHostPtr, nPeers));
|
|
for (int r = 0; r < nRanks; r++) {
|
|
uintptr_t addr = (uintptr_t)(comm->sharedRes->devPeers[channelId] + comm->topParentRanks[r]);
|
|
NCCLCHECK(ncclCudaMemcpyAsync((uintptr_t*)(channel->devPeers + r), (uintptr_t*)&addr, 1, sharedRes->deviceStream.cudaStream));
|
|
channel->devPeersHostPtr[r] = (struct ncclDevChannelPeer*)addr;
|
|
}
|
|
}
|
|
|
|
channel->ring.userRanks = ncclMemoryStackAlloc<int>(&comm->memPermanent, nRanks);
|
|
NCCLCHECK(ncclCudaCallocAsync(&channel->devRingUserRanks, nRanks, sharedRes->deviceStream.cudaStream));
|
|
ncclCommPushCudaFree(comm, channel->devRingUserRanks);
|
|
|
|
/* guarantee addr has been copied into channel->devPeers */
|
|
NCCLCHECK(ncclStrongStreamSynchronize(&sharedRes->deviceStream));
|
|
NCCLCHECK(ncclStrongStreamRelease(ncclCudaGraphNone(), &sharedRes->deviceStream));
|
|
|
|
return ncclSuccess;
|
|
}
|
|
|
|
ncclResult_t initNvlsChannel(struct ncclComm* comm, int channelId, struct ncclComm* parent, bool share) {
|
|
struct ncclChannel* channel = &comm->channels[channelId];
|
|
struct ncclSharedResources* sharedRes = comm->sharedRes;
|
|
|
|
if (channel->nvlsPeers != NULL)
|
|
return ncclSuccess;
|
|
|
|
if (channel->id == -1)
|
|
NCCLCHECK(initChannel(comm, channelId));
|
|
|
|
NCCLCHECK(ncclStrongStreamAcquireUncaptured(&sharedRes->deviceStream));
|
|
|
|
int nvlsRanks = comm->localRanks;
|
|
|
|
if (share) {
|
|
channel->nvlsPeers = parent->channels[channelId].nvlsPeers;
|
|
channel->nvlsDevPeers = parent->channels[channelId].nvlsDevPeers;
|
|
for (int r = 0; r < nvlsRanks; ++r) {
|
|
int tr = comm->topParentLocalRanks[r];
|
|
uintptr_t addr = (uintptr_t)(parent->channels[channelId].nvlsDevPeers + tr);
|
|
channel->peers[comm->nRanks + 1 + r] = parent->channels[channelId].nvlsPeers + tr;
|
|
NCCLCHECK(ncclCudaMemcpyAsync((uintptr_t*)(channel->devPeers + comm->nRanks + 1 + r), (uintptr_t*)&addr, 1, sharedRes->deviceStream.cudaStream));
|
|
channel->devPeersHostPtr[comm->nRanks + 1 + r] = (struct ncclDevChannelPeer*)addr;
|
|
ncclAtomicRefCountIncrement(&parent->channels[channelId].nvlsPeers[tr].refCount);
|
|
}
|
|
} else {
|
|
NCCLCHECK(ncclCalloc(&channel->nvlsPeers, nvlsRanks));
|
|
NCCLCHECK(ncclCudaCallocAsync(&channel->nvlsDevPeers, nvlsRanks, sharedRes->deviceStream.cudaStream));
|
|
for (int r = 0; r < nvlsRanks; ++r) {
|
|
uintptr_t addr = (uintptr_t)(channel->nvlsDevPeers + r);
|
|
channel->peers[comm->nRanks + 1 + r] = channel->nvlsPeers + r;
|
|
NCCLCHECK(ncclCudaMemcpyAsync((uintptr_t*)(channel->devPeers + comm->nRanks + 1 + r), (uintptr_t*)&addr, 1, sharedRes->deviceStream.cudaStream));
|
|
channel->devPeersHostPtr[comm->nRanks + 1 + r] = (struct ncclDevChannelPeer*)addr;
|
|
ncclAtomicRefCountIncrement(&channel->nvlsPeers[r].refCount);
|
|
}
|
|
}
|
|
|
|
NCCLCHECK(ncclStrongStreamSynchronize(&sharedRes->deviceStream));
|
|
NCCLCHECK(ncclStrongStreamRelease(ncclCudaGraphNone(), &sharedRes->deviceStream));
|
|
|
|
return ncclSuccess;
|
|
}
|
|
|
|
ncclResult_t initCollnetChannel(struct ncclComm* comm, int channelId, struct ncclComm* parent, bool share) {
|
|
struct ncclChannel* channel = &comm->channels[channelId];
|
|
struct ncclSharedResources* sharedRes = comm->sharedRes;
|
|
uintptr_t addr;
|
|
|
|
if (channel->collnetPeers != NULL)
|
|
return ncclSuccess;
|
|
|
|
if (channel->id == -1)
|
|
NCCLCHECK(initChannel(comm, channelId));
|
|
|
|
NCCLCHECK(ncclStrongStreamAcquireUncaptured(&sharedRes->deviceStream));
|
|
|
|
if (share) {
|
|
channel->collnetPeers = parent->channels[channelId].collnetPeers;
|
|
channel->collnetDevPeers = parent->channels[channelId].collnetDevPeers;
|
|
addr = (uintptr_t)parent->channels[channelId].collnetDevPeers;
|
|
channel->peers[comm->nRanks] = parent->channels[channelId].collnetPeers;
|
|
NCCLCHECK(ncclCudaMemcpyAsync((uintptr_t*)(channel->devPeers + comm->nRanks), (uintptr_t*)&addr, 1, sharedRes->deviceStream.cudaStream));
|
|
channel->devPeersHostPtr[comm->nRanks] = (struct ncclDevChannelPeer*)addr;
|
|
ncclAtomicRefCountIncrement(&parent->channels[channelId].collnetPeers->refCount);
|
|
} else {
|
|
NCCLCHECK(ncclCalloc(&channel->collnetPeers, 1));
|
|
NCCLCHECK(ncclCudaCallocAsync(&channel->collnetDevPeers, 1, sharedRes->deviceStream.cudaStream));
|
|
addr = (uintptr_t)channel->collnetDevPeers;
|
|
channel->peers[comm->nRanks] = channel->collnetPeers;
|
|
NCCLCHECK(ncclCudaMemcpyAsync((uintptr_t*)(channel->devPeers + comm->nRanks), (uintptr_t*)&addr, 1, sharedRes->deviceStream.cudaStream));
|
|
channel->devPeersHostPtr[comm->nRanks] = (struct ncclDevChannelPeer*)addr;
|
|
ncclAtomicRefCountIncrement(&channel->collnetPeers->refCount);
|
|
}
|
|
|
|
NCCLCHECK(ncclStrongStreamSynchronize(&sharedRes->deviceStream));
|
|
NCCLCHECK(ncclStrongStreamRelease(ncclCudaGraphNone(), &sharedRes->deviceStream));
|
|
|
|
return ncclSuccess;
|
|
}
|
|
|
|
ncclResult_t freeChannel(struct ncclChannel* channel, int nRanks, int collnetNRanks, int nvlsNRanks) {
|
|
int nPeers = nRanks + collnetNRanks + nvlsNRanks;
|
|
/* channel peers are only valid when async init thread completes commAlloc() and
|
|
* the channel is intialized with initChannel(); if either is not done, this channel
|
|
* should never be free. */
|
|
if (channel->id == -1 || channel->peers == NULL) return ncclSuccess;
|
|
|
|
// Free transport proxy resources
|
|
// Note: free all send resources first due to CollNet arrangement
|
|
for (int r = 0; r < nPeers; r++) {
|
|
struct ncclChannelPeer* peer = channel->peers[r];
|
|
if (peer) {
|
|
if (ncclAtomicRefCountDecrement(&peer->refCount) == 0) {
|
|
for (int b=0; b<NCCL_MAX_CONNS; b++) {
|
|
if (peer->send[b].transportComm) NCCLCHECK(peer->send[b].transportComm->free(peer->send+b));
|
|
if (peer->recv[b].transportComm) NCCLCHECK(peer->recv[b].transportComm->free(peer->recv+b));
|
|
}
|
|
if (r == nRanks) {
|
|
free(channel->collnetPeers);
|
|
ncclCudaFree(channel->collnetDevPeers);
|
|
} else if (r == nPeers - 1) {
|
|
free(channel->nvlsPeers);
|
|
ncclCudaFree(channel->nvlsDevPeers);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
free(channel->devPeersHostPtr);
|
|
return ncclSuccess;
|
|
}
|