60240fec77
Add scalable init API
* Add new ncclCommInitRankScalable to allow for passing multiple
unique IDs to the init function.
* Spreads the load onto multiple bootstrap roots, allowing for
constant bootstrap time.
* Requires multiple ranks to create a unique ID, and the CPU-side
ID exchange code to call allgather[v] instead of broadcast.
Accelerate init bootstrap operations
* Reduce the number of calls to allgather.
* Allow roots to reply early to ranks when information is already
available.
* Add an option to use ncclNet instead of sockets to perform
bootstrap allgather operations.
Add PAT algorithms for Allgather and ReduceScatter
* Parallel Aggregated Trees, variation of Bruck algorithm.
* Logarithmic number of network steps for small sizes at scale.
* Only supports one rank per node at the moment.
Add support for registered buffers for intra-node communication.
* Allow registered user buffers to be accessed directly intra-node
* Avoids extra copies in algorithms which permit it, saving
memory bandwidth and helping with compute overlap.
Add profiler plugin API
* New plugin API for profiling
* Supports various levels of profiling, with a hierarchy.
Asynchronous graph allocation
* Make calls to cudaMalloc and cudaMemcpy during graph allocation
asynchronous.
* Significantly speeds up graph capture.
Use fatal IB asynchronous events to stop network operation
* Avoids many other error messages
* Only fatal errors are affected; potentially transient errors
(e.g. port down) do not cause an immediate stop.
Set P2P level to PXB on AMD CPUs when using more than 2 GPUs per node
* P2P would cause a significant performance degradation when using
many GPUs, and therefore many interleaved data flows.
* Disable P2P through the CPU when we have 3+ GPUs per node; keep it
enabled when we only have 2 GPUs.
Improve the init logs to report the real NCCL function.
* Make the log report ncclCommInitRank or ncclCommSplit, rather than
the generic ncclCommInitRankFunc.
Add a parameter to set the location of the user configuration file.
* Add NCCL_CONF_FILE environment variable to set where the user's
configuration file resides.
Increase default IB timeout
* Increase IB timeout value from 18 to 20.
* Should help avoid fatal errors on large RoCE systems.
Add new check for nvidia peermem
* On linux kernels 6.6+, /sys/kernel/mm/memory_peers is no longer
present; check for /sys/module/nvidia_peermem/version instead.
Fix old performance regression when mixing small and large operations.
* Improves distribution of work on channels.
Fix crash when NUMA IDs are equal to -1.
* Can happen when a NIC is a virtual NIC, or when linux doesn't
know which NUMA node a device is attached to
* Issue NVIDIA/nccl-tests#233
Fix tree graph search when NCCL_CROSS_NIC is set to 1.
* Would force NCCL to use the balanced_tree pattern, thereby
disabling LL128 on platforms with 1 GPU+1 NIC per PCI switch.
* Would also try to use alternate rings even though it was not
needed.
Compiler tweaks and fixes
* PR #1177
* PR #1228
Fix stack smash
* PR #1325
Fixes for multi-node NVLink + IB operation
Coverity fixes and comments.
[ROCm/rccl commit: 68b542363f]
130 lines
4.3 KiB
C
130 lines
4.3 KiB
C
/*************************************************************************
|
|
* Copyright (c) 2022, NVIDIA CORPORATION. All rights reserved.
|
|
*
|
|
* See LICENSE.txt for license information
|
|
************************************************************************/
|
|
|
|
#ifndef NCCL_CUDAWRAP_H_
|
|
#define NCCL_CUDAWRAP_H_
|
|
|
|
#include <cuda.h>
|
|
#include <cuda_runtime.h>
|
|
#include "checks.h"
|
|
|
|
// Is cuMem API usage enabled
|
|
extern int ncclCuMemEnable();
|
|
extern int ncclCuMemHostEnable();
|
|
|
|
#if CUDART_VERSION >= 11030
|
|
#include <cudaTypedefs.h>
|
|
|
|
// Handle type used for cuMemCreate()
|
|
extern CUmemAllocationHandleType ncclCuMemHandleType;
|
|
|
|
#endif
|
|
|
|
#define CUPFN(symbol) pfn_##symbol
|
|
|
|
// Check CUDA PFN driver calls
|
|
#define CUCHECK(cmd) do { \
|
|
CUresult err = pfn_##cmd; \
|
|
if( err != CUDA_SUCCESS ) { \
|
|
const char *errStr; \
|
|
(void) pfn_cuGetErrorString(err, &errStr); \
|
|
WARN("Cuda failure %d '%s'", err, errStr); \
|
|
return ncclUnhandledCudaError; \
|
|
} \
|
|
} while(false)
|
|
|
|
#define CUCHECKGOTO(cmd, res, label) do { \
|
|
CUresult err = pfn_##cmd; \
|
|
if( err != CUDA_SUCCESS ) { \
|
|
const char *errStr; \
|
|
(void) pfn_cuGetErrorString(err, &errStr); \
|
|
WARN("Cuda failure %d '%s'", err, errStr); \
|
|
res = ncclUnhandledCudaError; \
|
|
goto label; \
|
|
} \
|
|
} while(false)
|
|
|
|
// Report failure but clear error and continue
|
|
#define CUCHECKIGNORE(cmd) do { \
|
|
CUresult err = pfn_##cmd; \
|
|
if( err != CUDA_SUCCESS ) { \
|
|
const char *errStr; \
|
|
(void) pfn_cuGetErrorString(err, &errStr); \
|
|
INFO(NCCL_ALL,"%s:%d Cuda failure %d '%s'", __FILE__, __LINE__, err, errStr); \
|
|
} \
|
|
} while(false)
|
|
|
|
#define CUCHECKTHREAD(cmd, args) do { \
|
|
CUresult err = pfn_##cmd; \
|
|
if (err != CUDA_SUCCESS) { \
|
|
INFO(NCCL_INIT,"%s:%d -> %d [Async thread]", __FILE__, __LINE__, err); \
|
|
args->ret = ncclUnhandledCudaError; \
|
|
return args; \
|
|
} \
|
|
} while(0)
|
|
|
|
#define DECLARE_CUDA_PFN_EXTERN(symbol) extern PFN_##symbol pfn_##symbol
|
|
|
|
#if CUDART_VERSION >= 11030
|
|
/* CUDA Driver functions loaded with cuGetProcAddress for versioning */
|
|
DECLARE_CUDA_PFN_EXTERN(cuDeviceGet);
|
|
DECLARE_CUDA_PFN_EXTERN(cuDeviceGetAttribute);
|
|
DECLARE_CUDA_PFN_EXTERN(cuGetErrorString);
|
|
DECLARE_CUDA_PFN_EXTERN(cuGetErrorName);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemGetAddressRange);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxCreate);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxDestroy);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxGetCurrent);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxSetCurrent);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxGetDevice);
|
|
DECLARE_CUDA_PFN_EXTERN(cuPointerGetAttribute);
|
|
DECLARE_CUDA_PFN_EXTERN(cuLaunchKernel);
|
|
#if CUDART_VERSION >= 11080
|
|
DECLARE_CUDA_PFN_EXTERN(cuLaunchKernelEx);
|
|
#endif
|
|
// cuMem API support
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemAddressReserve);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemAddressFree);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemCreate);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemGetAllocationGranularity);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemExportToShareableHandle);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemImportFromShareableHandle);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemMap);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemRelease);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemRetainAllocationHandle);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemSetAccess);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemUnmap);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemGetAllocationPropertiesFromHandle);
|
|
#if CUDA_VERSION >= 11070
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemGetHandleForAddressRange); // DMA-BUF support
|
|
#endif
|
|
#if CUDA_VERSION >= 12010
|
|
/* NVSwitch Multicast support */
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastAddDevice);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastBindMem);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastBindAddr);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastCreate);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastGetGranularity);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastUnbind);
|
|
#endif
|
|
#endif
|
|
|
|
ncclResult_t ncclCudaLibraryInit(void);
|
|
|
|
extern int ncclCudaDriverVersionCache;
|
|
extern bool ncclCudaLaunchBlocking; // initialized by ncclCudaLibraryInit()
|
|
|
|
inline ncclResult_t ncclCudaDriverVersion(int* driver) {
|
|
int version = __atomic_load_n(&ncclCudaDriverVersionCache, __ATOMIC_RELAXED);
|
|
if (version == -1) {
|
|
CUDACHECK(cudaDriverGetVersion(&version));
|
|
__atomic_store_n(&ncclCudaDriverVersionCache, version, __ATOMIC_RELAXED);
|
|
}
|
|
*driver = version;
|
|
return ncclSuccess;
|
|
}
|
|
#endif
|