68b542363f
Add scalable init API * Add new ncclCommInitRankScalable to allow for passing multiple unique IDs to the init function. * Spreads the load onto multiple bootstrap roots, allowing for constant bootstrap time. * Requires multiple ranks to create a unique ID, and the CPU-side ID exchange code to call allgather[v] instead of broadcast. Accelerate init bootstrap operations * Reduce the number of calls to allgather. * Allow roots to reply early to ranks when information is already available. * Add an option to use ncclNet instead of sockets to perform bootstrap allgather operations. Add PAT algorithms for Allgather and ReduceScatter * Parallel Aggregated Trees, variation of Bruck algorithm. * Logarithmic number of network steps for small sizes at scale. * Only supports one rank per node at the moment. Add support for registered buffers for intra-node communication. * Allow registered user buffers to be accessed directly intra-node * Avoids extra copies in algorithms which permit it, saving memory bandwidth and helping with compute overlap. Add profiler plugin API * New plugin API for profiling * Supports various levels of profiling, with a hierarchy. Asynchronous graph allocation * Make calls to cudaMalloc and cudaMemcpy during graph allocation asynchronous. * Significantly speeds up graph capture. Use fatal IB asynchronous events to stop network operation * Avoids many other error messages * Only fatal errors are affected; potentially transient errors (e.g. port down) do not cause an immediate stop. Set P2P level to PXB on AMD CPUs when using more than 2 GPUs per node * P2P would cause a significant performance degradation when using many GPUs, and therefore many interleaved data flows. * Disable P2P through the CPU when we have 3+ GPUs per node; keep it enabled when we only have 2 GPUs. Improve the init logs to report the real NCCL function. * Make the log report ncclCommInitRank or ncclCommSplit, rather than the generic ncclCommInitRankFunc. Add a parameter to set the location of the user configuration file. * Add NCCL_CONF_FILE environment variable to set where the user's configuration file resides. Increase default IB timeout * Increase IB timeout value from 18 to 20. * Should help avoid fatal errors on large RoCE systems. Add new check for nvidia peermem * On linux kernels 6.6+, /sys/kernel/mm/memory_peers is no longer present; check for /sys/module/nvidia_peermem/version instead. Fix old performance regression when mixing small and large operations. * Improves distribution of work on channels. Fix crash when NUMA IDs are equal to -1. * Can happen when a NIC is a virtual NIC, or when linux doesn't know which NUMA node a device is attached to * Issue NVIDIA/nccl-tests#233 Fix tree graph search when NCCL_CROSS_NIC is set to 1. * Would force NCCL to use the balanced_tree pattern, thereby disabling LL128 on platforms with 1 GPU+1 NIC per PCI switch. * Would also try to use alternate rings even though it was not needed. Compiler tweaks and fixes * PR #1177 * PR #1228 Fix stack smash * PR #1325 Fixes for multi-node NVLink + IB operation Coverity fixes and comments.
130 строки
4.3 KiB
C
130 строки
4.3 KiB
C
/*************************************************************************
|
|
* Copyright (c) 2022, NVIDIA CORPORATION. All rights reserved.
|
|
*
|
|
* See LICENSE.txt for license information
|
|
************************************************************************/
|
|
|
|
#ifndef NCCL_CUDAWRAP_H_
|
|
#define NCCL_CUDAWRAP_H_
|
|
|
|
#include <cuda.h>
|
|
#include <cuda_runtime.h>
|
|
#include "checks.h"
|
|
|
|
// Is cuMem API usage enabled
|
|
extern int ncclCuMemEnable();
|
|
extern int ncclCuMemHostEnable();
|
|
|
|
#if CUDART_VERSION >= 11030
|
|
#include <cudaTypedefs.h>
|
|
|
|
// Handle type used for cuMemCreate()
|
|
extern CUmemAllocationHandleType ncclCuMemHandleType;
|
|
|
|
#endif
|
|
|
|
#define CUPFN(symbol) pfn_##symbol
|
|
|
|
// Check CUDA PFN driver calls
|
|
#define CUCHECK(cmd) do { \
|
|
CUresult err = pfn_##cmd; \
|
|
if( err != CUDA_SUCCESS ) { \
|
|
const char *errStr; \
|
|
(void) pfn_cuGetErrorString(err, &errStr); \
|
|
WARN("Cuda failure %d '%s'", err, errStr); \
|
|
return ncclUnhandledCudaError; \
|
|
} \
|
|
} while(false)
|
|
|
|
#define CUCHECKGOTO(cmd, res, label) do { \
|
|
CUresult err = pfn_##cmd; \
|
|
if( err != CUDA_SUCCESS ) { \
|
|
const char *errStr; \
|
|
(void) pfn_cuGetErrorString(err, &errStr); \
|
|
WARN("Cuda failure %d '%s'", err, errStr); \
|
|
res = ncclUnhandledCudaError; \
|
|
goto label; \
|
|
} \
|
|
} while(false)
|
|
|
|
// Report failure but clear error and continue
|
|
#define CUCHECKIGNORE(cmd) do { \
|
|
CUresult err = pfn_##cmd; \
|
|
if( err != CUDA_SUCCESS ) { \
|
|
const char *errStr; \
|
|
(void) pfn_cuGetErrorString(err, &errStr); \
|
|
INFO(NCCL_ALL,"%s:%d Cuda failure %d '%s'", __FILE__, __LINE__, err, errStr); \
|
|
} \
|
|
} while(false)
|
|
|
|
#define CUCHECKTHREAD(cmd, args) do { \
|
|
CUresult err = pfn_##cmd; \
|
|
if (err != CUDA_SUCCESS) { \
|
|
INFO(NCCL_INIT,"%s:%d -> %d [Async thread]", __FILE__, __LINE__, err); \
|
|
args->ret = ncclUnhandledCudaError; \
|
|
return args; \
|
|
} \
|
|
} while(0)
|
|
|
|
#define DECLARE_CUDA_PFN_EXTERN(symbol) extern PFN_##symbol pfn_##symbol
|
|
|
|
#if CUDART_VERSION >= 11030
|
|
/* CUDA Driver functions loaded with cuGetProcAddress for versioning */
|
|
DECLARE_CUDA_PFN_EXTERN(cuDeviceGet);
|
|
DECLARE_CUDA_PFN_EXTERN(cuDeviceGetAttribute);
|
|
DECLARE_CUDA_PFN_EXTERN(cuGetErrorString);
|
|
DECLARE_CUDA_PFN_EXTERN(cuGetErrorName);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemGetAddressRange);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxCreate);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxDestroy);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxGetCurrent);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxSetCurrent);
|
|
DECLARE_CUDA_PFN_EXTERN(cuCtxGetDevice);
|
|
DECLARE_CUDA_PFN_EXTERN(cuPointerGetAttribute);
|
|
DECLARE_CUDA_PFN_EXTERN(cuLaunchKernel);
|
|
#if CUDART_VERSION >= 11080
|
|
DECLARE_CUDA_PFN_EXTERN(cuLaunchKernelEx);
|
|
#endif
|
|
// cuMem API support
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemAddressReserve);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemAddressFree);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemCreate);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemGetAllocationGranularity);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemExportToShareableHandle);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemImportFromShareableHandle);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemMap);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemRelease);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemRetainAllocationHandle);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemSetAccess);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemUnmap);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemGetAllocationPropertiesFromHandle);
|
|
#if CUDA_VERSION >= 11070
|
|
DECLARE_CUDA_PFN_EXTERN(cuMemGetHandleForAddressRange); // DMA-BUF support
|
|
#endif
|
|
#if CUDA_VERSION >= 12010
|
|
/* NVSwitch Multicast support */
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastAddDevice);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastBindMem);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastBindAddr);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastCreate);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastGetGranularity);
|
|
DECLARE_CUDA_PFN_EXTERN(cuMulticastUnbind);
|
|
#endif
|
|
#endif
|
|
|
|
ncclResult_t ncclCudaLibraryInit(void);
|
|
|
|
extern int ncclCudaDriverVersionCache;
|
|
extern bool ncclCudaLaunchBlocking; // initialized by ncclCudaLibraryInit()
|
|
|
|
inline ncclResult_t ncclCudaDriverVersion(int* driver) {
|
|
int version = __atomic_load_n(&ncclCudaDriverVersionCache, __ATOMIC_RELAXED);
|
|
if (version == -1) {
|
|
CUDACHECK(cudaDriverGetVersion(&version));
|
|
__atomic_store_n(&ncclCudaDriverVersionCache, version, __ATOMIC_RELAXED);
|
|
}
|
|
*driver = version;
|
|
return ncclSuccess;
|
|
}
|
|
#endif
|