60240fec77
Add scalable init API
* Add new ncclCommInitRankScalable to allow for passing multiple
unique IDs to the init function.
* Spreads the load onto multiple bootstrap roots, allowing for
constant bootstrap time.
* Requires multiple ranks to create a unique ID, and the CPU-side
ID exchange code to call allgather[v] instead of broadcast.
Accelerate init bootstrap operations
* Reduce the number of calls to allgather.
* Allow roots to reply early to ranks when information is already
available.
* Add an option to use ncclNet instead of sockets to perform
bootstrap allgather operations.
Add PAT algorithms for Allgather and ReduceScatter
* Parallel Aggregated Trees, variation of Bruck algorithm.
* Logarithmic number of network steps for small sizes at scale.
* Only supports one rank per node at the moment.
Add support for registered buffers for intra-node communication.
* Allow registered user buffers to be accessed directly intra-node
* Avoids extra copies in algorithms which permit it, saving
memory bandwidth and helping with compute overlap.
Add profiler plugin API
* New plugin API for profiling
* Supports various levels of profiling, with a hierarchy.
Asynchronous graph allocation
* Make calls to cudaMalloc and cudaMemcpy during graph allocation
asynchronous.
* Significantly speeds up graph capture.
Use fatal IB asynchronous events to stop network operation
* Avoids many other error messages
* Only fatal errors are affected; potentially transient errors
(e.g. port down) do not cause an immediate stop.
Set P2P level to PXB on AMD CPUs when using more than 2 GPUs per node
* P2P would cause a significant performance degradation when using
many GPUs, and therefore many interleaved data flows.
* Disable P2P through the CPU when we have 3+ GPUs per node; keep it
enabled when we only have 2 GPUs.
Improve the init logs to report the real NCCL function.
* Make the log report ncclCommInitRank or ncclCommSplit, rather than
the generic ncclCommInitRankFunc.
Add a parameter to set the location of the user configuration file.
* Add NCCL_CONF_FILE environment variable to set where the user's
configuration file resides.
Increase default IB timeout
* Increase IB timeout value from 18 to 20.
* Should help avoid fatal errors on large RoCE systems.
Add new check for nvidia peermem
* On linux kernels 6.6+, /sys/kernel/mm/memory_peers is no longer
present; check for /sys/module/nvidia_peermem/version instead.
Fix old performance regression when mixing small and large operations.
* Improves distribution of work on channels.
Fix crash when NUMA IDs are equal to -1.
* Can happen when a NIC is a virtual NIC, or when linux doesn't
know which NUMA node a device is attached to
* Issue NVIDIA/nccl-tests#233
Fix tree graph search when NCCL_CROSS_NIC is set to 1.
* Would force NCCL to use the balanced_tree pattern, thereby
disabling LL128 on platforms with 1 GPU+1 NIC per PCI switch.
* Would also try to use alternate rings even though it was not
needed.
Compiler tweaks and fixes
* PR #1177
* PR #1228
Fix stack smash
* PR #1325
Fixes for multi-node NVLink + IB operation
Coverity fixes and comments.
[ROCm/rccl commit: 68b542363f]
176 řádky
6.0 KiB
C
176 řádky
6.0 KiB
C
/*************************************************************************
|
|
* Copyright (c) 2019-2022, NVIDIA CORPORATION. All rights reserved.
|
|
*
|
|
* See LICENSE.txt for license information
|
|
************************************************************************/
|
|
|
|
#ifndef NCCL_CHECKS_H_
|
|
#define NCCL_CHECKS_H_
|
|
|
|
#include "debug.h"
|
|
|
|
// Check CUDA RT calls
|
|
#define CUDACHECK(cmd) do { \
|
|
cudaError_t err = cmd; \
|
|
if( err != cudaSuccess ) { \
|
|
WARN("Cuda failure '%s'", cudaGetErrorString(err)); \
|
|
return ncclUnhandledCudaError; \
|
|
} \
|
|
} while(false)
|
|
|
|
#define CUDACHECKGOTO(cmd, RES, label) do { \
|
|
cudaError_t err = cmd; \
|
|
if( err != cudaSuccess ) { \
|
|
WARN("Cuda failure '%s'", cudaGetErrorString(err)); \
|
|
RES = ncclUnhandledCudaError; \
|
|
goto label; \
|
|
} \
|
|
} while(false)
|
|
|
|
// Report failure but clear error and continue
|
|
#define CUDACHECKIGNORE(cmd) do { \
|
|
cudaError_t err = cmd; \
|
|
if( err != cudaSuccess ) { \
|
|
INFO(NCCL_ALL,"%s:%d Cuda failure '%s'", __FILE__, __LINE__, cudaGetErrorString(err)); \
|
|
(void) cudaGetLastError(); \
|
|
} \
|
|
} while(false)
|
|
|
|
#include <errno.h>
|
|
// Check system calls
|
|
#define SYSCHECK(statement, name) do { \
|
|
int retval; \
|
|
SYSCHECKSYNC((statement), name, retval); \
|
|
if (retval == -1) { \
|
|
WARN("Call to " name " failed: %s", strerror(errno)); \
|
|
return ncclSystemError; \
|
|
} \
|
|
} while (false)
|
|
|
|
#define SYSCHECKSYNC(statement, name, retval) do { \
|
|
retval = (statement); \
|
|
if (retval == -1 && (errno == EINTR || errno == EWOULDBLOCK || errno == EAGAIN)) { \
|
|
INFO(NCCL_ALL,"Call to " name " returned %s, retrying", strerror(errno)); \
|
|
} else { \
|
|
break; \
|
|
} \
|
|
} while(true)
|
|
|
|
#define SYSCHECKGOTO(statement, name, RES, label) do { \
|
|
int retval; \
|
|
SYSCHECKSYNC((statement), name, retval); \
|
|
if (retval == -1) { \
|
|
WARN("Call to " name " failed: %s", strerror(errno)); \
|
|
RES = ncclSystemError; \
|
|
goto label; \
|
|
} \
|
|
} while (0)
|
|
|
|
// Pthread calls don't set errno and never return EINTR.
|
|
#define PTHREADCHECK(statement, name) do { \
|
|
int retval = (statement); \
|
|
if (retval != 0) { \
|
|
WARN("Call to " name " failed: %s", strerror(retval)); \
|
|
return ncclSystemError; \
|
|
} \
|
|
} while (0)
|
|
|
|
#define PTHREADCHECKGOTO(statement, name, RES, label) do { \
|
|
int retval = (statement); \
|
|
if (retval != 0) { \
|
|
WARN("Call to " name " failed: %s", strerror(retval)); \
|
|
RES = ncclSystemError; \
|
|
goto label; \
|
|
} \
|
|
} while (0)
|
|
|
|
#define NEQCHECK(statement, value) do { \
|
|
if ((statement) != value) { \
|
|
/* Print the back trace*/ \
|
|
INFO(NCCL_ALL,"%s:%d -> %d (%s)", __FILE__, __LINE__, ncclSystemError, strerror(errno)); \
|
|
return ncclSystemError; \
|
|
} \
|
|
} while (0)
|
|
|
|
#define NEQCHECKGOTO(statement, value, RES, label) do { \
|
|
if ((statement) != value) { \
|
|
/* Print the back trace*/ \
|
|
RES = ncclSystemError; \
|
|
INFO(NCCL_ALL,"%s:%d -> %d (%s)", __FILE__, __LINE__, RES, strerror(errno)); \
|
|
goto label; \
|
|
} \
|
|
} while (0)
|
|
|
|
#define EQCHECK(statement, value) do { \
|
|
if ((statement) == value) { \
|
|
/* Print the back trace*/ \
|
|
INFO(NCCL_ALL,"%s:%d -> %d (%s)", __FILE__, __LINE__, ncclSystemError, strerror(errno)); \
|
|
return ncclSystemError; \
|
|
} \
|
|
} while (0)
|
|
|
|
#define EQCHECKGOTO(statement, value, RES, label) do { \
|
|
if ((statement) == value) { \
|
|
/* Print the back trace*/ \
|
|
RES = ncclSystemError; \
|
|
INFO(NCCL_ALL,"%s:%d -> %d (%s)", __FILE__, __LINE__, RES, strerror(errno)); \
|
|
goto label; \
|
|
} \
|
|
} while (0)
|
|
|
|
// Propagate errors up
|
|
#define NCCLCHECK(call) do { \
|
|
ncclResult_t RES = call; \
|
|
if (RES != ncclSuccess && RES != ncclInProgress) { \
|
|
/* Print the back trace*/ \
|
|
if (ncclDebugNoWarn == 0) INFO(NCCL_ALL,"%s:%d -> %d", __FILE__, __LINE__, RES); \
|
|
return RES; \
|
|
} \
|
|
} while (0)
|
|
|
|
#define NCCLCHECKGOTO(call, RES, label) do { \
|
|
RES = call; \
|
|
if (RES != ncclSuccess && RES != ncclInProgress) { \
|
|
/* Print the back trace*/ \
|
|
if (ncclDebugNoWarn == 0) INFO(NCCL_ALL,"%s:%d -> %d", __FILE__, __LINE__, RES); \
|
|
goto label; \
|
|
} \
|
|
} while (0)
|
|
|
|
#define NCCLWAIT(call, cond, abortFlagPtr) do { \
|
|
uint32_t* tmpAbortFlag = (abortFlagPtr); \
|
|
ncclResult_t RES = call; \
|
|
if (RES != ncclSuccess && RES != ncclInProgress) { \
|
|
if (ncclDebugNoWarn == 0) INFO(NCCL_ALL,"%s:%d -> %d", __FILE__, __LINE__, RES); \
|
|
return ncclInternalError; \
|
|
} \
|
|
if (__atomic_load(tmpAbortFlag, __ATOMIC_ACQUIRE)) NEQCHECK(*tmpAbortFlag, 0); \
|
|
} while (!(cond))
|
|
|
|
#define NCCLWAITGOTO(call, cond, abortFlagPtr, RES, label) do { \
|
|
uint32_t* tmpAbortFlag = (abortFlagPtr); \
|
|
RES = call; \
|
|
if (RES != ncclSuccess && RES != ncclInProgress) { \
|
|
if (ncclDebugNoWarn == 0) INFO(NCCL_ALL,"%s:%d -> %d", __FILE__, __LINE__, RES); \
|
|
goto label; \
|
|
} \
|
|
if (__atomic_load(tmpAbortFlag, __ATOMIC_ACQUIRE)) NEQCHECKGOTO(*tmpAbortFlag, 0, RES, label); \
|
|
} while (!(cond))
|
|
|
|
#define NCCLCHECKTHREAD(a, args) do { \
|
|
if (((args)->ret = (a)) != ncclSuccess && (args)->ret != ncclInProgress) { \
|
|
INFO(NCCL_INIT,"%s:%d -> %d [Async thread]", __FILE__, __LINE__, (args)->ret); \
|
|
return args; \
|
|
} \
|
|
} while(0)
|
|
|
|
#define CUDACHECKTHREAD(a) do { \
|
|
if ((a) != cudaSuccess) { \
|
|
INFO(NCCL_INIT,"%s:%d -> %d [Async thread]", __FILE__, __LINE__, args->ret); \
|
|
args->ret = ncclUnhandledCudaError; \
|
|
return args; \
|
|
} \
|
|
} while(0)
|
|
|
|
#endif
|