60240fec77
Add scalable init API
* Add new ncclCommInitRankScalable to allow for passing multiple
unique IDs to the init function.
* Spreads the load onto multiple bootstrap roots, allowing for
constant bootstrap time.
* Requires multiple ranks to create a unique ID, and the CPU-side
ID exchange code to call allgather[v] instead of broadcast.
Accelerate init bootstrap operations
* Reduce the number of calls to allgather.
* Allow roots to reply early to ranks when information is already
available.
* Add an option to use ncclNet instead of sockets to perform
bootstrap allgather operations.
Add PAT algorithms for Allgather and ReduceScatter
* Parallel Aggregated Trees, variation of Bruck algorithm.
* Logarithmic number of network steps for small sizes at scale.
* Only supports one rank per node at the moment.
Add support for registered buffers for intra-node communication.
* Allow registered user buffers to be accessed directly intra-node
* Avoids extra copies in algorithms which permit it, saving
memory bandwidth and helping with compute overlap.
Add profiler plugin API
* New plugin API for profiling
* Supports various levels of profiling, with a hierarchy.
Asynchronous graph allocation
* Make calls to cudaMalloc and cudaMemcpy during graph allocation
asynchronous.
* Significantly speeds up graph capture.
Use fatal IB asynchronous events to stop network operation
* Avoids many other error messages
* Only fatal errors are affected; potentially transient errors
(e.g. port down) do not cause an immediate stop.
Set P2P level to PXB on AMD CPUs when using more than 2 GPUs per node
* P2P would cause a significant performance degradation when using
many GPUs, and therefore many interleaved data flows.
* Disable P2P through the CPU when we have 3+ GPUs per node; keep it
enabled when we only have 2 GPUs.
Improve the init logs to report the real NCCL function.
* Make the log report ncclCommInitRank or ncclCommSplit, rather than
the generic ncclCommInitRankFunc.
Add a parameter to set the location of the user configuration file.
* Add NCCL_CONF_FILE environment variable to set where the user's
configuration file resides.
Increase default IB timeout
* Increase IB timeout value from 18 to 20.
* Should help avoid fatal errors on large RoCE systems.
Add new check for nvidia peermem
* On linux kernels 6.6+, /sys/kernel/mm/memory_peers is no longer
present; check for /sys/module/nvidia_peermem/version instead.
Fix old performance regression when mixing small and large operations.
* Improves distribution of work on channels.
Fix crash when NUMA IDs are equal to -1.
* Can happen when a NIC is a virtual NIC, or when linux doesn't
know which NUMA node a device is attached to
* Issue NVIDIA/nccl-tests#233
Fix tree graph search when NCCL_CROSS_NIC is set to 1.
* Would force NCCL to use the balanced_tree pattern, thereby
disabling LL128 on platforms with 1 GPU+1 NIC per PCI switch.
* Would also try to use alternate rings even though it was not
needed.
Compiler tweaks and fixes
* PR #1177
* PR #1228
Fix stack smash
* PR #1325
Fixes for multi-node NVLink + IB operation
Coverity fixes and comments.
[ROCm/rccl commit: 68b542363f]
151 regels
4.7 KiB
C
151 regels
4.7 KiB
C
/*************************************************************************
|
|
* Copyright (c) 2024, NVIDIA CORPORATION. All rights reserved.
|
|
*
|
|
* See LICENSE.txt for license information
|
|
************************************************************************/
|
|
|
|
#ifndef NCCL_PROFILER_V1_H_
|
|
#define NCCL_PROFILER_V1_H_
|
|
|
|
#include <stdint.h>
|
|
|
|
enum {
|
|
ncclProfileGroup = (1 << 0), // group event type
|
|
ncclProfileColl = (1 << 1), // host collective call event type
|
|
ncclProfileP2p = (1 << 2), // host point-to-point call event type
|
|
ncclProfileProxyOp = (1 << 3), // proxy operation event type
|
|
ncclProfileProxyStep = (1 << 4), // proxy step event type
|
|
ncclProfileProxyCtrl = (1 << 5), // proxy control event type
|
|
ncclProfileNumEvents = ( 6),
|
|
};
|
|
|
|
typedef struct {
|
|
uint8_t type; // event type descriptor: ncclProfileColl, ...
|
|
void* parentObj; // pointer to the profiler parent object (for coll is the group)
|
|
int rank; // originating rank
|
|
union {
|
|
struct {
|
|
const char* name;
|
|
uint64_t commHash;
|
|
uint64_t seqNumber;
|
|
uint8_t func;
|
|
void const* sendBuff;
|
|
void* recvBuff;
|
|
size_t count;
|
|
int root;
|
|
uint8_t datatype;
|
|
uint32_t op;
|
|
size_t trafficBytes;
|
|
uint8_t nMaxChannels;
|
|
uint8_t nWarps;
|
|
uint8_t algo;
|
|
uint8_t proto;
|
|
int isCollnet;
|
|
int isNvls;
|
|
} coll;
|
|
|
|
struct {
|
|
const char* name;
|
|
uint64_t commHash;
|
|
uint8_t func;
|
|
void* buff;
|
|
uint8_t datatype;
|
|
size_t count;
|
|
int peer;
|
|
} p2p;
|
|
|
|
struct {
|
|
pid_t pid; // pid of the originating process
|
|
uint8_t channelId; // channel id for this proxy operation
|
|
int peer; // remote rank for send/recv
|
|
int nSteps; // number of steps for this proxy operation
|
|
int chunkSize; // amount of data transferred by this proxy operation
|
|
int isSend;
|
|
} proxyOp;
|
|
|
|
struct {
|
|
int step;
|
|
} proxyStep;
|
|
};
|
|
} ncclProfilerEventDescr_v1_t;
|
|
|
|
typedef enum {
|
|
ncclProfilerProxyOpSendPosted,
|
|
ncclProfilerProxyOpSendRemFifoWait,
|
|
ncclProfilerProxyOpSendTransmitted,
|
|
ncclProfilerProxyOpSendDone,
|
|
ncclProfilerProxyOpRecvPosted,
|
|
ncclProfilerProxyOpRecvReceived,
|
|
ncclProfilerProxyOpRecvTransmitted,
|
|
ncclProfilerProxyOpRecvDone,
|
|
|
|
/* Legacy proxy profiler states */
|
|
ncclProfilerProxyStepSendGPUWait,
|
|
ncclProfilerProxyStepSendWait,
|
|
ncclProfilerProxyStepRecvWait,
|
|
ncclProfilerProxyStepRecvFlushWait,
|
|
ncclProfilerProxyStepRecvGPUWait,
|
|
|
|
/* Legacy proxy control states */
|
|
ncclProfilerProxyCtrlIdle,
|
|
ncclProfilerProxyCtrlActive,
|
|
ncclProfilerProxyCtrlSleep,
|
|
ncclProfilerProxyCtrlWakeup,
|
|
ncclProfilerProxyCtrlAppend,
|
|
ncclProfilerProxyCtrlAppendEnd,
|
|
} ncclProfilerEventState_v1_t;
|
|
|
|
typedef union {
|
|
struct {
|
|
size_t transSize;
|
|
int steps;
|
|
} proxyOp;
|
|
|
|
struct {
|
|
int appendedProxyOps;
|
|
} proxyCtrl;
|
|
} ncclProfilerEventStateArgs_v1_t;
|
|
|
|
typedef struct {
|
|
const char* name;
|
|
|
|
// init - initialize the profiler plugin
|
|
// Input
|
|
// - context : opaque profiler context object for separating profiler behavior across comms
|
|
// Output
|
|
// - eActivationMask: bitmask of active events set by the plugin
|
|
ncclResult_t (*init)(void** context, int* eActivationMask);
|
|
|
|
// startEvent - initialize and start a new event for the supplied event descriptor inside the eventset
|
|
// Input
|
|
// - context: opaque profiler context object
|
|
// - eDescr : pointer to ncclProfilerEventDescr_t object
|
|
// Output
|
|
// - eHandle: return event handle for supplied event descriptor object
|
|
ncclResult_t (*startEvent)(void* context, void** eHandle, ncclProfilerEventDescr_v1_t* eDescr);
|
|
|
|
// stopEvent - stop/finalize an event inside and event set
|
|
// Input
|
|
// - eHandle: handle to event object
|
|
ncclResult_t (*stopEvent)(void* eHandle);
|
|
|
|
// recordEventState - record event state transitions and event attribute updates
|
|
// Input
|
|
// - eHandle : handle to event object created through startEvent
|
|
// - eStateArgs: optional argument used to capture event attribute updates associated with the state transition
|
|
// - eState : event state transition
|
|
ncclResult_t (*recordEventState)(void* eHandle, ncclProfilerEventState_v1_t eState, ncclProfilerEventStateArgs_v1_t* eStateArgs);
|
|
|
|
// finalize - finalize the profiler plugin
|
|
// Input
|
|
// - context: opaque profiler context object
|
|
ncclResult_t (*finalize)(void* context);
|
|
} ncclProfiler_v1_t;
|
|
|
|
typedef ncclProfilerEventDescr_v1_t ncclProfilerEventDescr_t;
|
|
typedef ncclProfilerEventState_v1_t ncclProfilerEventState_t;
|
|
typedef ncclProfilerEventStateArgs_v1_t ncclProfilerEventStateArgs_t;
|
|
typedef ncclProfiler_v1_t ncclProfiler_t;
|
|
|
|
#endif
|