2.5.6-1 (#255)
Add LL128 Protocol.
Rewrite the topology detection and tree/ring creation (#179). Improve
tree performance by sending/receiving from different GPUs. Add
model-based tuning to switch between the different algorithms and
protocols.
Rework P2P/SHM detection in containers (#155, #248).
Detect duplicated devices and return an error (#231).
Add tuning for GCP
[ROCm/rccl commit: 299c554dcc]
Este cometimento está contido em:
cometido por
GitHub
ascendente
221b65bee1
cometimento
71560fd67b
@@ -7,14 +7,9 @@
|
||||
#ifndef NCCL_ENQUEUE_H_
|
||||
#define NCCL_ENQUEUE_H_
|
||||
|
||||
#include "core.h"
|
||||
#include "comm.h"
|
||||
#include "group.h"
|
||||
|
||||
// Channels / LL tuning
|
||||
#define NCCL_LL_CHANNEL_THRESHOLD 8 // Per thread size before we start increasing nrings
|
||||
#define NCCL_THREAD_THRESHOLD 64 // Per thread size before we switch to non-LL
|
||||
#define NCCL_THREAD_THRESHOLD_PREVOLTA 32 // Per thread size before we switch to non-LL for pre-Volta archs
|
||||
#define NCCL_LL_MIN_NTHREADS 64
|
||||
#include "collectives.h"
|
||||
|
||||
ncclResult_t ncclEnqueueCheck(struct ncclInfo* info);
|
||||
ncclResult_t ncclCpuBarrierIn(ncclComm_t comm, int* isLast);
|
||||
|
||||
Criar uma nova questão referindo esta
Bloquear um utilizador