2.3.5-5
Add support for inter-node communication using sockets and InfiniBand/RoCE.
Improve latency.
Add support for aggregation.
Improve LL/regular tuning.
Remove tests as those are now at github.com/nvidia/nccl-tests .
[ROCm/rccl commit: f93fe9bfd9]
Bu işleme şunda yer alıyor:
@@ -0,0 +1,248 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2017-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include "enqueue.h"
|
||||
#include "common_coll.h"
|
||||
#include "param.h"
|
||||
|
||||
#include "collectives/collectives.h"
|
||||
|
||||
#define NCCL_FUNC4(coll, op, dtype) \
|
||||
(void*)NCCL_KERN_NAME(coll, op, dtype), \
|
||||
(void*)NCCL_KERN_NAME(coll##LL, op, dtype)
|
||||
|
||||
// Must be consistent with ncclDataType_t
|
||||
#define NCCL_FUNCS3A(coll, op) \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, u8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i32), \
|
||||
(void*)NCCL_FUNC4(coll, op, u32), \
|
||||
(void*)NCCL_FUNC4(coll, op, i64), \
|
||||
(void*)NCCL_FUNC4(coll, op, u64), \
|
||||
(void*)NCCL_FUNC4(coll, op, f16), \
|
||||
(void*)NCCL_FUNC4(coll, op, f32), \
|
||||
(void*)NCCL_FUNC4(coll, op, f64)
|
||||
#define NCCL_FUNCS3B(coll, op) \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i8), \
|
||||
(void*)NCCL_FUNC4(coll, op, i8)
|
||||
|
||||
// Must be consistent with ncclRedOp_t
|
||||
#define NCCL_FUNCS2A(coll) \
|
||||
NCCL_FUNCS3A(coll, sum ), \
|
||||
NCCL_FUNCS3A(coll, prod), \
|
||||
NCCL_FUNCS3A(coll, max ), \
|
||||
NCCL_FUNCS3A(coll, min )
|
||||
#define NCCL_FUNCS2B(coll) \
|
||||
NCCL_FUNCS3B(coll, copy), \
|
||||
NCCL_FUNCS3B(coll, copy), \
|
||||
NCCL_FUNCS3B(coll, copy), \
|
||||
NCCL_FUNCS3B(coll, copy)
|
||||
|
||||
// Must be consistent with the ncclFuncSet enum
|
||||
static void* const ncclKerns[ncclCollCount*ncclNumOps*ncclNumTypes*2] = {
|
||||
NCCL_FUNCS2B(ncclBroadcast),
|
||||
NCCL_FUNCS2A(ncclReduce),
|
||||
NCCL_FUNCS2B(ncclAllGather),
|
||||
NCCL_FUNCS2A(ncclReduceScatter),
|
||||
NCCL_FUNCS2A(ncclAllReduce)
|
||||
};
|
||||
|
||||
ncclResult_t ncclLaunchCooperativeKernelMultiDevice(struct cudaLaunchParams *paramsList, int* cudaDevs, int numDevices, int cgMode) {
|
||||
#if __CUDACC_VER_MAJOR__ >= 9
|
||||
if (cgMode & 0x01) {
|
||||
CUDACHECK(cudaLaunchCooperativeKernelMultiDevice(paramsList, numDevices,
|
||||
// These flags are to reduce the latency of using this API
|
||||
cudaCooperativeLaunchMultiDeviceNoPreSync|cudaCooperativeLaunchMultiDeviceNoPostSync));
|
||||
return ncclSuccess;
|
||||
}
|
||||
#endif
|
||||
int savedDev;
|
||||
CUDACHECK(cudaGetDevice(&savedDev));
|
||||
for (int i = 0; i < numDevices; i++) {
|
||||
struct cudaLaunchParams* params = paramsList+i;
|
||||
CUDACHECK(cudaSetDevice(cudaDevs[i]));
|
||||
CUDACHECK(cudaLaunchKernel(params->func, params->gridDim, params->blockDim, params->args, params->sharedMem, params->stream));
|
||||
}
|
||||
CUDACHECK(cudaSetDevice(savedDev));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t setupLaunch(struct ncclComm* comm, struct cudaLaunchParams* params) {
|
||||
params->gridDim.x = std::min((int) params->gridDim.x, comm->nRings);
|
||||
|
||||
// Set active = 2 for the last operation
|
||||
for (int r=0; r<params->gridDim.x; r++) {
|
||||
struct ncclRing* ring = comm->rings+r;
|
||||
ring->collectives[(ring->collStart+ring->collCount-1)%NCCL_MAX_OPS].active = 2;
|
||||
}
|
||||
|
||||
// Find the first operation, choose the kernel accordingly and pass it
|
||||
// as the first argument.
|
||||
struct ncclColl* coll = comm->rings[0].collectives+comm->rings[0].collStart;
|
||||
memcpy(&comm->args, coll, sizeof(struct ncclColl));
|
||||
// As we pass that coll directly, we can free it immediately.
|
||||
coll->active = 0;
|
||||
|
||||
params->func = ncclKerns[coll->funcIndex];
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclCpuBarrierIn(struct ncclComm* comm, int* isLast) {
|
||||
volatile int* ptr = (volatile int*)(comm->intraBarrier+comm->intraPhase);
|
||||
int val = *ptr;
|
||||
bool done = false;
|
||||
while (done == false) {
|
||||
if (val >= comm->intraRanks) {
|
||||
WARN("Trying to launch too many collectives");
|
||||
return ncclInvalidUsage;
|
||||
}
|
||||
if (val+1 == comm->intraRanks) {
|
||||
// Reset the barrier.
|
||||
comm->intraBarrier[comm->intraPhase^1] = 0;
|
||||
*isLast = 1;
|
||||
return ncclSuccess;
|
||||
}
|
||||
done = __sync_bool_compare_and_swap(ptr, val, val+1);
|
||||
val++;
|
||||
}
|
||||
*isLast = 0;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclCpuBarrierLast(struct ncclComm* comm) {
|
||||
volatile int* ptr = (volatile int*)(comm->intraBarrier+comm->intraPhase);
|
||||
int val = *ptr;
|
||||
if (__sync_bool_compare_and_swap(ptr, val, val+1) != true) {
|
||||
WARN("Trying to launch too many collectives");
|
||||
return ncclInternalError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclCpuBarrierOut(struct ncclComm* comm) {
|
||||
volatile int* ptr = (volatile int*)(comm->intraBarrier+comm->intraPhase);
|
||||
while (*ptr < comm->intraRanks) pthread_yield();
|
||||
comm->intraPhase ^= 1;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclBarrierEnqueue(struct ncclComm* comm) {
|
||||
if (comm->nRanks == 1) return ncclSuccess;
|
||||
struct cudaLaunchParams* params = comm->myParams;
|
||||
|
||||
NCCLCHECK(setupLaunch(comm, params));
|
||||
|
||||
// Use internal NCCL stream for CGMD/GROUP launch if required or if the user stream is NULL
|
||||
if (comm->launchMode == ncclComm::GROUP && (comm->groupCudaStream || comm->userStream == NULL)) {
|
||||
// Enqueue event in user stream
|
||||
CUDACHECK(cudaEventRecord(comm->doneEvent, comm->userStream));
|
||||
// Create dependency between user stream and internal NCCL stream
|
||||
CUDACHECK(cudaStreamWaitEvent(comm->groupStream, comm->doneEvent, 0));
|
||||
params->stream = comm->groupStream;
|
||||
} else {
|
||||
if (comm->userStream != params->stream) {
|
||||
// Stream changed from last call, create dependency against last NCCL kernel launch
|
||||
CUDACHECK(cudaStreamWaitEvent(comm->userStream, comm->doneEvent, 0));
|
||||
}
|
||||
params->stream = comm->userStream;
|
||||
}
|
||||
|
||||
int isLast = 0;
|
||||
NCCLCHECK(ncclCpuBarrierIn(comm, &isLast));
|
||||
|
||||
if (isLast) {
|
||||
if (comm->launchMode == ncclComm::GROUP) {
|
||||
// I'm the last. Launch all operations.
|
||||
NCCLCHECK(ncclLaunchCooperativeKernelMultiDevice(comm->intraParams, comm->intraCudaDevs, comm->intraRanks, *comm->intraCGMode));
|
||||
}
|
||||
NCCLCHECK(ncclCpuBarrierLast(comm));
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclBarrierEnqueueWait(ncclComm_t comm) {
|
||||
if (comm->nRanks == 1) return ncclSuccess;
|
||||
// We can't print the CG mode before the first barrier happened.
|
||||
if (comm->rank == 0 && *comm->intraCGMode & 0x10) {
|
||||
*comm->intraCGMode ^= 0x10;
|
||||
INFO(INIT,"Launch mode %s%s%s",
|
||||
comm->launchMode == ncclComm::GROUP ? "Group" : "Parallel",
|
||||
*comm->intraCGMode ? "/CGMD" : "",
|
||||
(comm->launchMode == ncclComm::GROUP && comm->groupCudaStream) ? "/Stream" : "");
|
||||
}
|
||||
|
||||
NCCLCHECK(ncclCpuBarrierOut(comm));
|
||||
|
||||
struct cudaLaunchParams *params = comm->myParams;
|
||||
if (comm->launchMode == ncclComm::PARALLEL) {
|
||||
CUDACHECK(cudaLaunchKernel(params->func, params->gridDim, params->blockDim, params->args, params->sharedMem, params->stream));
|
||||
}
|
||||
// Start the network proxies as soon as the kernel has been launched. We can't
|
||||
// perform any CUDA call between the two or having a cudaFree between the CUDA
|
||||
// launch and the transportStartProxies call could cause a deadlock.
|
||||
// Also, starting the proxies after the CUDA launch seems to be better for
|
||||
// performance (latency).
|
||||
for (int r=0; r<params->gridDim.x; r++) {
|
||||
struct ncclRing* ring = comm->rings+r;
|
||||
ring->collStart = ring->collFifoTail;
|
||||
ring->collCount = 0;
|
||||
}
|
||||
params->gridDim.x = params->blockDim.x = 0;
|
||||
NCCLCHECK(transportStartProxies(comm));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclEnqueueEvents(ncclComm_t comm) {
|
||||
struct cudaLaunchParams *params = comm->myParams;
|
||||
// Enqueue event after NCCL kernel
|
||||
CUDACHECK(cudaEventRecord(comm->doneEvent, params->stream));
|
||||
// Use internal NCCL stream for CGMD/GROUP launch if required or if the user stream is NULL
|
||||
if (comm->launchMode == ncclComm::GROUP && (comm->groupCudaStream || comm->userStream == NULL)) {
|
||||
// Create dependency between NCCL internal stream and user stream
|
||||
CUDACHECK(cudaStreamWaitEvent(comm->userStream, comm->doneEvent, 0));
|
||||
}
|
||||
comm->userStreamSet = false;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclEnqueueCheck(ncclFunc_t func, const char* primName, const void* sendbuff,
|
||||
void* recvbuff, size_t count, ncclDataType_t type, ncclRedOp_t op, int root,
|
||||
ncclComm_t comm, cudaStream_t stream) {
|
||||
if (comm == NULL) return ncclInvalidArgument;
|
||||
// Launch asynchronously if needed
|
||||
if (ncclAsyncMode()) {
|
||||
ncclResult_t ret = ncclSuccess;
|
||||
int savedDev = -1;
|
||||
if (comm->checkPointers) {
|
||||
CUDACHECKGOTO(cudaGetDevice(&savedDev), ret, end);
|
||||
CUDACHECKGOTO(cudaSetDevice(comm->cudaDev), ret, end);
|
||||
}
|
||||
// Check arguments
|
||||
NCCLCHECKGOTO(ArgsCheck(sendbuff, recvbuff, count, type, op, root, comm, primName), ret, end);
|
||||
// Always register comm even in case of error to make sure ncclGroupEnd
|
||||
// cleans it up.
|
||||
NCCLCHECK(ncclAsyncColl(comm));
|
||||
NCCLCHECKGOTO(func(sendbuff, recvbuff, count, type, op, root, comm, stream), ret, end);
|
||||
end:
|
||||
if (savedDev != -1) CUDACHECK(cudaSetDevice(savedDev));
|
||||
ncclAsyncErrCheck(ret);
|
||||
return ret;
|
||||
} else {
|
||||
NCCLCHECK(ArgsCheck(sendbuff, recvbuff, count, type, op, root, comm, primName));
|
||||
NCCLCHECK(func(sendbuff, recvbuff, count, type, op, root, comm, stream));
|
||||
NCCLCHECK(ncclBarrierEnqueue(comm));
|
||||
NCCLCHECK(ncclBarrierEnqueueWait(comm));
|
||||
NCCLCHECK(ncclEnqueueEvents(comm));
|
||||
return ncclSuccess;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2015-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include "group.h"
|
||||
#include "debug.h"
|
||||
#include "enqueue.h"
|
||||
|
||||
#define MAX_ASYNC_OPS 128
|
||||
thread_local pthread_t ncclGroupThreads[MAX_ASYNC_OPS];
|
||||
thread_local int ncclGroupIndex = 0;
|
||||
thread_local int ncclGroupMode = 0;
|
||||
thread_local ncclResult_t ncclGroupError = ncclSuccess;
|
||||
|
||||
bool ncclAsyncMode() {
|
||||
return ncclGroupMode > 0;
|
||||
}
|
||||
|
||||
ncclResult_t ncclAsyncErrCheck(ncclResult_t ret) {
|
||||
if (ncclGroupError == ncclSuccess || ret != ncclSuccess) ncclGroupError = ret;
|
||||
return ret;
|
||||
}
|
||||
|
||||
struct ncclInitArgs {
|
||||
ncclInitFunc_t func;
|
||||
int cudaDev;
|
||||
ncclComm_t* newcomm;
|
||||
int ndev;
|
||||
ncclUniqueId commId;
|
||||
int myrank;
|
||||
};
|
||||
struct ncclCollArgs {
|
||||
ncclComm_t comm;
|
||||
};
|
||||
|
||||
enum ncclAsyncFuncType {
|
||||
ASYNC_FUNC_INVALID = 0,
|
||||
ASYNC_FUNC_INIT = 1,
|
||||
ASYNC_FUNC_COLL = 2,
|
||||
};
|
||||
struct ncclAsyncArgs {
|
||||
ncclResult_t ret;
|
||||
enum ncclAsyncFuncType funcType;
|
||||
union {
|
||||
ncclCollArgs coll;
|
||||
ncclInitArgs init;
|
||||
};
|
||||
};
|
||||
|
||||
thread_local struct ncclAsyncArgs ncclGroupArgs[MAX_ASYNC_OPS];
|
||||
|
||||
ncclResult_t ncclSetDevice(int cudaDev) {
|
||||
CUDACHECK(cudaSetDevice(cudaDev));
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
#define CHECK(a) do { \
|
||||
if ((args->ret = (a)) != ncclSuccess) { \
|
||||
INFO(INIT,"%s:%d -> %d [Async thread]", __FILE__, __LINE__, args->ret); \
|
||||
return args; \
|
||||
} \
|
||||
} while(0)
|
||||
|
||||
void* ncclAsyncThreadMain(void* args_) {
|
||||
struct ncclAsyncArgs* args = (struct ncclAsyncArgs*)args_;
|
||||
CHECK(ncclSetDevice(args->init.cudaDev));
|
||||
CHECK(args->init.func(args->init.newcomm, args->init.ndev, args->init.commId, args->init.myrank));
|
||||
return args;
|
||||
}
|
||||
|
||||
ncclResult_t ncclAsyncInit(ncclInitFunc_t func, int cudaDev, ncclComm_t* newcomm, int ndev, ncclUniqueId commId, int myrank) {
|
||||
if (ncclGroupIndex >= MAX_ASYNC_OPS) {
|
||||
WARN("Too many async operations in progress, max is %d", MAX_ASYNC_OPS);
|
||||
return ncclAsyncErrCheck(ncclInternalError);
|
||||
}
|
||||
int index = ncclGroupIndex++;
|
||||
struct ncclAsyncArgs* args = ncclGroupArgs+index;
|
||||
args->funcType = ASYNC_FUNC_INIT;
|
||||
args->init.func = func;
|
||||
args->init.cudaDev = cudaDev;
|
||||
args->init.newcomm = newcomm;
|
||||
args->init.ndev = ndev;
|
||||
memcpy(&args->init.commId, &commId, sizeof(commId));
|
||||
args->init.myrank = myrank;
|
||||
// We need to use threads for Init
|
||||
pthread_create(ncclGroupThreads+index, NULL, ncclAsyncThreadMain, args);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t ncclAsyncColl(ncclComm_t comm) {
|
||||
struct ncclAsyncArgs* args = ncclGroupArgs;
|
||||
for (int i=0; i<ncclGroupIndex; i++) {
|
||||
if (args->coll.comm == comm) return ncclSuccess;
|
||||
args++;
|
||||
}
|
||||
if (ncclGroupIndex >= MAX_ASYNC_OPS) {
|
||||
WARN("Too many async operations in progress, max is %d", MAX_ASYNC_OPS);
|
||||
return ncclAsyncErrCheck(ncclInternalError);
|
||||
}
|
||||
ncclGroupIndex++;
|
||||
args->funcType = ASYNC_FUNC_COLL;
|
||||
args->coll.comm = comm;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
NCCL_API(ncclResult_t, ncclGroupStart);
|
||||
ncclResult_t ncclGroupStart() {
|
||||
ncclGroupMode++;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
NCCL_API(ncclResult_t, ncclGroupEnd);
|
||||
ncclResult_t ncclGroupEnd() {
|
||||
ncclGroupMode--;
|
||||
if (ncclGroupMode > 0) return ncclSuccess;
|
||||
int savedDev;
|
||||
CUDACHECK(cudaGetDevice(&savedDev));
|
||||
int done = ncclGroupIndex;
|
||||
int doneArray[ncclGroupIndex];
|
||||
for (int i=0; i<ncclGroupIndex; i++) doneArray[i] = 0;
|
||||
|
||||
ncclResult_t ret = ncclGroupError;
|
||||
if (ret != ncclSuccess) goto group_cleanup;
|
||||
|
||||
/* Collectives are done in three steps :
|
||||
* 1. Barrier Check In. Only the last call may call cudaLaunchKernel[cooperative]
|
||||
* 2. Barrier Wait. No CUDA call is permitted
|
||||
* 3. Enqueue Events. CUDA event wait/enqueue.
|
||||
* This is needed because step 2 cannot call any CUDA primitive, otherwise if
|
||||
* cudaFree happens between 1 and 3, it could block that CUDA call and
|
||||
* prevent some ranks from launching their network threads, which would
|
||||
* prevent the NCCL call from completing, blocking the cudaFree call.
|
||||
*/
|
||||
for (int i=0; i<ncclGroupIndex; i++) {
|
||||
struct ncclAsyncArgs* args = ncclGroupArgs+i;
|
||||
if (args->funcType == ASYNC_FUNC_COLL) {
|
||||
if (args->coll.comm->userStream == NULL)
|
||||
CUDACHECKGOTO(cudaSetDevice(args->coll.comm->cudaDev), ret, end);
|
||||
NCCLCHECKGOTO(ncclBarrierEnqueue(args->coll.comm), ret, end);
|
||||
}
|
||||
}
|
||||
for (int i=0; i<ncclGroupIndex; i++) {
|
||||
struct ncclAsyncArgs* args = ncclGroupArgs+i;
|
||||
if (args->funcType == ASYNC_FUNC_COLL) {
|
||||
CUDACHECKGOTO(cudaSetDevice(args->coll.comm->cudaDev), ret, end);
|
||||
NCCLCHECKGOTO(ncclBarrierEnqueueWait(args->coll.comm), ret, end);
|
||||
}
|
||||
}
|
||||
for (int i=0; i<ncclGroupIndex; i++) {
|
||||
struct ncclAsyncArgs* args = ncclGroupArgs+i;
|
||||
if (args->funcType == ASYNC_FUNC_COLL) {
|
||||
if (args->coll.comm->userStream == NULL)
|
||||
CUDACHECKGOTO(cudaSetDevice(args->coll.comm->cudaDev), ret, end);
|
||||
NCCLCHECKGOTO(ncclEnqueueEvents(args->coll.comm), ret, end);
|
||||
doneArray[i] = 1;
|
||||
done--;
|
||||
}
|
||||
}
|
||||
|
||||
/* For init, since we use threads, we just wait for threads to complete */
|
||||
while (done) {
|
||||
for (int i=0; i<ncclGroupIndex; i++) {
|
||||
struct ncclAsyncArgs* args = ncclGroupArgs+i;
|
||||
if (args->funcType == ASYNC_FUNC_INIT && doneArray[i] == 0) {
|
||||
int err = pthread_tryjoin_np(ncclGroupThreads[i], NULL);
|
||||
if (err == EBUSY) continue;
|
||||
if (err != 0) { ret = ncclSystemError; goto end; }
|
||||
if (args->ret != ncclSuccess) { ret = args->ret; goto end; }
|
||||
doneArray[i] = 1;
|
||||
done--;
|
||||
}
|
||||
}
|
||||
}
|
||||
goto end;
|
||||
group_cleanup:
|
||||
// At least one call in the group failed. Since we want to make that group
|
||||
// an atomic operation, we need to cancel all operations.
|
||||
for (int i=0; i<ncclGroupIndex; i++) {
|
||||
struct ncclComm* comm = ncclGroupArgs[i].coll.comm;
|
||||
for (int r=0; r<comm->nRings; r++) {
|
||||
struct ncclRing* ring = comm->rings+r;
|
||||
for (int i=0; i<ring->collCount; i++) {
|
||||
ring->collectives[(ring->collStart + i)%NCCL_MAX_OPS].active = 0;
|
||||
}
|
||||
ring->collFifoTail = ring->collStart;
|
||||
ring->collCount = 0;
|
||||
}
|
||||
comm->myParams->gridDim.x = comm->myParams->blockDim.x = 0;
|
||||
comm->userStreamSet = false;
|
||||
}
|
||||
end:
|
||||
ncclGroupError = ncclSuccess;
|
||||
ncclGroupIndex = 0;
|
||||
CUDACHECK(cudaSetDevice(savedDev)); // do other clean-ups first before calling cudaSetDevice, because this call can fail too
|
||||
return ret;
|
||||
}
|
||||
@@ -0,0 +1,290 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2015-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include "ibvwrap.h"
|
||||
#include <sys/types.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include <dlfcn.h>
|
||||
#include "core.h"
|
||||
|
||||
static enum { ibvUninitialized, ibvInitializing, ibvInitialized, ibvError } ibvState = ibvUninitialized;
|
||||
|
||||
/*Function Pointers*/
|
||||
int (*ibv_internal_fork_init)(void);
|
||||
struct ibv_device** (*ibv_internal_get_device_list)(int *num_devices);
|
||||
void (*ibv_internal_free_device_list)(struct ibv_device **list);
|
||||
const char * (*ibv_internal_get_device_name)(struct ibv_device *device);
|
||||
struct ibv_context* (*ibv_internal_open_device)(struct ibv_device* device);
|
||||
int (*ibv_internal_close_device)(struct ibv_context *context);
|
||||
int (*ibv_internal_get_async_event)(struct ibv_context *context, struct ibv_async_event *event);
|
||||
void (*ibv_internal_ack_async_event)(struct ibv_async_event *event);
|
||||
int (*ibv_internal_query_device)(struct ibv_context *context, struct ibv_device_attr *device_attr);
|
||||
int (*ibv_internal_query_port)(struct ibv_context *context, uint8_t port_num, struct ibv_port_attr *port_attr);
|
||||
int (*ibv_internal_query_gid)(struct ibv_context *context, uint8_t port_num, int index, union ibv_gid *gid);
|
||||
int (*ibv_internal_query_qp)(struct ibv_qp *qp, struct ibv_qp_attr *attr, int attr_mask, struct ibv_qp_init_attr *init_attr);
|
||||
struct ibv_pd * (*ibv_internal_alloc_pd)(struct ibv_context *context);
|
||||
int (*ibv_internal_dealloc_pd)(struct ibv_pd *pd);
|
||||
struct ibv_mr * (*ibv_internal_reg_mr)(struct ibv_pd *pd, void *addr, size_t length, int access);
|
||||
int (*ibv_internal_dereg_mr)(struct ibv_mr *mr);
|
||||
struct ibv_cq * (*ibv_internal_create_cq)(struct ibv_context *context, int cqe, void *cq_context, struct ibv_comp_channel *channel, int comp_vector);
|
||||
int (*ibv_internal_destroy_cq)(struct ibv_cq *cq);
|
||||
struct ibv_qp * (*ibv_internal_create_qp)(struct ibv_pd *pd, struct ibv_qp_init_attr *qp_init_attr);
|
||||
int (*ibv_internal_modify_qp)(struct ibv_qp *qp, struct ibv_qp_attr *attr, int attr_mask);
|
||||
int (*ibv_internal_destroy_qp)(struct ibv_qp *qp);
|
||||
const char * (*ibv_internal_event_type_str)(enum ibv_event_type event);
|
||||
|
||||
// IBVERBS Library versioning
|
||||
#define IBVERBS_VERSION "IBVERBS_1.1"
|
||||
|
||||
ncclResult_t wrap_ibv_symbols(void) {
|
||||
if (ibvState == ibvInitialized)
|
||||
return ncclSuccess;
|
||||
if (ibvState == ibvError)
|
||||
return ncclSystemError;
|
||||
|
||||
if (__sync_bool_compare_and_swap(&ibvState, ibvUninitialized, ibvInitializing) == false) {
|
||||
// Another thread raced in front of us. Wait for it to be done.
|
||||
while (ibvState == ibvInitializing) pthread_yield();
|
||||
return (ibvState == ibvInitialized) ? ncclSuccess : ncclSystemError;
|
||||
}
|
||||
|
||||
static void* ibvhandle = NULL;
|
||||
void* tmp;
|
||||
void** cast;
|
||||
|
||||
ibvhandle=dlopen("libibverbs.so", RTLD_NOW);
|
||||
if (!ibvhandle) {
|
||||
ibvhandle=dlopen("libibverbs.so.1", RTLD_NOW);
|
||||
if (!ibvhandle) {
|
||||
WARN("Failed to open libibverbs.so[.1]");
|
||||
goto teardown;
|
||||
}
|
||||
}
|
||||
|
||||
#define LOAD_SYM(handle, symbol, funcptr) do { \
|
||||
cast = (void**)&funcptr; \
|
||||
tmp = dlvsym(handle, symbol, IBVERBS_VERSION); \
|
||||
if (tmp == NULL) { \
|
||||
WARN("dlvsym failed on %s - %s version %s", symbol, dlerror(), IBVERBS_VERSION); \
|
||||
goto teardown; \
|
||||
} \
|
||||
*cast = tmp; \
|
||||
} while (0)
|
||||
|
||||
LOAD_SYM(ibvhandle, "ibv_get_device_list", ibv_internal_get_device_list);
|
||||
LOAD_SYM(ibvhandle, "ibv_free_device_list", ibv_internal_free_device_list);
|
||||
LOAD_SYM(ibvhandle, "ibv_get_device_name", ibv_internal_get_device_name);
|
||||
LOAD_SYM(ibvhandle, "ibv_open_device", ibv_internal_open_device);
|
||||
LOAD_SYM(ibvhandle, "ibv_close_device", ibv_internal_close_device);
|
||||
LOAD_SYM(ibvhandle, "ibv_get_async_event", ibv_internal_get_async_event);
|
||||
LOAD_SYM(ibvhandle, "ibv_ack_async_event", ibv_internal_ack_async_event);
|
||||
LOAD_SYM(ibvhandle, "ibv_query_device", ibv_internal_query_device);
|
||||
LOAD_SYM(ibvhandle, "ibv_query_port", ibv_internal_query_port);
|
||||
LOAD_SYM(ibvhandle, "ibv_query_gid", ibv_internal_query_gid);
|
||||
LOAD_SYM(ibvhandle, "ibv_query_qp", ibv_internal_query_qp);
|
||||
LOAD_SYM(ibvhandle, "ibv_alloc_pd", ibv_internal_alloc_pd);
|
||||
LOAD_SYM(ibvhandle, "ibv_dealloc_pd", ibv_internal_dealloc_pd);
|
||||
LOAD_SYM(ibvhandle, "ibv_reg_mr", ibv_internal_reg_mr);
|
||||
LOAD_SYM(ibvhandle, "ibv_dereg_mr", ibv_internal_dereg_mr);
|
||||
LOAD_SYM(ibvhandle, "ibv_create_cq", ibv_internal_create_cq);
|
||||
LOAD_SYM(ibvhandle, "ibv_destroy_cq", ibv_internal_destroy_cq);
|
||||
LOAD_SYM(ibvhandle, "ibv_create_qp", ibv_internal_create_qp);
|
||||
LOAD_SYM(ibvhandle, "ibv_modify_qp", ibv_internal_modify_qp);
|
||||
LOAD_SYM(ibvhandle, "ibv_destroy_qp", ibv_internal_destroy_qp);
|
||||
LOAD_SYM(ibvhandle, "ibv_fork_init", ibv_internal_fork_init);
|
||||
LOAD_SYM(ibvhandle, "ibv_event_type_str", ibv_internal_event_type_str);
|
||||
|
||||
ibvState = ibvInitialized;
|
||||
return ncclSuccess;
|
||||
|
||||
teardown:
|
||||
ibv_internal_get_device_list = NULL;
|
||||
ibv_internal_free_device_list = NULL;
|
||||
ibv_internal_get_device_name = NULL;
|
||||
ibv_internal_open_device = NULL;
|
||||
ibv_internal_close_device = NULL;
|
||||
ibv_internal_get_async_event = NULL;
|
||||
ibv_internal_ack_async_event = NULL;
|
||||
ibv_internal_query_device = NULL;
|
||||
ibv_internal_query_port = NULL;
|
||||
ibv_internal_query_gid = NULL;
|
||||
ibv_internal_query_qp = NULL;
|
||||
ibv_internal_alloc_pd = NULL;
|
||||
ibv_internal_dealloc_pd = NULL;
|
||||
ibv_internal_reg_mr = NULL;
|
||||
ibv_internal_dereg_mr = NULL;
|
||||
ibv_internal_create_cq = NULL;
|
||||
ibv_internal_destroy_cq = NULL;
|
||||
ibv_internal_create_qp = NULL;
|
||||
ibv_internal_modify_qp = NULL;
|
||||
ibv_internal_destroy_qp = NULL;
|
||||
ibv_internal_fork_init = NULL;
|
||||
ibv_internal_event_type_str = NULL;
|
||||
|
||||
if (ibvhandle != NULL) dlclose(ibvhandle);
|
||||
ibvState = ibvError;
|
||||
return ncclSystemError;
|
||||
}
|
||||
|
||||
#define IBV_PTR_CHECK_ERRNO(name_internal, call, retval, error_retval, name) \
|
||||
if (name_internal == NULL) { \
|
||||
WARN("lib wrapper not initialized."); \
|
||||
return ncclInternalError; \
|
||||
} \
|
||||
retval = call; \
|
||||
if (retval == error_retval) { \
|
||||
WARN("Call to " name " failed with error %s", strerror(errno)); \
|
||||
return ncclSystemError; \
|
||||
} \
|
||||
return ncclSuccess;
|
||||
|
||||
#define IBV_PTR_CHECK(name_internal, call, retval, error_retval, name) \
|
||||
if (name_internal == NULL) { \
|
||||
WARN("lib wrapper not initialized."); \
|
||||
return ncclInternalError; \
|
||||
} \
|
||||
retval = call; \
|
||||
if (retval == error_retval) { \
|
||||
WARN("Call to " name " failed"); \
|
||||
return ncclSystemError; \
|
||||
} \
|
||||
return ncclSuccess;
|
||||
|
||||
#define IBV_INT_CHECK_RET_ERRNO(name_internal, call, success_retval, name) \
|
||||
if (name_internal == NULL) { \
|
||||
WARN("lib wrapper not initialized."); \
|
||||
return ncclInternalError; \
|
||||
} \
|
||||
int ret = call; \
|
||||
if (ret != success_retval) { \
|
||||
WARN("Call to " name " failed with error %s", strerror(ret)); \
|
||||
return ncclSystemError; \
|
||||
} \
|
||||
return ncclSuccess;
|
||||
|
||||
#define IBV_INT_CHECK(name_internal, call, error_retval, name) \
|
||||
if (name_internal == NULL) { \
|
||||
WARN("lib wrapper not initialized."); \
|
||||
return ncclInternalError; \
|
||||
} \
|
||||
int ret = call; \
|
||||
if (ret == error_retval) { \
|
||||
WARN("Call to " name " failed"); \
|
||||
return ncclSystemError; \
|
||||
} \
|
||||
return ncclSuccess;
|
||||
|
||||
#define IBV_PASSTHRU(name_internal, call) \
|
||||
if (name_internal == NULL) { \
|
||||
WARN("lib wrapper not initialized."); \
|
||||
return ncclInternalError; \
|
||||
} \
|
||||
call; \
|
||||
return ncclSuccess;
|
||||
|
||||
ncclResult_t wrap_ibv_fork_init() {
|
||||
IBV_INT_CHECK(ibv_internal_fork_init, ibv_internal_fork_init(), -1, "ibv_fork_init");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_get_device_list(struct ibv_device ***ret, int *num_devices) {
|
||||
*ret = ibv_internal_get_device_list(num_devices);
|
||||
if (*ret == NULL) *num_devices = 0;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_free_device_list(struct ibv_device **list) {
|
||||
IBV_PASSTHRU(ibv_internal_free_device_list, ibv_internal_free_device_list(list));
|
||||
}
|
||||
|
||||
const char *wrap_ibv_get_device_name(struct ibv_device *device) {
|
||||
if (ibv_internal_get_device_name == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
exit(-1);
|
||||
}
|
||||
return ibv_internal_get_device_name(device);
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_open_device(struct ibv_context **ret, struct ibv_device *device) { /*returns 0 on success, -1 on failure*/
|
||||
IBV_PTR_CHECK(ibv_internal_open_device, ibv_internal_open_device(device), *ret, NULL, "ibv_open_device");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_close_device(struct ibv_context *context) { /*returns 0 on success, -1 on failure*/
|
||||
IBV_INT_CHECK(ibv_internal_close_device, ibv_internal_close_device(context), -1, "ibv_close_device");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_get_async_event(struct ibv_context *context, struct ibv_async_event *event) { /*returns 0 on success, and -1 on error*/
|
||||
IBV_INT_CHECK(ibv_internal_get_async_event, ibv_internal_get_async_event(context, event), -1, "ibv_get_async_event");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_ack_async_event(struct ibv_async_event *event) {
|
||||
IBV_PASSTHRU(ibv_internal_ack_async_event, ibv_internal_ack_async_event(event));
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_query_device(struct ibv_context *context, struct ibv_device_attr *device_attr) { /*returns 0 on success, or the value of errno on failure (which indicates the failure reason)*/
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_query_device, ibv_internal_query_device(context, device_attr), 0, "ibv_query_device");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_query_port(struct ibv_context *context, uint8_t port_num, struct ibv_port_attr *port_attr) { /*returns 0 on success, or the value of errno on failure (which indicates the failure reason)*/
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_query_port, ibv_internal_query_port(context, port_num, port_attr), 0, "ibv_query_port");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_query_gid(struct ibv_context *context, uint8_t port_num, int index, union ibv_gid *gid) {
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_query_gid, ibv_internal_query_gid(context, port_num, index, gid), 0, "ibv_query_gid");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_query_qp(struct ibv_qp *qp, struct ibv_qp_attr *attr, int attr_mask, struct ibv_qp_init_attr *init_attr) {
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_query_qp, ibv_internal_query_qp(qp, attr, attr_mask, init_attr), 0, "ibv_query_qp");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_alloc_pd(struct ibv_pd **ret, struct ibv_context *context) {
|
||||
IBV_PTR_CHECK(ibv_internal_alloc_pd, ibv_internal_alloc_pd(context), *ret, NULL, "ibv_alloc_pd");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_dealloc_pd(struct ibv_pd *pd) { /*returns 0 on success, or the value of errno on failure (which indicates the failure reason)*/
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_dealloc_pd, ibv_internal_dealloc_pd(pd), 0, "ibv_dealloc_pd");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_reg_mr(struct ibv_mr **ret, struct ibv_pd *pd, void *addr, size_t length, int access) {
|
||||
IBV_PTR_CHECK(ibv_internal_reg_mr, ibv_internal_reg_mr(pd, addr, length, access), *ret, NULL, "ibv_reg_mr");
|
||||
}
|
||||
|
||||
struct ibv_mr * wrap_direct_ibv_reg_mr(struct ibv_pd *pd, void *addr, size_t length, int access) {
|
||||
if (ibv_internal_reg_mr == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
return NULL;
|
||||
}
|
||||
return ibv_internal_reg_mr(pd, addr, length, access);
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_dereg_mr(struct ibv_mr *mr) { /*returns 0 on success, or the value of errno on failure (which indicates the failure reason)*/
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_dereg_mr, ibv_internal_dereg_mr(mr), 0, "ibv_dereg_mr");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_create_cq(struct ibv_cq **ret, struct ibv_context *context, int cqe, void *cq_context, struct ibv_comp_channel *channel, int comp_vector) {
|
||||
IBV_PTR_CHECK(ibv_internal_create_cq, ibv_internal_create_cq(context, cqe, cq_context, channel, comp_vector), *ret, NULL, "ibv_create_cq");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_destroy_cq(struct ibv_cq *cq) {
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_destroy_cq, ibv_internal_destroy_cq(cq), 0, "ibv_destroy_cq");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_destroy_qp(struct ibv_qp *qp) {
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_destroy_qp, ibv_internal_destroy_qp(qp), 0, "ibv_destroy_qp");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_create_qp(struct ibv_qp **ret, struct ibv_pd *pd, struct ibv_qp_init_attr *qp_init_attr) {
|
||||
IBV_PTR_CHECK(ibv_internal_create_qp, ibv_internal_create_qp(pd, qp_init_attr), *ret, NULL, "ibv_create_qp");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_modify_qp(struct ibv_qp *qp, struct ibv_qp_attr *attr, int attr_mask) { /*returns 0 on success, or the value of errno on failure (which indicates the failure reason)*/
|
||||
IBV_INT_CHECK_RET_ERRNO(ibv_internal_modify_qp, ibv_internal_modify_qp(qp, attr, attr_mask), 0, "ibv_modify_qp");
|
||||
}
|
||||
|
||||
ncclResult_t wrap_ibv_event_type_str(char **ret, enum ibv_event_type event) {
|
||||
*ret = (char *) ibv_internal_event_type_str(event);
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -0,0 +1,248 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2015-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include "nvmlwrap.h"
|
||||
|
||||
#ifndef NVML_DIRECT
|
||||
#include <dlfcn.h>
|
||||
#include "core.h"
|
||||
|
||||
static enum { nvmlUninitialized, nvmlInitializing, nvmlInitialized, nvmlError } nvmlState = nvmlUninitialized;
|
||||
|
||||
static nvmlReturn_t (*nvmlInternalInit)(void);
|
||||
static nvmlReturn_t (*nvmlInternalShutdown)(void);
|
||||
static nvmlReturn_t (*nvmlInternalDeviceGetHandleByPciBusId)(const char* pciBusId, nvmlDevice_t* device);
|
||||
static nvmlReturn_t (*nvmlInternalDeviceGetIndex)(nvmlDevice_t device, unsigned* index);
|
||||
static nvmlReturn_t (*nvmlInternalDeviceSetCpuAffinity)(nvmlDevice_t device);
|
||||
static nvmlReturn_t (*nvmlInternalDeviceClearCpuAffinity)(nvmlDevice_t device);
|
||||
static const char* (*nvmlInternalErrorString)(nvmlReturn_t r);
|
||||
static nvmlReturn_t (*nvmlInternalDeviceGetNvLinkState)(nvmlDevice_t device, unsigned int link, nvmlEnableState_t *isActive);
|
||||
static nvmlReturn_t (*nvmlInternalDeviceGetPciInfo)(nvmlDevice_t device, nvmlPciInfo_t* pci);
|
||||
static nvmlReturn_t (*nvmlInternalDeviceGetNvLinkRemotePciInfo)(nvmlDevice_t device, unsigned int link, nvmlPciInfo_t *pci);
|
||||
static nvmlReturn_t (*nvmlInternalDeviceGetNvLinkCapability)(nvmlDevice_t device, unsigned int link,
|
||||
nvmlNvLinkCapability_t capability, unsigned int *capResult);
|
||||
|
||||
ncclResult_t wrapNvmlSymbols(void) {
|
||||
if (nvmlState == nvmlInitialized)
|
||||
return ncclSuccess;
|
||||
if (nvmlState == nvmlError)
|
||||
return ncclSystemError;
|
||||
|
||||
if (__sync_bool_compare_and_swap(&nvmlState, nvmlUninitialized, nvmlInitializing) == false) {
|
||||
// Another thread raced in front of us. Wait for it to be done.
|
||||
while (nvmlState == nvmlInitializing) pthread_yield();
|
||||
return (nvmlState == nvmlInitialized) ? ncclSuccess : ncclSystemError;
|
||||
}
|
||||
|
||||
static void* nvmlhandle = NULL;
|
||||
void* tmp;
|
||||
void** cast;
|
||||
|
||||
nvmlhandle=dlopen("libnvidia-ml.so.1", RTLD_NOW);
|
||||
if (!nvmlhandle) {
|
||||
WARN("Failed to open libnvidia-ml.so.1");
|
||||
goto teardown;
|
||||
}
|
||||
|
||||
#define LOAD_SYM(handle, symbol, funcptr) do { \
|
||||
cast = (void**)&funcptr; \
|
||||
tmp = dlsym(handle, symbol); \
|
||||
if (tmp == NULL) { \
|
||||
WARN("dlsym failed on %s - %s", symbol, dlerror());\
|
||||
goto teardown; \
|
||||
} \
|
||||
*cast = tmp; \
|
||||
} while (0)
|
||||
|
||||
#define LOAD_SYM_OPTIONAL(handle, symbol, funcptr) do {\
|
||||
cast = (void**)&funcptr; \
|
||||
tmp = dlsym(handle, symbol); \
|
||||
if (tmp == NULL) { \
|
||||
INFO(INIT,"dlsym failed on %s, ignoring", symbol); \
|
||||
} \
|
||||
*cast = tmp; \
|
||||
} while (0)
|
||||
|
||||
LOAD_SYM(nvmlhandle, "nvmlInit", nvmlInternalInit);
|
||||
LOAD_SYM(nvmlhandle, "nvmlShutdown", nvmlInternalShutdown);
|
||||
LOAD_SYM(nvmlhandle, "nvmlDeviceGetHandleByPciBusId", nvmlInternalDeviceGetHandleByPciBusId);
|
||||
LOAD_SYM(nvmlhandle, "nvmlDeviceGetIndex", nvmlInternalDeviceGetIndex);
|
||||
LOAD_SYM(nvmlhandle, "nvmlDeviceSetCpuAffinity", nvmlInternalDeviceSetCpuAffinity);
|
||||
LOAD_SYM(nvmlhandle, "nvmlDeviceClearCpuAffinity", nvmlInternalDeviceClearCpuAffinity);
|
||||
LOAD_SYM(nvmlhandle, "nvmlErrorString", nvmlInternalErrorString);
|
||||
LOAD_SYM(nvmlhandle, "nvmlDeviceGetPciInfo", nvmlInternalDeviceGetPciInfo);
|
||||
LOAD_SYM_OPTIONAL(nvmlhandle, "nvmlDeviceGetNvLinkState", nvmlInternalDeviceGetNvLinkState);
|
||||
LOAD_SYM_OPTIONAL(nvmlhandle, "nvmlDeviceGetNvLinkRemotePciInfo", nvmlInternalDeviceGetNvLinkRemotePciInfo);
|
||||
LOAD_SYM_OPTIONAL(nvmlhandle, "nvmlDeviceGetNvLinkCapability", nvmlInternalDeviceGetNvLinkCapability);
|
||||
|
||||
nvmlState = nvmlInitialized;
|
||||
return ncclSuccess;
|
||||
|
||||
teardown:
|
||||
nvmlInternalInit = NULL;
|
||||
nvmlInternalShutdown = NULL;
|
||||
nvmlInternalDeviceGetHandleByPciBusId = NULL;
|
||||
nvmlInternalDeviceGetIndex = NULL;
|
||||
nvmlInternalDeviceSetCpuAffinity = NULL;
|
||||
nvmlInternalDeviceClearCpuAffinity = NULL;
|
||||
nvmlInternalDeviceGetPciInfo = NULL;
|
||||
nvmlInternalDeviceGetNvLinkState = NULL;
|
||||
nvmlInternalDeviceGetNvLinkRemotePciInfo = NULL;
|
||||
nvmlInternalDeviceGetNvLinkCapability = NULL;
|
||||
|
||||
if (nvmlhandle != NULL) dlclose(nvmlhandle);
|
||||
nvmlState = nvmlError;
|
||||
return ncclSystemError;
|
||||
}
|
||||
|
||||
|
||||
ncclResult_t wrapNvmlInit(void) {
|
||||
if (nvmlInternalInit == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalInit();
|
||||
if (ret != NVML_SUCCESS) {
|
||||
WARN("nvmlInit() failed: %s",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlShutdown(void) {
|
||||
if (nvmlInternalShutdown == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalShutdown();
|
||||
if (ret != NVML_SUCCESS) {
|
||||
WARN("nvmlShutdown() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlDeviceGetHandleByPciBusId(const char* pciBusId, nvmlDevice_t* device) {
|
||||
if (nvmlInternalDeviceGetHandleByPciBusId == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalDeviceGetHandleByPciBusId(pciBusId, device);
|
||||
if (ret != NVML_SUCCESS) {
|
||||
WARN("nvmlDeviceGetHandleByPciBusId() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlDeviceGetIndex(nvmlDevice_t device, unsigned* index) {
|
||||
if (nvmlInternalDeviceGetIndex == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalDeviceGetIndex(device, index);
|
||||
if (ret != NVML_SUCCESS) {
|
||||
WARN("nvmlDeviceGetIndex() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlDeviceSetCpuAffinity(nvmlDevice_t device) {
|
||||
if (nvmlInternalDeviceSetCpuAffinity == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
// Workaround : it seems SetCpuAffinity is not thread safe.
|
||||
static pthread_mutex_t lock = PTHREAD_MUTEX_INITIALIZER;
|
||||
pthread_mutex_lock(&lock);
|
||||
nvmlReturn_t ret = nvmlInternalDeviceSetCpuAffinity(device);
|
||||
pthread_mutex_unlock(&lock);
|
||||
if (ret != NVML_SUCCESS) {
|
||||
WARN("nvmlDeviceSetCpuAffinity() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlDeviceClearCpuAffinity(nvmlDevice_t device) {
|
||||
if (nvmlInternalInit == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalDeviceClearCpuAffinity(device);
|
||||
if (ret != NVML_SUCCESS) {
|
||||
WARN("nvmlDeviceClearCpuAffinity() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlDeviceGetPciInfo(nvmlDevice_t device, nvmlPciInfo_t* pci) {
|
||||
if (nvmlInternalDeviceGetPciInfo == NULL) {
|
||||
WARN("lib wrapper not initialized.");
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalDeviceGetPciInfo(device, pci);
|
||||
if (ret != NVML_SUCCESS) {
|
||||
WARN("nvmlDeviceGetPciInfo() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlDeviceGetNvLinkState(nvmlDevice_t device, unsigned int link, nvmlEnableState_t *isActive) {
|
||||
if (nvmlInternalDeviceGetNvLinkState == NULL) {
|
||||
/* Do not warn, this symbol is optional. */
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalDeviceGetNvLinkState(device, link, isActive);
|
||||
if (ret != NVML_SUCCESS) {
|
||||
INFO(INIT,"nvmlDeviceGetNvLinkState() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlDeviceGetNvLinkRemotePciInfo(nvmlDevice_t device, unsigned int link, nvmlPciInfo_t *pci) {
|
||||
if (nvmlInternalDeviceGetNvLinkRemotePciInfo == NULL) {
|
||||
/* Do not warn, this symbol is optional. */
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalDeviceGetNvLinkRemotePciInfo(device, link, pci);
|
||||
if (ret != NVML_SUCCESS) {
|
||||
if (ret != NVML_ERROR_NOT_SUPPORTED)
|
||||
INFO(INIT,"nvmlDeviceGetNvLinkRemotePciInfo() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclResult_t wrapNvmlDeviceGetNvLinkCapability(nvmlDevice_t device, unsigned int link,
|
||||
nvmlNvLinkCapability_t capability, unsigned int *capResult) {
|
||||
if (nvmlInternalDeviceGetNvLinkCapability == NULL) {
|
||||
/* Do not warn, this symbol is optional. */
|
||||
return ncclInternalError;
|
||||
}
|
||||
nvmlReturn_t ret = nvmlInternalDeviceGetNvLinkCapability(device, link, capability, capResult);
|
||||
if (ret != NVML_SUCCESS) {
|
||||
if (ret != NVML_ERROR_NOT_SUPPORTED)
|
||||
INFO(INIT,"nvmlDeviceGetNvLinkCapability() failed: %s ",
|
||||
nvmlInternalErrorString(ret));
|
||||
return ncclSystemError;
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
#endif
|
||||
@@ -0,0 +1,355 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2016-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include "core.h"
|
||||
#include "net.h"
|
||||
#include "param.h"
|
||||
|
||||
/* Parse user defined rings. Format is like :
|
||||
* "0 1|1 0|0 1 2 3|3 2 1 0|0 2 3 1|1 3 2 0|0 1 2 3 4 5 6 7|7 6 5 4 3 2 1 0"
|
||||
* Rings with a non-matching number of ranks are ignored so we can provide
|
||||
* rings for multiple cases.
|
||||
*/
|
||||
#define MAX_ENV_RANKS 512
|
||||
static ncclResult_t parseRings(const char* str, int* nringsRet, int nranks, int* prev, int* next) {
|
||||
int ranks[MAX_ENV_RANKS];
|
||||
int nrings = 0;
|
||||
int rank = 0;
|
||||
int offset = 0;
|
||||
int status = 0; // 0 : between numbers, 1 : inside number
|
||||
do {
|
||||
int digit = str[offset] - '0';
|
||||
if (digit >= 0 && digit <= 9) {
|
||||
if (status == 0) {
|
||||
ranks[rank] = digit;
|
||||
status = 1;
|
||||
} else {
|
||||
ranks[rank] = ranks[rank]*10+digit;
|
||||
}
|
||||
} else {
|
||||
if (status == 1) {
|
||||
rank++;
|
||||
if (rank == MAX_ENV_RANKS) goto end;
|
||||
}
|
||||
status = 0;
|
||||
if (str[offset] == '|' || str[offset] == '\0') {
|
||||
int prevRank = ranks[rank-1];
|
||||
// Ignore rings if nranks doesn't match
|
||||
if (rank != nranks) goto newring;
|
||||
|
||||
for (int r=0; r<nranks; r++) {
|
||||
int rank = ranks[r];
|
||||
// Ignore rings with ranks out of bounds
|
||||
if (rank < 0 || rank >= nranks) goto newring;
|
||||
// Ignore rings with duplicate ranks
|
||||
for (int i=0; i<r; i++)
|
||||
if (ranks[i] == rank) goto newring;
|
||||
|
||||
next[nrings*nranks+prevRank] = rank;
|
||||
prev[nrings*nranks+rank] = prevRank;
|
||||
prevRank = rank;
|
||||
}
|
||||
nrings++;
|
||||
newring:
|
||||
rank = 0;
|
||||
}
|
||||
}
|
||||
} while (str[offset++] != 0);
|
||||
end:
|
||||
*nringsRet = nrings;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
/*
|
||||
* Ring creation algorithm
|
||||
*
|
||||
* First, we establish hierarchical coordinates depending on the way ranks can
|
||||
* communicate. After fillCoords, we have for each rank a unique 3-int array
|
||||
* { node, pci_domain, rank } corresponding to the three transports :
|
||||
* { 2[NET], 1[SHM], 0[P2P] }.
|
||||
* Also, we renumber ranks (to indexes) based on their growing coordinates.
|
||||
*
|
||||
* Then, we ask transports to connect groups together. We start with net, then
|
||||
* shm, then p2p. We maintain two arrays, prev and next, where values are equal
|
||||
* to -1 when ranks are not yet connected, and a rank otherwise. We never
|
||||
* connect ranks outside our group, meaning that on 4 nodes of 2 sockets of 4
|
||||
* ranks, if we are rank 13, we should see something like (provided we have a
|
||||
* single net interface, hence a single ring) :
|
||||
*
|
||||
* Connecting all nodes <13>
|
||||
* 2[NET] : prev 31 -1 -1 -1 -1 -1 -1 -1 7 -1 -1 -1 -1 -1 -1 -1 15 -1 -1 -1 -1 -1 -1 -1 23 -1 -1 -1 -1 -1 -1 -1
|
||||
* next -1 -1 -1 -1 -1 -1 -1 8 -1 -1 -1 -1 -1 -1 -1 16 -1 -1 -1 -1 -1 -1 -1 24 -1 -1 -1 -1 -1 -1 -1 0
|
||||
*
|
||||
* Connecting P2P domains with shared memory <13>
|
||||
* 1[SHM] : prev 31 -1 -1 -1 -1 -1 -1 -1 7 -1 -1 -1 11 -1 -1 -1 15 -1 -1 -1 -1 -1 -1 -1 23 -1 -1 -1 -1 -1 -1 -1
|
||||
* next -1 -1 -1 -1 -1 -1 -1 8 -1 -1 -1 12 -1 -1 -1 16 -1 -1 -1 -1 -1 -1 -1 24 -1 -1 -1 -1 -1 -1 -1 0
|
||||
*
|
||||
* Connecting ranks (only inside the P2P domain) <13>
|
||||
* 0[P2P] : prev 31 -1 -1 -1 -1 -1 -1 -1 7 -1 -1 -1 11 12 13 14 15 -1 -1 -1 -1 -1 -1 -1 23 -1 -1 -1 -1 -1 -1 -1
|
||||
* next -1 -1 -1 -1 -1 -1 -1 8 -1 -1 -1 12 13 14 15 16 -1 -1 -1 -1 -1 -1 -1 24 -1 -1 -1 -1 -1 -1 -1 0
|
||||
*
|
||||
* Hence, when we ask a transport to connect groups, we provide it with a subview of the ranks (except for net
|
||||
* which always sees the full world). That way, P2P can bruteforce all combinations inside the node without
|
||||
* risking to explode in terms of combinations, and we scale better.
|
||||
*
|
||||
* Finally, we loop over Network scores to try to create rings with high scores (=locality) and decrease until
|
||||
* we get at least one ring.
|
||||
*/
|
||||
|
||||
static void recIsConnected(int rank, int* connected, int nranks, int* matrix, int transport) {
|
||||
connected[rank] = 1;
|
||||
for (int r=0; r<nranks; r++) {
|
||||
if (connected[r] == 0 && matrix[rank*nranks+r] == transport) {
|
||||
recIsConnected(r, connected, nranks, matrix, transport);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void isConnected(int rank, int* connected, int nranks, int* matrix, int transport) {
|
||||
for (int r=0; r<nranks; r++) connected[r] = 0;
|
||||
recIsConnected(rank, connected, nranks, matrix, transport);
|
||||
}
|
||||
|
||||
#define NEW_IDX(rank) do { \
|
||||
rankToIdx[rank] = idx; \
|
||||
idxToRank[idx] = rank; \
|
||||
for (int t=0; t<NTRANSPORTS; t++) coords[rank*NTRANSPORTS+t] = current[t]; \
|
||||
idx++; \
|
||||
} while (0)
|
||||
|
||||
int findConnected(int rank, int* matrix, int nranks, int transport, int* coords) {
|
||||
for (int r=0; r<nranks; r++) {
|
||||
if (coords[r*NTRANSPORTS] == -1 && matrix[rank*nranks+r] == transport) return r;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
static ncclResult_t fillCoords(int nranks, int* matrix, int* coords, int* rankToIdx, int* idxToRank) {
|
||||
int current[NTRANSPORTS];
|
||||
int* p2pConnected;
|
||||
NCCLCHECK(ncclCalloc(&p2pConnected, nranks));
|
||||
for (int i=0; i<NTRANSPORTS; i++) current[i] = 0;
|
||||
int curRank = 0, idx = 0;
|
||||
while (1) {
|
||||
// P2P is handled separately as there is no level below it and we need to
|
||||
// cover the case of being connected to another GPU indirectly.
|
||||
// So we detect all GPUs in the same P2P domain once and add them all at
|
||||
// once.
|
||||
isConnected(curRank, p2pConnected, nranks, matrix, 0);
|
||||
for (int r=0; r<nranks; r++) {
|
||||
if (p2pConnected[r]) {
|
||||
NEW_IDX(r);
|
||||
curRank = r;
|
||||
current[0]++;
|
||||
}
|
||||
}
|
||||
current[0] = 0;
|
||||
|
||||
if (idx == nranks) {
|
||||
free(p2pConnected);
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
// Find next group, either connected through SHM or NET.
|
||||
int rank;
|
||||
int transport = 1;
|
||||
while ((rank = findConnected(curRank, matrix, nranks, transport, coords)) == -1) {
|
||||
current[transport] = 0;
|
||||
transport++;
|
||||
if (transport == NTRANSPORTS) { free(p2pConnected); return ncclInternalError; }
|
||||
}
|
||||
curRank = rank;
|
||||
current[transport]++;
|
||||
}
|
||||
}
|
||||
|
||||
NCCL_PARAM(MinNrings, "MIN_NRINGS", 0);
|
||||
NCCL_PARAM(MaxNrings, "MAX_NRINGS", 0);
|
||||
|
||||
/* Users can force the number of threads with an environment variable */
|
||||
NCCL_PARAM(Nthreads, "NTHREADS", -2);
|
||||
ncclResult_t getEnvThreads(int* nthreads) {
|
||||
int64_t nt = ncclParamNthreads();
|
||||
if (nt != -2)
|
||||
*nthreads = nt;
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
/* Main ring creation function */
|
||||
ncclResult_t ncclGetRings(int* nrings, int* nthreads, int rank, int nranks, int* transports, ncclTvalue_t* values, int* prev, int* next) {
|
||||
*nrings = 0;
|
||||
|
||||
if (nranks == 1) return ncclSuccess;
|
||||
|
||||
char* str = getenv("NCCL_RINGS");
|
||||
if (str && strlen(str)>0) {
|
||||
int ret = parseRings(str, nrings, nranks, prev, next);
|
||||
if (ret == ncclSuccess && *nrings > 0) {
|
||||
if (rank == 0) INFO(INIT,"%d ring(s) set by environment", *nrings);
|
||||
NCCLCHECK(getEnvThreads(nthreads));
|
||||
return ncclSuccess;
|
||||
}
|
||||
if (rank == 0) INFO(INIT,"No valid ring found in environment, ignoring");
|
||||
*nrings = 0;
|
||||
}
|
||||
|
||||
// Compute hierarchical topology groups, indexes, and rank<->index tables
|
||||
int* coords, *globalIdxToRank, *globalRankToIdx;
|
||||
NCCLCHECK(ncclCalloc(&coords, nranks*NTRANSPORTS));
|
||||
for (int i=0; i<nranks*NTRANSPORTS; i++) coords[i] = -1;
|
||||
NCCLCHECK(ncclCalloc(&globalIdxToRank, nranks));
|
||||
NCCLCHECK(ncclCalloc(&globalRankToIdx, nranks));
|
||||
|
||||
NCCLCHECK(fillCoords(nranks, transports, coords, globalRankToIdx, globalIdxToRank));
|
||||
|
||||
// Start with a high score, then decrease until we find rings
|
||||
int minScore = NCCL_MAX_SCORE;
|
||||
int nringsTmp;
|
||||
int *prevTmp, *nextTmp, *idxToRank, *rankToIdx, *groups, *subgroups;
|
||||
NCCLCHECK(ncclCalloc(&prevTmp, nranks*MAXRINGS));
|
||||
NCCLCHECK(ncclCalloc(&nextTmp, nranks*MAXRINGS));
|
||||
NCCLCHECK(ncclCalloc(&idxToRank, nranks));
|
||||
NCCLCHECK(ncclCalloc(&rankToIdx, nranks));
|
||||
NCCLCHECK(ncclCalloc(&groups, nranks));
|
||||
NCCLCHECK(ncclCalloc(&subgroups, nranks));
|
||||
|
||||
int nThreads;
|
||||
do {
|
||||
nThreads = *nthreads;
|
||||
for (int i=0; i<nranks*MAXRINGS; i++) prevTmp[i] = nextTmp[i] = -1;
|
||||
nringsTmp = MAXRINGS;
|
||||
// Loop over transports to connect groups
|
||||
for (int t=NTRANSPORTS-1; t>=0; t--) {
|
||||
for (int i=0; i<nranks; i++) idxToRank[i] = rankToIdx[i] = -1;
|
||||
|
||||
int nidx = 0;
|
||||
for (int i=0; i<nranks; i++) {
|
||||
// Extract only ranks in the same local area as rank
|
||||
// We need to extract them in the topological order, hence we iterate over indexes, not ranks
|
||||
int r = globalIdxToRank[i];
|
||||
int sameLocal = 1;
|
||||
for (int tr = NTRANSPORTS-1; tr > t; tr--) if (coords[r*NTRANSPORTS+tr] != coords[rank*NTRANSPORTS+tr]) sameLocal = 0;
|
||||
if (!sameLocal) continue;
|
||||
|
||||
groups[nidx] = coords[r*NTRANSPORTS+t];
|
||||
subgroups[nidx] = t ? coords[r*NTRANSPORTS+t-1] : nidx;
|
||||
rankToIdx[r] = nidx;
|
||||
idxToRank[nidx] = r;
|
||||
nidx++;
|
||||
}
|
||||
|
||||
int ngroups = groups[nidx-1] + 1; // Coords should be ordered
|
||||
|
||||
ncclTvalue_t* subvalues;
|
||||
int *subprev, *subnext;
|
||||
NCCLCHECK(ncclCalloc(&subvalues, nidx*nidx));
|
||||
NCCLCHECK(ncclCalloc(&subprev, nidx*nringsTmp));
|
||||
NCCLCHECK(ncclCalloc(&subnext, nidx*nringsTmp));
|
||||
if (ngroups > 1) {
|
||||
/* Extract subvalues */
|
||||
for (int i=0; i<nidx; i++) {
|
||||
for (int j=0; j<nidx; j++) {
|
||||
if (transports[idxToRank[i]*nranks+idxToRank[j]] == t)
|
||||
subvalues[i*nidx+j] = values[idxToRank[i]*nranks+idxToRank[j]];
|
||||
else
|
||||
subvalues[i*nidx+j] = 0;
|
||||
}
|
||||
}
|
||||
/* Extract subprev/subnext */
|
||||
for (int i=0; i<nidx*nringsTmp; i++) {
|
||||
subprev[i] = subnext[i] = -1;
|
||||
}
|
||||
for (int r=0; r<nringsTmp; r++) {
|
||||
int start = -1, end = -1;
|
||||
for (int i=0; i<nranks; i++) {
|
||||
if (rankToIdx[i] == -1) continue;
|
||||
if (prevTmp[r*nranks+i] != -1) start = i;
|
||||
if (nextTmp[r*nranks+i] != -1) end = i;
|
||||
}
|
||||
if (start != -1 && end != -1) {
|
||||
subprev[r*nidx+rankToIdx[start]] = rankToIdx[end];
|
||||
subnext[r*nidx+rankToIdx[end]] = rankToIdx[start];
|
||||
}
|
||||
}
|
||||
/* Get rings */
|
||||
NCCLCHECK(ncclTransports[t].getRings(nidx, groups, subgroups, subvalues, &nringsTmp, subprev, subnext, minScore, &nThreads));
|
||||
/* Merge subprev/subnext into prev/next */
|
||||
for (int r=0; r<nringsTmp; r++) {
|
||||
for (int i=0; i<nidx; i++) {
|
||||
if ((prevTmp[r*nranks+idxToRank[i]] == -1) && (subprev[r*nidx+i] != -1)) prevTmp[r*nranks+idxToRank[i]] = idxToRank[subprev[r*nidx+i]];
|
||||
if ((nextTmp[r*nranks+idxToRank[i]] == -1) && (subnext[r*nidx+i] != -1)) nextTmp[r*nranks+idxToRank[i]] = idxToRank[subnext[r*nidx+i]];
|
||||
}
|
||||
}
|
||||
//for (int r=0; r<nringsTmp; r++) {
|
||||
//printf("[%d] [%d] [%d] [%d] Prev ", rank, minScore, t, r); for (int i=0; i<nranks; i++) printf("%d ", prevTmp[r*nranks+i]); printf("\n");
|
||||
//printf("[%d] [%d] [%d] [%d] Next ", rank, minScore, t, r); for (int i=0; i<nranks; i++) printf("%d ", nextTmp[r*nranks+i]); printf("\n");
|
||||
//}
|
||||
}
|
||||
free(subvalues);
|
||||
free(subprev);
|
||||
free(subnext);
|
||||
if (nringsTmp == 0) break;
|
||||
}
|
||||
minScore--;
|
||||
if (nringsTmp > *nrings) {
|
||||
*nrings = nringsTmp;
|
||||
for (int i=0; i<nranks*(*nrings); i++) {
|
||||
prev[i] = prevTmp[i];
|
||||
next[i] = nextTmp[i];
|
||||
}
|
||||
}
|
||||
} while (nringsTmp == 0 && minScore);
|
||||
|
||||
free(coords);
|
||||
free(globalRankToIdx);
|
||||
free(globalIdxToRank);
|
||||
free(prevTmp);
|
||||
free(nextTmp);
|
||||
free(idxToRank);
|
||||
free(rankToIdx);
|
||||
free(groups);
|
||||
free(subgroups);
|
||||
|
||||
*nthreads = nThreads;
|
||||
|
||||
if (*nrings == 0) {
|
||||
WARN("Could not create rings, falling back on simple ring");
|
||||
*nrings = 1;
|
||||
prev[rank] = (rank-1+nranks) % nranks;
|
||||
next[rank] = (rank+1)%nranks;
|
||||
}
|
||||
|
||||
int maxNrings = ncclParamMaxNrings();
|
||||
int minNrings = ncclParamMinNrings();
|
||||
if (maxNrings > 0 && minNrings > maxNrings) {
|
||||
if (rank == 0) WARN("NCCL_MIN_NRINGS set to a value greater than NCCL_MAX_NRINGS, ignoring NCCL_MIN_NRINGS");
|
||||
minNrings = 0;
|
||||
}
|
||||
if (minNrings > MAXRINGS) {
|
||||
if (rank == 0) WARN("NCCL_MIN_NRINGS set to a value greater than the maximum number of rings supported (%d), limiting it to %d", MAXRINGS, MAXRINGS);
|
||||
minNrings = MAXRINGS;
|
||||
}
|
||||
if (maxNrings > 0 && maxNrings <= *nrings) {
|
||||
if (rank == 0) INFO(INIT,"Limiting to %d rings per user request.", maxNrings);
|
||||
*nrings = maxNrings;
|
||||
} else {
|
||||
int defaultMinNrings = ncclCudaCompCap() == 3 ? 2 : 1;
|
||||
if (minNrings < defaultMinNrings) minNrings = defaultMinNrings;
|
||||
if (minNrings > 0 && minNrings > *nrings) {
|
||||
if (rank == 0 && minNrings > defaultMinNrings) INFO(INIT,"Duplicating rings to %d per user request.", minNrings);
|
||||
for (int r=*nrings; r<MAXRINGS && r <minNrings; r++) {
|
||||
for (int i=0; i<nranks; i++) {
|
||||
prev[r*nranks+i] = prev[(r-*nrings)*nranks+i];
|
||||
next[r*nranks+i] = next[(r-*nrings)*nranks+i];
|
||||
}
|
||||
}
|
||||
*nrings = minNrings;
|
||||
}
|
||||
}
|
||||
|
||||
NCCLCHECK(getEnvThreads(nthreads));
|
||||
return ncclSuccess;
|
||||
}
|
||||
@@ -0,0 +1,129 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2016-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include "utils.h"
|
||||
#include "debug.h"
|
||||
#include <unistd.h>
|
||||
#include <string.h>
|
||||
|
||||
ncclResult_t getHostName(char* hostname, int maxlen) {
|
||||
if (gethostname(hostname, maxlen) != 0) {
|
||||
strncpy(hostname, "unknown", maxlen);
|
||||
return ncclSystemError;
|
||||
}
|
||||
int i = 0;
|
||||
while ((hostname[i] != '.') && (hostname[i] != '\0') && (i < maxlen-1)) i++;
|
||||
hostname[i] = '\0';
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
uint64_t getHash(const char* string) {
|
||||
// Based on DJB2, result = result * 33 + char
|
||||
uint64_t result = 5381;
|
||||
for (int c = 0; string[c] != '\0'; c++) {
|
||||
result = ((result << 5) + result) + string[c];
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/* Generate a hash of the unique identifying string for this host
|
||||
* that will be unique for both bare-metal and container instances
|
||||
* Equivalent of a hash of;
|
||||
*
|
||||
* $(hostname) $(readlink /proc/self/ns/uts)
|
||||
*/
|
||||
uint64_t getHostHash(void) {
|
||||
char uname[1024];
|
||||
// Start off with the hostname
|
||||
(void) getHostName(uname, sizeof(uname));
|
||||
int hlen = strlen(uname);
|
||||
int len = readlink("/proc/self/ns/uts", uname+hlen, sizeof(uname)-1-hlen);
|
||||
if (len < 0) len = 0;
|
||||
|
||||
uname[hlen+len]='\0';
|
||||
TRACE(INIT,"unique hostname '%s'", uname);
|
||||
|
||||
return getHash(uname);
|
||||
}
|
||||
|
||||
/* Generate a hash of the unique identifying string for this process
|
||||
* that will be unique for both bare-metal and container instances
|
||||
* Equivalent of a hash of;
|
||||
*
|
||||
* $$ $(readlink /proc/self/ns/pid)
|
||||
*/
|
||||
uint64_t getPidHash(void) {
|
||||
char pname[1024];
|
||||
// Start off with our pid ($$)
|
||||
sprintf(pname, "%ld", (long) getpid());
|
||||
int plen = strlen(pname);
|
||||
int len = readlink("/proc/self/ns/pid", pname+plen, sizeof(pname)-1-plen);
|
||||
if (len < 0) len = 0;
|
||||
|
||||
pname[plen+len]='\0';
|
||||
TRACE(INIT,"unique PID '%s'", pname);
|
||||
|
||||
return getHash(pname);
|
||||
}
|
||||
|
||||
int parseStringList(const char* string, struct netIf* ifList, int maxList) {
|
||||
if (!string) return 0;
|
||||
|
||||
const char* ptr = string;
|
||||
// Ignore "^" prefix, will be detected outside of this function
|
||||
if (ptr[0] == '^') ptr++;
|
||||
|
||||
int ifNum = 0;
|
||||
int ifC = 0;
|
||||
char c;
|
||||
do {
|
||||
c = *ptr;
|
||||
if (c == ':') {
|
||||
if (ifC > 0) {
|
||||
ifList[ifNum].prefix[ifC] = '\0';
|
||||
ifList[ifNum].port = atoi(ptr+1);
|
||||
ifNum++; ifC = 0;
|
||||
}
|
||||
while (c != ',' && c != '\0') c = *(++ptr);
|
||||
} else if (c == ',' || c == '\0') {
|
||||
if (ifC > 0) {
|
||||
ifList[ifNum].prefix[ifC] = '\0';
|
||||
ifList[ifNum].port = -1;
|
||||
ifNum++; ifC = 0;
|
||||
}
|
||||
} else {
|
||||
ifList[ifNum].prefix[ifC] = c;
|
||||
ifC++;
|
||||
}
|
||||
ptr++;
|
||||
} while (c);
|
||||
return ifNum;
|
||||
}
|
||||
|
||||
static bool matchPrefix(const char* string, const char* prefix) {
|
||||
return (strncmp(string, prefix, strlen(prefix)) == 0);
|
||||
}
|
||||
|
||||
static bool matchPort(const int port1, const int port2) {
|
||||
if (port1 == -1) return true;
|
||||
if (port2 == -1) return true;
|
||||
if (port1 == port2) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
bool matchIfList(const char* string, int port, struct netIf* ifList, int listSize) {
|
||||
// Make an exception for the case where no user list is defined
|
||||
if (listSize == 0) return true;
|
||||
|
||||
for (int i=0; i<listSize; i++) {
|
||||
if (matchPrefix(string, ifList[i].prefix)
|
||||
&& matchPort(port, ifList[i].port)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
Yeni konuda referans
Bir kullanıcı engelle