Disable Colltrace for --fast option (#778)

* Disable Colltrace for --fast option

* Limit nprocs for CI
This commit is contained in:
Bertan Dogancay
2023-06-21 16:16:09 -04:00
committed by GitHub
parent 399d31ed40
commit 0c77c66221
8 changed files with 85 additions and 43 deletions
+13 -1
View File
@@ -341,6 +341,10 @@ class ncclFunction {
collTrace->type = ncclCollTraceDataType; \
}
#else
#define traceColl(launch_type)
#define traceKernelLaunch(firstLaunch)
#define traceKernelEnd()
#define traceAbort()
#define traceData(data2, data4, data8_0, data8_1)
#endif
@@ -519,7 +523,6 @@ __forceinline__ __device__ void ncclKernel(
}
#endif
if (tid == 0) __insert_timestamp(__LINE__);
if (COLLTRACE && tid == 0) traceKernelLaunch(true);
while (true) {
@@ -563,6 +566,7 @@ __forceinline__ __device__ void ncclKernel(
if (COLLTRACE && tid == 0) traceColl(false);
}
if (COLLTRACE && tid == 0) traceKernelEnd();
#ifdef ENABLE_PROFILING
if (ncclShmem.comm.devProf->seq < PROFILE_NUM_LAUNCHES) {
__synclds();
@@ -572,6 +576,7 @@ __forceinline__ __device__ void ncclKernel(
#endif
}
#ifdef ENABLE_COLLTRACE
#define IMPL_COLL_KERN(func, algo, proto, devredop, type, fIndex) \
__launch_bounds__(NCCL_MAX_NTHREADS, 1) \
__global__ void NCCL_KERN_NAME(func, algo, proto, devredop, type)(struct ncclDevComm* comm, uint64_t channelMask, struct ncclWork* workHead) { \
@@ -582,6 +587,13 @@ __launch_bounds__(NCCL_MAX_NTHREADS, 1) \
__global__ void NCCL_KERN_NAME_DEBUG(func, algo, proto, devredop, type)(struct ncclDevComm* comm, uint64_t channelMask, struct ncclWork* workHead) { \
ncclKernel<ncclFunc##func, type, Func##devredop<type>, NCCL_ALGO_##algo, NCCL_PROTO_##proto, fIndex, true>(comm, channelMask, workHead); \
}
#else
#define IMPL_COLL_KERN(func, algo, proto, devredop, type, fIndex) \
__launch_bounds__(NCCL_MAX_NTHREADS, 1) \
__global__ void NCCL_KERN_NAME(func, algo, proto, devredop, type)(struct ncclDevComm* comm, uint64_t channelMask, struct ncclWork* workHead) { \
ncclKernel<ncclFunc##func, type, Func##devredop<type>, NCCL_ALGO_##algo, NCCL_PROTO_##proto, fIndex, false>(comm, channelMask, workHead); \
}
#endif
// Examples : AllReduce, RING, LL, Sum, uint8
/* Functions for aggregation case */
+7
View File
@@ -28,11 +28,18 @@ struct ncclKernelMatch {
};
typedef void(*ncclKern_t)(struct ncclDevComm* comm, uint64_t channelMask, struct ncclWork* workHead);
// Must be consistent with the ncclFuncSet enum
#ifdef ENABLE_COLLTRACE
static ncclKernelMatch const ncclKerns[2] = {
{(void *)NCCL_KERN_NAME(SendRecv, RING, SIMPLE, Sum, int8_t), true},
{(void *)NCCL_KERN_NAME_DEBUG(SendRecv, RING, SIMPLE, Sum, int8_t), true},
};
#else
static ncclKernelMatch const ncclKerns[1] = {
{(void*)NCCL_KERN_NAME(SendRecv, RING, SIMPLE, Sum, int8_t), true}
};
#endif
static ncclResult_t computeColl(struct ncclInfo* info /* input */, int* workFuncIndex, struct ncclWorkElem* work, struct ncclProxyOp* proxyOp /* output */);
+5 -2
View File
@@ -224,7 +224,10 @@ static float ncclTopoXGMISpeed(int gcn) {
return gcn == 910 ? MI200_XGMI_WIDTH : VEGA_XGMI_WIDTH;
}
#define ncclGetKernelIndex(p_comm) \
((p_comm)->collTraceThread ? 1 : 0)
#if ENABLE_COLLTRACE
#define ncclGetKernelIndex(p_comm) ((p_comm)->collTraceThread ? 1 : 0)
#else
#define ncclGetKernelIndex(p_comm) (0)
#endif
#endif