@@ -0,0 +1,34 @@
|
||||
# Find all C source files in current directory
|
||||
set(SRC_FILES
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/plugin.cc
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/print_event.cc
|
||||
)
|
||||
|
||||
# Create shared library
|
||||
add_library(nccl-profiler-example SHARED ${SRC_FILES})
|
||||
|
||||
# Set include directories
|
||||
target_include_directories(nccl-profiler-example PRIVATE
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/nccl
|
||||
${CUDAToolkit_INCLUDE_DIRS}
|
||||
)
|
||||
|
||||
# Set output name to match Makefile
|
||||
set_target_properties(nccl-profiler-example PROPERTIES
|
||||
OUTPUT_NAME "nccl-profiler-example"
|
||||
PREFIX "lib"
|
||||
POSITION_INDEPENDENT_CODE ON
|
||||
LIBRARY_OUTPUT_DIRECTORY ${CMAKE_BINARY_DIR}/lib
|
||||
)
|
||||
|
||||
add_custom_command(TARGET nccl-profiler-example POST_BUILD
|
||||
COMMAND ${CMAKE_COMMAND} -E make_directory ${CMAKE_BINARY_DIR}/test/unit/plugins
|
||||
COMMAND ${CMAKE_COMMAND} -E copy ${CMAKE_BINARY_DIR}/lib/libnccl-profiler-example.so ${CMAKE_BINARY_DIR}/test/unit/plugins
|
||||
)
|
||||
|
||||
# Add custom target for clean (equivalent to Makefile clean target)
|
||||
add_custom_target(clean-profiler-lib
|
||||
COMMAND ${CMAKE_COMMAND} -E remove -f ${CMAKE_BINARY_DIR}/lib/libnccl-profiler-example.so
|
||||
COMMAND ${CMAKE_COMMAND} -E remove -f ${CMAKE_BINARY_DIR}/test/unit/plugins/libnccl-profiler-example.so
|
||||
COMMENT "Cleaning libnccl-profiler-example.so"
|
||||
)
|
||||
@@ -4,19 +4,26 @@
|
||||
# See LICENSE.txt for license information
|
||||
#
|
||||
.DEFAULT_GOAL: build
|
||||
include ../../makefiles/common.mk
|
||||
SRCDIR ?= $(abspath ../..)
|
||||
ROCM_PATH ?= $(wildcard /opt/rocm)
|
||||
CXX = $(ROCM_PATH)/lib/llvm/bin/amdclang++
|
||||
BUILDDIR ?= .
|
||||
NCCLDIR := $(BUILDDIR)
|
||||
HIPIFY_DIR := hipify-profiler
|
||||
|
||||
SRC_FILES := $(wildcard *.c)
|
||||
SRC_FILES := $(wildcard *.cc)
|
||||
HIPIFY_SRC := $(addprefix $(HIPIFY_DIR)/,$(SRC_FILES))
|
||||
|
||||
build: ${BUILDDIR}/librccl-profiler.so
|
||||
build: ${BUILDDIR}/librccl-profiler-example.so
|
||||
|
||||
${BUILDDIR}/librccl-profiler.so: ${SRC_FILES}
|
||||
${BUILDDIR}/librccl-profiler-example.so: $(HIPIFY_SRC)
|
||||
@printf "Compiling %-35s > %s\n" $< $@
|
||||
@mkdir -p ${BUILDDIR}
|
||||
$(CC) -Inccl -fPIC -shared -o $@ $^
|
||||
$(CXX) -D__HIP_PLATFORM_AMD__ -I$(HIPIFY_DIR) -I$(HIPIFY_DIR)/nccl -I$(ROCM_PATH)/include -fPIC -shared -o $@ $^
|
||||
|
||||
$(HIPIFY_DIR)/%.cc: %.cc
|
||||
@mkdir -p $(HIPIFY_DIR)/nccl
|
||||
@cp *.cc *.h $(HIPIFY_DIR)/
|
||||
@cp nccl/*.h $(HIPIFY_DIR)/nccl/
|
||||
@hipify-perl -inplace -quiet-warnings $(HIPIFY_DIR)/*.cc $(HIPIFY_DIR)/*.h
|
||||
|
||||
clean:
|
||||
rm -f ${BUILDDIR}/librccl-profiler.so
|
||||
rm -rf ${BUILDDIR}/librccl-profiler-example.so $(HIPIFY_DIR)
|
||||
@@ -13,8 +13,7 @@ change the size of the event window the profiler keeps track of.
|
||||
|
||||
## Building the profiler plugin
|
||||
|
||||
To use the example plugin, just type `make`. You will need a NCCL build's include directory present.
|
||||
You can override `NCCL_HOME` to where the NCCL installation is on your system.
|
||||
To build the example plugin shipped as part of NCCL, just type `make`.
|
||||
|
||||
## Using the profiler plugin
|
||||
|
||||
@@ -27,13 +26,13 @@ You can override `NCCL_HOME` to where the NCCL installation is on your system.
|
||||
|
||||
As an example, setting:
|
||||
|
||||
`NCCL_PROFILE_EVENT_MASK` to 1 (`ncclProfileGroup`) | 2 (`ncclProfileColl`) | 8 (`ncclProfileProxyOp`)
|
||||
`NCCL_PROFILE_EVENT_MASK` to 256 (`ncclProfileGroupApi`) | 2 (`ncclProfileColl`) | 8 (`ncclProfileProxyOp`)
|
||||
|
||||
enables the profiling of the group, the collective and the proxy op events. The same events can be
|
||||
enables the profiling of the group API, the collective and the proxy op events. The same events can be
|
||||
expressed more concisely by setting `NCCL_PROFILE_EVENT_MASK` to 8 (`ncclProfileProxyOp`). Indeed,
|
||||
in NCCL all the events above (in the event hierarchy) the one requested are also captured. The advantage
|
||||
is that the profiler can easily correlate events that belong to the same NCCL operation and present
|
||||
them accordingly.
|
||||
them accordingly. Setting `NCCL_PROFILE_EVENT_MASK` to 4095 enables all events supported by the v5 profiler.
|
||||
|
||||
3. Set `NCCL_PROFILE_DUMP_FILE` to the name of the dump file for the collected traces. A file named
|
||||
${NCCL_PROFILE_DUMP_FILE}-hostname-tid.txt is created. Profiler traces are saved using the chrome
|
||||
@@ -57,11 +56,14 @@ The group, collective and p2p pools contain objects for the corresponding events
|
||||
contains objects for `ProxyCtrl` events and the `ProxyDetach` pool contains objects for `ProxyOp` events
|
||||
generated by remote proxies. A list of pools and their size is reported below:
|
||||
|
||||
- `NCCL_PROFILE_GROUP_POOL_SIZE` (16)
|
||||
- `NCCL_PROFILE_COLL_POOL_SIZE` (16)
|
||||
- `NCCL_PROFILE_P2P_POOL_SIZE` (1024)
|
||||
- `NCCL_PROFILE_GROUP_API_POOL_SIZE` (256)
|
||||
- `NCCL_PROFILE_COLL_API_POOL_SIZE` (256)
|
||||
- `NCCL_PROFILE_P2P_API_POOL_SIZE` (256)
|
||||
- `NCCL_PROFILE_KERNEL_LAUNCH_POOL_SIZE` (256)
|
||||
- `NCCL_PROFILE_COLL_POOL_SIZE` (256)
|
||||
- `NCCL_PROFILE_P2P_POOL_SIZE` (256)
|
||||
- `NCCL_PROFILE_PROXY_CTRL_POOL_SIZE` (16)
|
||||
- `NCCL_PROFILE_PROXY_DETACH_POOL_SIZE` (128)
|
||||
- `NCCL_PROFILE_PROXY_DETACH_POOL_SIZE` (256)
|
||||
|
||||
Remote proxy operations are generated when PXN is in use. Refer to this article for more information
|
||||
about PXN and how it works:
|
||||
@@ -73,76 +75,58 @@ The example profiler generates traces using the json format. An example of trace
|
||||
|
||||
```
|
||||
[
|
||||
{"name": "Group", "cat": "GROUP", "ph": "b", "id": 0, "pid": 4157654, "tid": 1, "ts": 764234.611328, "args": {"groupId": 0}},
|
||||
{"name": "AllReduce", "cat": "COLL", "ph": "b", "id": 0, "pid": 4157654, "tid": 1, "ts": 764237.294922, "args": {"SeqNum": 0, "CommHash": 673864846479792718, "Rank": 1, "Count": 32768, "Datatype": "ncclFloat32", "Algorithm": "RING", "Protocol": "LL", "nMaxChannels": 2}},
|
||||
{"name": "Recv", "cat": "PROXY", "ph": "b", "id": 0, "pid": 4157654, "tid": 1, "ts": 768464.936523, "args": {"Channel": 0, "Peer": 0, "Steps": 14, "ChunkSize": 32768, "transSize": 229376, "POSTED": {"step": 14, "ts": 772020.300781}, "RECEIVED": {"step": 14, "ts": 772196.049805}, "TRANSMITTED": {"step": 14, "ts": 772197.326172}, "DONE": {"step": 14, "ts": 772201.538086}}},
|
||||
{"name": "RecvBufferWait", "cat": "NET", "ph": "b", "id": 0, "pid": 4157654, "tid": 1, "ts": 768465.158203, "args": {"Step": 0}},
|
||||
{"name": "RecvBufferWait", "cat": "NET", "ph": "e", "id": 0, "pid": 4157654, "tid": 1, "ts": 768477.924805},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "b", "id": 0, "pid": 4157654, "tid": 1, "ts": 768477.924805, "args": {"Step": 0}},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "e", "id": 0, "pid": 4157654, "tid": 1, "ts": 768547.197266},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "b", "id": 0, "pid": 4157654, "tid": 1, "ts": 768547.197266, "args": {"Step": 0}},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "e", "id": 0, "pid": 4157654, "tid": 1, "ts": 768564.174805},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "b", "id": 0, "pid": 4157654, "tid": 1, "ts": 768564.174805, "args": {"Step": 0}},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "e", "id": 0, "pid": 4157654, "tid": 1, "ts": 768568.276367},
|
||||
{"name": "RecvBufferWait", "cat": "NET", "ph": "b", "id": 1, "pid": 4157654, "tid": 1, "ts": 768503.604492, "args": {"Step": 1}},
|
||||
{"name": "RecvBufferWait", "cat": "NET", "ph": "e", "id": 1, "pid": 4157654, "tid": 1, "ts": 768504.549805},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "b", "id": 1, "pid": 4157654, "tid": 1, "ts": 768504.549805, "args": {"Step": 1}},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "e", "id": 1, "pid": 4157654, "tid": 1, "ts": 769994.490234},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "b", "id": 1, "pid": 4157654, "tid": 1, "ts": 769994.490234, "args": {"Step": 1}},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "e", "id": 1, "pid": 4157654, "tid": 1, "ts": 769995.012695},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "b", "id": 1, "pid": 4157654, "tid": 1, "ts": 769995.012695, "args": {"Step": 1}},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "e", "id": 1, "pid": 4157654, "tid": 1, "ts": 770006.914062},
|
||||
{"name": "RecvBufferWait", "cat": "NET", "ph": "b", "id": 2, "pid": 4157654, "tid": 1, "ts": 768506.941406, "args": {"Step": 2}},
|
||||
{"name": "RecvBufferWait", "cat": "NET", "ph": "e", "id": 2, "pid": 4157654, "tid": 1, "ts": 768507.435547},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "b", "id": 2, "pid": 4157654, "tid": 1, "ts": 768507.435547, "args": {"Step": 2}},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "e", "id": 2, "pid": 4157654, "tid": 1, "ts": 771452.536133},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "b", "id": 2, "pid": 4157654, "tid": 1, "ts": 771452.536133, "args": {"Step": 2}},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "e", "id": 2, "pid": 4157654, "tid": 1, "ts": 771453.060547},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "b", "id": 2, "pid": 4157654, "tid": 1, "ts": 771453.060547, "args": {"Step": 2}},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "e", "id": 2, "pid": 4157654, "tid": 1, "ts": 771468.458008},
|
||||
{"name": "RecvBufferWait", "cat": "NET", "ph": "b", "id": 3, "pid": 4157654, "tid": 1, "ts": 768509.484375, "args": {"Step": 3}},
|
||||
{"name": "RecvBufferWait", "cat": "NET", "ph": "e", "id": 3, "pid": 4157654, "tid": 1, "ts": 768510.250000},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "b", "id": 3, "pid": 4157654, "tid": 1, "ts": 768510.250000, "args": {"Step": 3}},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "e", "id": 3, "pid": 4157654, "tid": 1, "ts": 771904.499023},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "b", "id": 3, "pid": 4157654, "tid": 1, "ts": 771904.499023, "args": {"Step": 3}},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "e", "id": 3, "pid": 4157654, "tid": 1, "ts": 771904.991211},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "b", "id": 3, "pid": 4157654, "tid": 1, "ts": 771904.991211, "args": {"Step": 3}},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "e", "id": 3, "pid": 4157654, "tid": 1, "ts": 771910.500000},
|
||||
{"name": "Send", "cat": "PROXY", "ph": "b", "id": 1, "pid": 4157654, "tid": 1, "ts": 768482.878906, "args": {"Channel": 0, "Peer": 2, "Steps": 14, "ChunkSize": 32768, "transSize": 229376, "POSTED": {"step": 14, "ts": 771995.675781}, "REM_FIFO_WAIT": {"step": 14, "ts": 772190.692383}, "TRANSMITTED": {"step": 14, "ts": 772191.516602}, "DONE": {"step": 14, "ts": 772208.473633}}},
|
||||
{"name": "SendBufferWait", "cat": "NET", "ph": "b", "id": 14, "pid": 4157654, "tid": 1, "ts": 768483.019531, "args": {"Step": 0}},
|
||||
{"name": "SendBufferWait", "cat": "NET", "ph": "e", "id": 14, "pid": 4157654, "tid": 1, "ts": 768483.300781},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "b", "id": 14, "pid": 4157654, "tid": 1, "ts": 768483.300781, "args": {"Step": 0}},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "e", "id": 14, "pid": 4157654, "tid": 1, "ts": 769594.615234},
|
||||
{"name": "SendWait", "cat": "NET", "ph": "b", "id": 14, "pid": 4157654, "tid": 1, "ts": 769594.615234, "args": {"Step": 0}},
|
||||
{"name": "SendWait", "cat": "NET", "ph": "e", "id": 14, "pid": 4157654, "tid": 1, "ts": 769618.889648},
|
||||
{"name": "SendBufferWait", "cat": "NET", "ph": "b", "id": 15, "pid": 4157654, "tid": 1, "ts": 768505.083008, "args": {"Step": 1}},
|
||||
{"name": "SendBufferWait", "cat": "NET", "ph": "e", "id": 15, "pid": 4157654, "tid": 1, "ts": 768505.163086},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "b", "id": 15, "pid": 4157654, "tid": 1, "ts": 768505.163086, "args": {"Step": 1}},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "e", "id": 15, "pid": 4157654, "tid": 1, "ts": 769610.555664},
|
||||
{"name": "SendWait", "cat": "NET", "ph": "b", "id": 15, "pid": 4157654, "tid": 1, "ts": 769610.555664, "args": {"Step": 1}},
|
||||
{"name": "SendWait", "cat": "NET", "ph": "e", "id": 15, "pid": 4157654, "tid": 1, "ts": 769622.517578},
|
||||
{"name": "SendBufferWait", "cat": "NET", "ph": "b", "id": 16, "pid": 4157654, "tid": 1, "ts": 768507.937500, "args": {"Step": 2}},
|
||||
{"name": "SendBufferWait", "cat": "NET", "ph": "e", "id": 16, "pid": 4157654, "tid": 1, "ts": 768508.017578},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "b", "id": 16, "pid": 4157654, "tid": 1, "ts": 768508.017578, "args": {"Step": 2}},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "e", "id": 16, "pid": 4157654, "tid": 1, "ts": 770002.129883},
|
||||
{"name": "SendWait", "cat": "NET", "ph": "b", "id": 16, "pid": 4157654, "tid": 1, "ts": 770002.129883, "args": {"Step": 2}},
|
||||
{"name": "SendWait", "cat": "NET", "ph": "e", "id": 16, "pid": 4157654, "tid": 1, "ts": 770013.848633},
|
||||
{"name": "SendBufferWait", "cat": "NET", "ph": "b", "id": 17, "pid": 4157654, "tid": 1, "ts": 768510.742188, "args": {"Step": 3}},
|
||||
{"name": "SendBufferWait", "cat": "NET", "ph": "e", "id": 17, "pid": 4157654, "tid": 1, "ts": 768510.822266},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "b", "id": 17, "pid": 4157654, "tid": 1, "ts": 768510.822266, "args": {"Step": 3}},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "e", "id": 17, "pid": 4157654, "tid": 1, "ts": 771461.563477},
|
||||
{"name": "SendWait", "cat": "NET", "ph": "b", "id": 17, "pid": 4157654, "tid": 1, "ts": 771461.563477, "args": {"Step": 3}},
|
||||
{"name": "SendWait", "cat": "NET", "ph": "e", "id": 17, "pid": 4157654, "tid": 1, "ts": 771469.171875},
|
||||
{"name": "Group API", "cat": "GROUP_API", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 3433.595001, "args": {"groupApiId": 0, "groupDepth":1}},
|
||||
{"name": "KernelLaunch", "cat": "KERNEL_LAUNCH", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 0.000000, "args": {"groupId": 0, "Stream": 0x5020000567d0}},
|
||||
{"name": "KernelLaunch", "cat": "KERNEL_LAUNCH", "ph": "e", "id": 0, "pid": 225798, "tid": 1, "ts": 111991.558990},
|
||||
{"name": "AllReduce", "cat": "COLL_API", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 0.000000, "args": {"count": 262144, "datatype": ncclFloat32, "root": 0, "GraphCaptured":0, "Stream": 0x5020000567d0}},
|
||||
{"name": "AllReduce", "cat": "COLL", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 111994.477997, "args": {"SeqNum": 0, "CommHash": 1493613951195738943, "Rank": 0, "Count": 262144, "Datatype": "ncclFloat32", "Algorithm": "RING", "Protocol": "SIMPLE", "nChannels": 2}},
|
||||
{"name": "KernelCh", "cat": "GPU", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 119711.888000, "args": {"Channel": 0, "StartGpuClk": 1756135989724672000, "StopGpuClk": 1756135989732831232}},
|
||||
{"name": "ScheduleRecv", "cat": "PROXY", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 119652.709991, "args": {"Channel": 0, "Peer": 1, "Steps": 4, "ChunkSize": 4194304, "transSize": 524288}},
|
||||
{"name": "ScheduleRecv", "cat": "PROXY", "ph": "e", "id": 0, "pid": 225798, "tid": 1, "ts": 119686.300995},
|
||||
{"name": "ProgressRecv", "cat": "PROXY", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 119686.300995, "args": {"Channel": 0, "Peer": 1, "Steps": 4, "ChunkSize": 4194304, "transSize": 524288}},
|
||||
{“name": "RecvWait", "cat": "NET", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 119707.677979, "args": {"Step": 0}},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "e", "id": 0, "pid": 225798, "tid": 1, "ts": 119807.691986},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 119807.691986, "args": {"Step": 0}},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "e", "id": 0, "pid": 225798, "tid": 1, "ts": 119867.338989},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "b", "id": 0, "pid": 225798, "tid": 1, "ts": 119867.338989, "args": {"Step": 0}},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "e", "id": 0, "pid": 225798, "tid": 1, "ts": 120120.983002},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "b", "id": 1, "pid": 225798, "tid": 1, "ts": 119733.647980, "args": {"Step": 1}},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "e", "id": 1, "pid": 225798, "tid": 1, "ts": 119844.401001},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "b", "id": 1, "pid": 225798, "tid": 1, "ts": 119844.401001, "args": {"Step": 1}},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "e", "id": 1, "pid": 225798, "tid": 1, "ts": 119890.567993},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "b", "id": 1, "pid": 225798, "tid": 1, "ts": 119890.567993, "args": {"Step": 1}},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "e", "id": 1, "pid": 225798, "tid": 1, "ts": 120121.129974},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "b", "id": 2, "pid": 225798, "tid": 1, "ts": 119753.023987, "args": {"Step": 2}},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "e", "id": 2, "pid": 225798, "tid": 1, "ts": 120038.847992},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "b", "id": 2, "pid": 225798, "tid": 1, "ts": 120038.847992, "args": {"Step": 2}},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "e", "id": 2, "pid": 225798, "tid": 1, "ts": 120085.685974},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "b", "id": 2, "pid": 225798, "tid": 1, "ts": 120085.685974, "args": {"Step": 2}},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "e", "id": 2, "pid": 225798, "tid": 1, "ts": 120121.244995},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "b", "id": 3, "pid": 225798, "tid": 1, "ts": 119772.510986, "args": {"Step": 3}},
|
||||
{"name": "RecvWait", "cat": "NET", "ph": "e", "id": 3, "pid": 225798, "tid": 1, "ts": 120062.944977},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "b", "id": 3, "pid": 225798, "tid": 1, "ts": 120062.944977, "args": {"Step": 3}},
|
||||
{"name": "RecvFlushWait", "cat": "NET", "ph": "e", "id": 3, "pid": 225798, "tid": 1, "ts": 120101.089996},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "b", "id": 3, "pid": 225798, "tid": 1, "ts": 120101.089996, "args": {"Step": 3}},
|
||||
{"name": "RecvGpuWait", "cat": "NET", "ph": "e", "id": 3, "pid": 225798, "tid": 1, "ts": 120165.115997},
|
||||
{"name": "ProgressRecv", "cat": "PROXY", "ph": "e", "id": 0, "pid": 225798, "tid": 1, "ts": 120165.356995},
|
||||
{"name": "ScheduleSend", "cat": "PROXY", "ph": "b", "id": 1, "pid": 225798, "tid": 1, "ts": 119656.950989, "args": {"Channel": 0, "Peer": 1, "Steps": 4, "ChunkSize": 4194304, "transSize": 524288}},
|
||||
{"name": "ScheduleSend", "cat": "PROXY", "ph": "e", "id": 1, "pid": 225798, "tid": 1, "ts": 119709.078979},
|
||||
{"name": "ProgressSend", "cat": "PROXY", "ph": "b", "id": 1, "pid": 225798, "tid": 1, "ts": 119709.078979, "args": {"Channel": 0, "Peer": 1, "Steps": 4, "ChunkSize": 4194304, "transSize": 524288}},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "b", "id": 4, "pid": 225798, "tid": 1, "ts": 119710.632996, "args": {"Step": 0}},
|
||||
{"name": "SendGpuWait", "cat": "NET", "ph": "e", "id": 4, "pid": 225798, "tid": 1, "ts": 119808.636993},
|
||||
{"name": "SendPeerWait", "cat": "NET", "ph": "b", "id": 4, "pid": 225798, "tid": 1, "ts": 119808.636993, "args": {"Step": 0}},
|
||||
{"name": "SendPeerWait", "cat": "NET", "ph": "e", "id": 4, "pid": 225798, "tid": 1, "ts": 119818.972992},
|
||||
... [ trace truncated for brevity ]
|
||||
{"name": "AllReduce", "cat": "COLL", "ph": "e", "id": 0, "pid": 4157654, "tid": 1, "ts": 772209.317383},
|
||||
{"name": "Group", "cat": "GROUP", "ph": "e", "id": 0, "pid": 4157654, "tid": 1, "ts": 772209.418945},
|
||||
{"name": "AllReduce", "cat": "COLL", "ph": "e", "id": 17, "pid": 225798, "tid": 1, "ts": 170633.535980},
|
||||
{"name": "AllReduce", "cat": "COLL_API", "ph": "e", "id": 17, "pid": 225798, "tid": 1, "ts": 170582.923981},
|
||||
{"name": "Group API", "cat": "GROUP_API", "ph": "e", "id": 17, "pid": 225798, "tid": 1, "ts": 170637.582001},
|
||||
{}]
|
||||
```
|
||||
|
||||
Details about the fields used in the trace can be found at this link:
|
||||
https://docs.google.com/document/d/1CvAClvFfyA5R-PhYUmn5OOQtYMH4h6I0nSsKchNAySU/preview?tab=t.0#heading=h.yr4qxyxotyw
|
||||
|
||||
The trace above is obtained by running a `ncclAllReduce` operation on 8 GPUs, communicating with each other through
|
||||
The trace above is obtained by running a `ncclAllReduce` operation on 2 GPUs, communicating with each other through
|
||||
the network interface. The `Group` event encloses all traces that are related to the single `ncclAllReduce` call.
|
||||
(Note that for single collective invocations, where there are no explicit group calls, NCCL creates a group with only
|
||||
one collective and this is what is presented in the traces above).
|
||||
@@ -161,38 +145,17 @@ The `AllReduce` entry presents information about the `ncclAllReduce` operation.
|
||||
- datatype : NCCL datatype
|
||||
- algorithm : algorithm used to process the ncclAllReduce
|
||||
- protocol : protocol used to process the ncclAllReduce
|
||||
- nMaxChannels: max number of channels used to process the ncclAllReduce
|
||||
- nChannels : Number of channels used to process the ncclAllReduce
|
||||
|
||||
If the proxy events are not active (e.g., the `ncclAllReduce` is intranode) the end timestamp will match the time
|
||||
consumed by the CPU to launch the collective. For more details refer to `ext-profiler/README.md`, section `Profiling
|
||||
of collective and p2p operations`.
|
||||
|
||||
### Proxy Send
|
||||
The `Send` entry presents information about the `ProxyOp` processing in the progress thread. It contains the following
|
||||
info in the args field:
|
||||
|
||||
- Channel : id of the channel used by this proxy operation to send data to the peer
|
||||
- Peer : peer rank
|
||||
- Steps : number of network steps required to transfer transSize bytes to the peer
|
||||
- ChunkSize : chunk size used by NCCL to pipeline data through the proxy thread
|
||||
- transSize : bytes transferred across the channel by this proxy operation
|
||||
- POSTED : struct containing the number of buffer posts to the GPU and the time stamp for the last post
|
||||
- REM_FIFO_WAIT: struct containing the number of remote buffer waits and the time stamp for the last wait
|
||||
- TRANSMITTED : struct containing the number of network sends and the time stamp of the last send
|
||||
- DONE : struct containing the number of network sends completed and the time stamp of the last send completed
|
||||
|
||||
In case of a network problem the POSTED, REM_FIFO_WAIT, TRANSMITTED and DONE might all have partially updated steps,
|
||||
which could help identify at which point the network problem occurred.
|
||||
|
||||
The Proxy send trace gives a summary of the proxy progress thread activity for the channel. If more details are
|
||||
needed, these can be obtained by enabling the proxy step event (`ncclProfileProxyStep`). In which case the trace
|
||||
entries below are also reported by the profiler.
|
||||
|
||||
#### Proxy SendBufferWait
|
||||
|
||||
Presents, for every network step, the time the CPU proxy spends waiting for the channel staging buffer to become available.
|
||||
|
||||
#### Proxy SendGPUWait
|
||||
#### Proxy SendGpuWait
|
||||
|
||||
Presents, for every network step, the time the CPU proxy spends waiting for the GPU to provide the data in the staging
|
||||
buffer.
|
||||
@@ -201,31 +164,6 @@ buffer.
|
||||
|
||||
Presents, for every network step, the time the CPU proxy spends waiting for the `isend` to complete
|
||||
|
||||
### Proxy Recv
|
||||
|
||||
The `Recv` entry presents information about the `ProxyOp` processing in the progress thread. It contains the following
|
||||
info in the args field:
|
||||
|
||||
- Channel : id of the channel used by this proxy operation to recv data from the peer
|
||||
- Peer : peer rank
|
||||
- Steps : number of network steps required to transfer transSize bytes from the peer
|
||||
- ChunkSize : chunk size used by NCCL to pipeline data through the proxy thread
|
||||
- transSize : bytes transferred across the channel by this proxy operation
|
||||
- POSTED : struct containing the number of recvs posted and the time stamp for the last recv posted
|
||||
- RECEIVED : struct containing the number of recvs completed and the time stamp for the last recv completed
|
||||
- TRANSMITTED: struct containing the number of recvs flushed to the GPU memory and the time stamp for the last recv flushed
|
||||
- DONE : struct containing the number of flush completed and the time stamp for the last flush completed
|
||||
|
||||
The Proxy Recv trace gives a summary of the proxy progress thread activity for the channel. If more details are
|
||||
needed, these can be obtained by enabling the proxy step event (`ncclProfileProxyStep`). In which case the trace
|
||||
entries below are also reported by the profiler.
|
||||
|
||||
|
||||
#### Proxy RecvBufferWait
|
||||
|
||||
Presents, for every network step, the time the CPU proxy spends waiting for the staging buffer for the channel to
|
||||
become available.
|
||||
|
||||
#### Proxy RecvWait
|
||||
|
||||
Presents, for every network step, the time the CPU proxy spends waiting for a posted `irecv` to complete
|
||||
@@ -234,6 +172,6 @@ Presents, for every network step, the time the CPU proxy spends waiting for a po
|
||||
|
||||
Presents, for every network step, the time the CPU proxy spends waitng for the recv data to be flushed to the GPU
|
||||
|
||||
#### Proxy RecvGPUWait
|
||||
#### Proxy RecvGpuWait
|
||||
|
||||
Presents, for every network step, the time the CPU proxy spends waiting for the GPU to consume the recv data
|
||||
|
||||
@@ -1,30 +0,0 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2024, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "event.h"
|
||||
|
||||
int taskEventQueueEmpty(struct group* g) {
|
||||
return g->eventHead == NULL;
|
||||
}
|
||||
|
||||
void taskEventQueueEnqueue(struct group* g, struct taskEventBase* event) {
|
||||
event->next = NULL;
|
||||
if (g->eventHead) g->eventTail->next = event;
|
||||
else g->eventHead = event;
|
||||
g->eventTail = event;
|
||||
}
|
||||
|
||||
struct taskEventBase* taskEventQueueHead(struct group* g) {
|
||||
return g->eventHead;
|
||||
}
|
||||
|
||||
struct taskEventBase* taskEventQueueDequeue(struct group* g) {
|
||||
struct taskEventBase* tmp = g->eventHead;
|
||||
g->eventHead = g->eventHead->next;
|
||||
if (g->eventHead == NULL) g->eventTail = NULL;
|
||||
return tmp;
|
||||
}
|
||||
@@ -10,10 +10,14 @@
|
||||
#include <sys/types.h>
|
||||
#include <stdint.h>
|
||||
#include <unistd.h>
|
||||
#include <cstring>
|
||||
#include "err.h"
|
||||
#include "profiler.h"
|
||||
#include "queue.h"
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#define MAX_CHANNELS 128 // Match RCCL's MAXCHANNELS
|
||||
#define MAX_STEPS 16
|
||||
#define MAX_STEPS 1024
|
||||
#define MAX_OPS 16 // Up to 64K ranks for PAT
|
||||
#define MAX_EVENTS_PER_REQ (8)
|
||||
|
||||
@@ -21,7 +25,7 @@ struct proxyOp;
|
||||
struct proxyStep;
|
||||
|
||||
struct netPlugin {
|
||||
uint8_t type;
|
||||
uint64_t type;
|
||||
int pluginType;
|
||||
int pluginVer;
|
||||
uint8_t pluginEvent;
|
||||
@@ -63,7 +67,7 @@ struct kernelCh {
|
||||
#define PROXY_STEP_MAX_STATES 3
|
||||
|
||||
struct proxyStep {
|
||||
uint8_t type; // type of event: network transfer
|
||||
uint64_t type; // type of event: network transfer
|
||||
int state;
|
||||
int step; // network transfer id in given channel
|
||||
int isSend; // send/recv channel operation
|
||||
@@ -76,7 +80,7 @@ struct proxyStep {
|
||||
};
|
||||
|
||||
struct proxyOp {
|
||||
uint8_t type; // type of event: proxy operation
|
||||
uint64_t type; // type of event: proxy operation
|
||||
uint8_t channelId; // channel id for this proxy operation
|
||||
pid_t pid;
|
||||
int rank;
|
||||
@@ -97,7 +101,7 @@ struct group;
|
||||
struct context;
|
||||
|
||||
struct proxyCtrl {
|
||||
uint8_t type;
|
||||
uint64_t type;
|
||||
struct context* ctx; // profiler context
|
||||
double startTs;
|
||||
double stopTs;
|
||||
@@ -107,12 +111,12 @@ struct proxyCtrl {
|
||||
|
||||
// task level event base structure
|
||||
struct taskEventBase {
|
||||
uint8_t type; // event type: collective/p2p
|
||||
uint64_t type; // event type: collective/p2p
|
||||
int rank; // rank of the operation in NCCL communicator
|
||||
const char* func; // ncclFunc*
|
||||
int refCount; // number of references for this operation
|
||||
struct group* parent; // parent event group
|
||||
struct taskEventBase* next; // next top level event in group
|
||||
void* parent; // parent API event
|
||||
struct taskEventBase* next; // next top level event
|
||||
double startTs;
|
||||
double stopTs;
|
||||
};
|
||||
@@ -147,7 +151,7 @@ struct p2p {
|
||||
};
|
||||
|
||||
struct group {
|
||||
uint8_t type;
|
||||
uint64_t type;
|
||||
struct context* ctx; // profiler context
|
||||
int groupId;
|
||||
int refCount;
|
||||
@@ -158,6 +162,70 @@ struct group {
|
||||
struct group* next; // next group event in queue
|
||||
};
|
||||
|
||||
struct collApi {
|
||||
uint64_t type;
|
||||
struct groupApi* parent;
|
||||
struct context* ctx; // profiler context
|
||||
int collApiId;
|
||||
int refCount;
|
||||
cudaStream_t stream;
|
||||
const char* func;
|
||||
size_t count;
|
||||
const char* datatype;
|
||||
int root;
|
||||
bool graphCaptured;
|
||||
struct taskEventBase* eventHead; // queue head for task events
|
||||
struct taskEventBase* eventTail; // queue tail for task events
|
||||
double startTs;
|
||||
double stopTs;
|
||||
struct collApi* next;
|
||||
};
|
||||
|
||||
struct p2pApi {
|
||||
uint64_t type;
|
||||
struct groupApi* parent;
|
||||
struct context* ctx; // profiler context
|
||||
int p2pApiId;
|
||||
int refCount;
|
||||
const char* func;
|
||||
cudaStream_t stream;
|
||||
size_t count;
|
||||
const char* datatype;
|
||||
bool graphCaptured;
|
||||
struct taskEventBase* eventHead; // queue head for task events
|
||||
struct taskEventBase* eventTail; // queue tail for task events
|
||||
double startTs;
|
||||
double stopTs;
|
||||
struct p2pApi* next;
|
||||
};
|
||||
|
||||
struct kernelLaunch {
|
||||
uint64_t type;
|
||||
struct groupApi* parent;
|
||||
cudaStream_t stream;
|
||||
int kernelLaunchId;
|
||||
double startTs;
|
||||
double stopTs;
|
||||
struct kernelLaunch* next;
|
||||
};
|
||||
|
||||
struct groupApi {
|
||||
uint64_t type;
|
||||
struct context* ctx;
|
||||
int groupApiId;
|
||||
int refCount;
|
||||
bool graphCaptured;
|
||||
int groupDepth;
|
||||
struct profilerQueue<struct p2pApi, &p2pApi::next> p2pApiEvents;
|
||||
struct profilerQueue<struct collApi, &collApi::next> collApiEvents;
|
||||
struct profilerQueue<struct kernelLaunch, &kernelLaunch::next> kernelLaunchEvents;
|
||||
double endOfncclGroupStartTs;
|
||||
double startOfncclGroupEndTs;
|
||||
double startTs;
|
||||
double stopTs;
|
||||
struct groupApi* next;
|
||||
};
|
||||
|
||||
// arrays for different event objects
|
||||
struct context {
|
||||
const char* commName;
|
||||
@@ -165,6 +233,26 @@ struct context {
|
||||
int nranks;
|
||||
int rank;
|
||||
|
||||
int groupApiPoolSize;
|
||||
int groupApiPoolBase;
|
||||
int groupApiPoolIndex;
|
||||
struct groupApi* groupApiPool;
|
||||
|
||||
int collApiPoolSize;
|
||||
int collApiPoolBase;
|
||||
int collApiPoolIndex;
|
||||
struct collApi* collApiPool;
|
||||
|
||||
int p2pApiPoolSize;
|
||||
int p2pApiPoolBase;
|
||||
int p2pApiPoolIndex;
|
||||
struct p2pApi* p2pApiPool;
|
||||
|
||||
int kernelLaunchPoolSize;
|
||||
int kernelLaunchPoolBase;
|
||||
int kernelLaunchPoolIndex;
|
||||
struct kernelLaunch* kernelLaunchPool;
|
||||
|
||||
int groupPoolSize;
|
||||
int groupPoolBase;
|
||||
int groupPoolIndex;
|
||||
@@ -186,9 +274,50 @@ struct context {
|
||||
struct proxyCtrl* proxyCtrlPool;
|
||||
};
|
||||
|
||||
int taskEventQueueEmpty(struct group* g);
|
||||
void taskEventQueueEnqueue(struct group* g, struct taskEventBase* event);
|
||||
struct taskEventBase* taskEventQueueHead(struct group* g);
|
||||
struct taskEventBase* taskEventQueueDequeue(struct group* g);
|
||||
template <typename T>
|
||||
inline int taskEventQueueEmpty(T *obj) {
|
||||
return obj->eventHead == NULL;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
inline void taskEventQueueEnqueue(T* obj, struct taskEventBase* event) {
|
||||
event->next = NULL;
|
||||
if (obj->eventHead) obj->eventTail->next = event;
|
||||
else obj->eventHead = event;
|
||||
obj->eventTail = event;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
inline struct taskEventBase* taskEventQueueHead(T *obj) {
|
||||
return obj->eventHead;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
inline struct taskEventBase* taskEventQueueDequeue(T* obj) {
|
||||
struct taskEventBase* tmp = obj->eventHead;
|
||||
obj->eventHead = obj->eventHead->next;
|
||||
if (obj->eventHead == NULL) obj->eventTail = NULL;
|
||||
return tmp;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
inline void resetTaskEvents(T *obj, struct context* ctx) {
|
||||
while (!taskEventQueueEmpty(obj)) {
|
||||
struct taskEventBase* base = taskEventQueueDequeue(obj);
|
||||
if (base->type == ncclProfileColl) {
|
||||
struct collective* c = (struct collective *)base;
|
||||
// reset event proxyOps & proxySteps
|
||||
memset(c->nProxyOps, 0, sizeof(int)*MAX_CHANNELS);
|
||||
// release collective events in the group and return them to the collective pool
|
||||
__atomic_fetch_add(&ctx->collPoolBase, 1, __ATOMIC_RELAXED);
|
||||
} else if (base->type == ncclProfileP2p) {
|
||||
struct p2p* p = (struct p2p *)base;
|
||||
// reset event proxyOp and proxySteps
|
||||
memset(&p->op, 0, sizeof(struct proxyOp)*MAX_CHANNELS);
|
||||
// release p2p events in the group and return them to the p2p pool
|
||||
__atomic_fetch_add(&ctx->p2pPoolBase, 1, __ATOMIC_RELAXED);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -11,17 +11,20 @@
|
||||
#include <stdlib.h>
|
||||
|
||||
#include "common.h"
|
||||
#include "err.h"
|
||||
|
||||
enum {
|
||||
ncclProfileGroup = (1 << 0), // group event type
|
||||
ncclProfileColl = (1 << 1), // host collective call event type
|
||||
ncclProfileP2p = (1 << 2), // host point-to-point call event type
|
||||
ncclProfileProxyOp = (1 << 3), // proxy operation event type
|
||||
ncclProfileProxyStep = (1 << 4), // proxy step event type
|
||||
ncclProfileProxyCtrl = (1 << 5), // proxy control event type
|
||||
ncclProfileKernelCh = (1 << 6), // kernel channel event type
|
||||
ncclProfileNetPlugin = (1 << 7), // network plugin-defined, events
|
||||
ncclProfileGroup = (1 << 0), // group event type
|
||||
ncclProfileColl = (1 << 1), // host collective call event type
|
||||
ncclProfileP2p = (1 << 2), // host point-to-point call event type
|
||||
ncclProfileProxyOp = (1 << 3), // proxy operation event type
|
||||
ncclProfileProxyStep = (1 << 4), // proxy step event type
|
||||
ncclProfileProxyCtrl = (1 << 5), // proxy control event type
|
||||
ncclProfileKernelCh = (1 << 6), // kernel channel event type
|
||||
ncclProfileNetPlugin = (1 << 7), // network plugin-defined, events
|
||||
ncclProfileGroupApi = (1 << 8), // Group API events
|
||||
ncclProfileCollApi = (1 << 9), // Collective API events
|
||||
ncclProfileP2pApi = (1 << 10), // Point-to-Point API events
|
||||
ncclProfileKernelLaunch = (1 << 11), // Kernel launch events
|
||||
};
|
||||
|
||||
typedef enum {
|
||||
@@ -56,21 +59,27 @@ typedef enum {
|
||||
|
||||
/* Kernel event states */
|
||||
ncclProfilerKernelChStop = 22,
|
||||
|
||||
/* Group API States */
|
||||
ncclProfilerEndGroupApiStart = 23,
|
||||
ncclProfilerBeginGroupApiEnd = 24
|
||||
} ncclProfilerEventState_t;
|
||||
|
||||
typedef ncclProfilerEventState_t ncclProfilerEventState_v1_t;
|
||||
typedef ncclProfilerEventState_t ncclProfilerEventState_v2_t;
|
||||
typedef ncclProfilerEventState_t ncclProfilerEventState_v3_t;
|
||||
typedef ncclProfilerEventState_t ncclProfilerEventState_v4_t;
|
||||
typedef ncclProfilerEventState_t ncclProfilerEventState_v5_t;
|
||||
|
||||
#include "profiler_v5.h"
|
||||
#include "profiler_v4.h"
|
||||
#include "profiler_v3.h"
|
||||
#include "profiler_v2.h"
|
||||
#include "profiler_v1.h"
|
||||
#include "profiler_net.h"
|
||||
|
||||
typedef ncclProfiler_v4_t ncclProfiler_t;
|
||||
typedef ncclProfilerEventDescr_v4_t ncclProfilerEventDescr_t;
|
||||
typedef ncclProfilerEventStateArgs_v4_t ncclProfilerEventStateArgs_t;
|
||||
typedef ncclProfiler_v5_t ncclProfiler_t;
|
||||
typedef ncclProfilerEventDescr_v5_t ncclProfilerEventDescr_t;
|
||||
typedef ncclProfilerEventStateArgs_v5_t ncclProfilerEventStateArgs_t;
|
||||
|
||||
#endif // end include guard
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2024, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
|
||||
#ifndef PROFILER_V5_H_
|
||||
#define PROFILER_V5_H_
|
||||
#include <stdbool.h>
|
||||
|
||||
typedef struct {
|
||||
uint64_t type; // event type descriptor: ncclProfileGroupApi, ...
|
||||
void* parentObj; // pointer to the profiler parent object
|
||||
int rank; // originating rank
|
||||
union {
|
||||
struct {
|
||||
int graphCaptured;
|
||||
int groupDepth;
|
||||
} groupApi;
|
||||
|
||||
struct {
|
||||
const char* func;
|
||||
size_t count;
|
||||
const char* datatype;
|
||||
int root;
|
||||
void* stream;
|
||||
bool graphCaptured;
|
||||
} collApi;
|
||||
|
||||
struct {
|
||||
const char* func;
|
||||
size_t count;
|
||||
const char* datatype;
|
||||
void* stream;
|
||||
bool graphCaptured;
|
||||
} p2pApi;
|
||||
|
||||
struct {
|
||||
void* stream;
|
||||
} kernelLaunch;
|
||||
|
||||
struct {
|
||||
uint64_t seqNumber;
|
||||
const char* func;
|
||||
void const* sendBuff;
|
||||
void* recvBuff;
|
||||
size_t count;
|
||||
int root;
|
||||
const char* datatype;
|
||||
uint8_t nChannels;
|
||||
uint8_t nWarps;
|
||||
const char* algo;
|
||||
const char* proto;
|
||||
void* parentGroup; // for backward compatibility with v4
|
||||
} coll;
|
||||
|
||||
struct {
|
||||
const char* func;
|
||||
void* buff;
|
||||
const char* datatype;
|
||||
size_t count;
|
||||
int peer;
|
||||
uint8_t nChannels;
|
||||
void* parentGroup; // for backward compatibility with v4
|
||||
} p2p;
|
||||
|
||||
struct {
|
||||
pid_t pid; // pid of the originating process
|
||||
uint8_t channelId; // channel id for this proxy operation
|
||||
int peer; // remote rank for send/recv
|
||||
int nSteps; // number of steps for this proxy operation
|
||||
int chunkSize; // amount of data transferred by this proxy operation
|
||||
int isSend;
|
||||
} proxyOp;
|
||||
|
||||
struct {
|
||||
int step;
|
||||
} proxyStep;
|
||||
|
||||
struct {
|
||||
uint8_t channelId;
|
||||
uint64_t pTimer; // start timestamp from GPU globaltimer
|
||||
} kernelCh;
|
||||
|
||||
struct {
|
||||
int64_t id;
|
||||
void* data;
|
||||
} netPlugin;
|
||||
};
|
||||
} ncclProfilerEventDescr_v5_t;
|
||||
|
||||
typedef union {
|
||||
struct {
|
||||
size_t transSize;
|
||||
} proxyStep;
|
||||
|
||||
struct {
|
||||
int appendedProxyOps;
|
||||
} proxyCtrl;
|
||||
|
||||
struct {
|
||||
void* data;
|
||||
} netPlugin;
|
||||
|
||||
struct {
|
||||
uint64_t pTimer;
|
||||
} kernelCh;
|
||||
} ncclProfilerEventStateArgs_v5_t;
|
||||
|
||||
typedef struct {
|
||||
const char* name;
|
||||
|
||||
// init - initialize the profiler plugin
|
||||
// Input
|
||||
// - context : opaque profiler context object for separating profiler behavior across comms
|
||||
// - commId : communicator id
|
||||
// - commName : user assigned communicator name
|
||||
// - nNodes : number of nodes in communicator
|
||||
// - nranks : number of ranks in communicator
|
||||
// - rank : rank identifier in communicator
|
||||
// - logfn : logger function
|
||||
// Output
|
||||
// - eActivationMask: bitmask of active events set by the plugin
|
||||
ncclResult_t (*init)(void** context, uint64_t commId, int* eActivationMask, const char* commName, int nNodes, int nranks, int rank, ncclDebugLogger_t logfn);
|
||||
|
||||
// startEvent - initialize and start a new event for the supplied event descriptor inside the eventset
|
||||
// Input
|
||||
// - context: opaque profiler context object
|
||||
// - eDescr : pointer to ncclProfilerEventDescr_t object
|
||||
// Output
|
||||
// - eHandle: return event handle for supplied event descriptor object
|
||||
ncclResult_t (*startEvent)(void* context, void** eHandle, ncclProfilerEventDescr_v5_t* eDescr);
|
||||
|
||||
// stopEvent - stop/finalize an event inside and event set
|
||||
// Input
|
||||
// - eHandle: handle to event object
|
||||
ncclResult_t (*stopEvent)(void* eHandle);
|
||||
|
||||
// recordEventState - record event state transitions and event attribute updates
|
||||
// Input
|
||||
// - eHandle : handle to event object created through startEvent
|
||||
// - eStateArgs: optional argument used to capture event attribute updates associated with the state transition
|
||||
// - eState : event state transition
|
||||
ncclResult_t (*recordEventState)(void* eHandle, ncclProfilerEventState_v5_t eState, ncclProfilerEventStateArgs_v5_t* eStateArgs);
|
||||
|
||||
// finalize - finalize the profiler plugin
|
||||
// Input
|
||||
// - context: opaque profiler context object
|
||||
ncclResult_t (*finalize)(void* context);
|
||||
} ncclProfiler_v5_t;
|
||||
|
||||
#endif
|
||||
+230
-34
@@ -6,7 +6,7 @@
|
||||
|
||||
#include <stdio.h>
|
||||
#include <pthread.h>
|
||||
#include <string.h>
|
||||
#include <cstring>
|
||||
#include <linux/limits.h>
|
||||
#include <sys/time.h>
|
||||
#include <sys/types.h>
|
||||
@@ -22,12 +22,20 @@ static int initialized; // initialization counter for profiler
|
||||
static double startTime; // profiler start time
|
||||
|
||||
static const int defaultEActivationMask = ncclProfileColl | ncclProfileP2p;
|
||||
static const int defaultGroupPoolSize = 16;
|
||||
static const int defaultCollPoolSize = 16;
|
||||
static const int defaultP2pPoolSize = 1024;
|
||||
static const int defaultGroupApiPoolSize = 256;
|
||||
static const int defaultCollApiPoolSize = 256;
|
||||
static const int defaultP2pApiPoolSize = 256;
|
||||
static const int defaultKernelLaunchPoolSize = 256;
|
||||
static const int defaultGroupPoolSize = 256;
|
||||
static const int defaultCollPoolSize = 256;
|
||||
static const int defaultP2pPoolSize = 256;
|
||||
static const int defaultProxyCtrlPoolSize = 16;
|
||||
static const int defaultDetachPoolSize = 128;
|
||||
static const int defaultDetachPoolSize = 256;
|
||||
|
||||
static int groupApiPoolSize;
|
||||
static int collApiPoolSize;
|
||||
static int p2pApiPoolSize;
|
||||
static int kernelLaunchPoolSize;
|
||||
static int groupPoolSize;
|
||||
static int collPoolSize;
|
||||
static int p2pPoolSize;
|
||||
@@ -51,7 +59,7 @@ static pthread_mutex_t lock = PTHREAD_MUTEX_INITIALIZER;
|
||||
static pid_t pid;
|
||||
static int* eActivationMaskPtr;
|
||||
|
||||
__hidden ncclResult_t exampleProfilerInit(void** context, int* eActivationMask, const char* commName, uint64_t commHash, int nNodes, int nranks, int rank, ncclDebugLogger_t logfn) {
|
||||
__hidden ncclResult_t exampleProfilerInit(void** context, uint64_t commId, int* eActivationMask, const char* commName, int nNodes, int nranks, int rank, ncclDebugLogger_t logfn) {
|
||||
pthread_mutex_lock(&lock);
|
||||
if (__atomic_fetch_add(&initialized, 1, __ATOMIC_RELAXED) == 0) {
|
||||
// first thread initializes event mask, environment and detach pool
|
||||
@@ -59,6 +67,18 @@ __hidden ncclResult_t exampleProfilerInit(void** context, int* eActivationMask,
|
||||
str = getenv("NCCL_PROFILE_EVENT_MASK");
|
||||
__atomic_store_n(eActivationMask, str ? atoi(str) : 0, __ATOMIC_RELAXED);
|
||||
|
||||
str = getenv("NCCL_PROFILE_GROUP_API_POOL_SIZE");
|
||||
groupApiPoolSize = str ? atoi(str) : defaultGroupApiPoolSize;
|
||||
|
||||
str = getenv("NCCL_PROFILE_COLL_API_POOL_SIZE");
|
||||
collApiPoolSize = str ? atoi(str) : defaultCollApiPoolSize;
|
||||
|
||||
str = getenv("NCCL_PROFILE_P2P_API_POOL_SIZE");
|
||||
p2pApiPoolSize = str ? atoi(str) : defaultP2pApiPoolSize;
|
||||
|
||||
str = getenv("NCCL_PROFILE_KERNEL_LAUNCH_POOL_SIZE");
|
||||
kernelLaunchPoolSize = str ? atoi(str) : defaultKernelLaunchPoolSize;
|
||||
|
||||
str = getenv("NCCL_PROFILE_GROUP_POOL_SIZE");
|
||||
groupPoolSize = str ? atoi(str) : defaultGroupPoolSize;
|
||||
|
||||
@@ -96,11 +116,23 @@ __hidden ncclResult_t exampleProfilerInit(void** context, int* eActivationMask,
|
||||
// pre-allocate memory for event object pools in dedicated profiler context
|
||||
struct context* ctx = (struct context *)calloc(1, sizeof(*ctx));
|
||||
ctx->commName = commName;
|
||||
ctx->commHash = commHash;
|
||||
ctx->commHash = commId;
|
||||
ctx->nranks = nranks;
|
||||
ctx->rank = rank;
|
||||
logFn = logfn;
|
||||
INFO(NCCL_INIT, "PROFILER/Plugin: init commName: %s commHash: %lu nranks: %d rank: %d", commName ? commName : "", commHash, nranks, rank);
|
||||
INFO(NCCL_INIT, "PROFILER/Plugin: init commName: %s commHash: %lu nranks: %d rank: %d", commName ? commName : "", commId, nranks, rank);
|
||||
|
||||
ctx->groupApiPool = (struct groupApi *)calloc(groupApiPoolSize, sizeof(*ctx->groupApiPool));
|
||||
if (ctx->groupApiPool == NULL) goto fail;
|
||||
|
||||
ctx->collApiPool = (struct collApi *)calloc(collApiPoolSize, sizeof(*ctx->collApiPool));
|
||||
if (ctx->collApiPool == NULL) goto fail;
|
||||
|
||||
ctx->p2pApiPool = (struct p2pApi *)calloc(p2pApiPoolSize, sizeof(*ctx->p2pApiPool));
|
||||
if (ctx->p2pApiPool == NULL) goto fail;
|
||||
|
||||
ctx->kernelLaunchPool = (struct kernelLaunch *)calloc(kernelLaunchPoolSize, sizeof(*ctx->kernelLaunchPool));
|
||||
if (ctx->kernelLaunchPool == NULL) goto fail;
|
||||
|
||||
ctx->groupPool = (struct group *)calloc(groupPoolSize, sizeof(*ctx->groupPool));
|
||||
if (ctx->groupPool == NULL) goto fail;
|
||||
@@ -130,16 +162,22 @@ fail:
|
||||
if (ctx->p2pPool) free(ctx->p2pPool);
|
||||
if (ctx->collPool) free(ctx->collPool);
|
||||
if (ctx->groupPool) free(ctx->groupPool);
|
||||
if (ctx->collApiPool) free(ctx->collApiPool);
|
||||
if (ctx->p2pApiPool) free(ctx->p2pApiPool);
|
||||
if (ctx->kernelLaunchPool) free(ctx->kernelLaunchPool);
|
||||
if (ctx->groupApiPool) free(ctx->groupApiPool);
|
||||
free(ctx);
|
||||
if (detachPool) free(detachPool);
|
||||
return ncclSystemError;
|
||||
}
|
||||
|
||||
static const char* profilerDumpFile;
|
||||
|
||||
__hidden ncclResult_t exampleProfilerFinalize(void* context) {
|
||||
FILE* fh = NULL;
|
||||
char filename[PATH_MAX] = { 0 };
|
||||
struct context* ctx = (struct context *)context;
|
||||
const char* dump = getenv("NCCL_PROFILE_DUMP_FILE");
|
||||
const char* dump = profilerDumpFile ? profilerDumpFile : getenv("NCCL_PROFILE_DUMP_FILE");
|
||||
if (dump) {
|
||||
sprintf(filename, "%s_%lu_%d.json", dump, ctx->commHash, ctx->rank);
|
||||
fh = fopen(filename, "w");
|
||||
@@ -148,10 +186,12 @@ __hidden ncclResult_t exampleProfilerFinalize(void* context) {
|
||||
INFO(NCCL_INIT, "PROFILER/Plugin: finalize commName: %s commHash: %lu nranks: %d rank: %d", ctx->commName ? ctx->commName : "", ctx->commHash, ctx->nranks, ctx->rank);
|
||||
|
||||
// print last N groups/collectives/p2ps
|
||||
int start = (ctx->groupPoolIndex - groupPoolSize >= 0) ? ctx->groupPoolIndex - groupPoolSize : 0;
|
||||
int end = ctx->groupPoolIndex;
|
||||
// Note that since the v5 version of the profiler, group API events are now at the top of the hierarchy.
|
||||
// Legacy Group events from v4 are still emitted for compatibility purposes when using the v4 profiler but excluded from this example.
|
||||
int start = (ctx->groupApiPoolIndex - groupApiPoolSize >= 0) ? ctx->groupApiPoolIndex - groupApiPoolSize : 0;
|
||||
int end = ctx->groupApiPoolIndex;
|
||||
for (int i = start; i < end; i++) {
|
||||
printEvent(fh, &ctx->groupPool[i%groupPoolSize]);
|
||||
printEvent(fh, &ctx->groupApiPool[i%groupApiPoolSize]);
|
||||
}
|
||||
|
||||
start = (ctx->proxyCtrlPoolIndex - proxyCtrlPoolSize >= 0) ? ctx->proxyCtrlPoolIndex - proxyCtrlPoolSize : 0;
|
||||
@@ -161,6 +201,10 @@ __hidden ncclResult_t exampleProfilerFinalize(void* context) {
|
||||
}
|
||||
|
||||
free(ctx->groupPool);
|
||||
free(ctx->collApiPool);
|
||||
free(ctx->p2pApiPool);
|
||||
free(ctx->kernelLaunchPool);
|
||||
free(ctx->groupApiPool);
|
||||
free(ctx->collPool);
|
||||
free(ctx->p2pPool);
|
||||
free(ctx->proxyCtrlPool);
|
||||
@@ -187,7 +231,113 @@ __hidden void updateEvent(void* handle);
|
||||
__hidden ncclResult_t exampleProfilerStartEvent(void* context, void** eHandle, ncclProfilerEventDescr_t* eDescr) {
|
||||
*eHandle = NULL;
|
||||
struct context* ctx = (struct context *)context;
|
||||
if (eDescr->type == ncclProfileGroup) {
|
||||
if (eDescr->type == ncclProfileGroupApi) {
|
||||
struct groupApi* event;
|
||||
int groupApiId = __atomic_fetch_add(&ctx->groupApiPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
if ((groupApiId - __atomic_load_n(&ctx->groupApiPoolBase, __ATOMIC_RELAXED)) < groupApiPoolSize) {
|
||||
// if there are available group API events grab one
|
||||
event = &ctx->groupApiPool[groupApiId%groupApiPoolSize];
|
||||
// Make sure all child events of the picked group API event are cleared
|
||||
while (!profilerQueueEmpty(&event->collApiEvents)) {
|
||||
struct collApi *collApiEvent = profilerQueueDequeue(&event->collApiEvents);
|
||||
resetTaskEvents(collApiEvent, ctx);
|
||||
__atomic_fetch_add(&ctx->collApiPoolBase, 1, __ATOMIC_RELAXED);
|
||||
}
|
||||
while (!profilerQueueEmpty(&event->p2pApiEvents)) {
|
||||
struct p2pApi *p2pApiEvent = profilerQueueDequeue(&event->p2pApiEvents);
|
||||
resetTaskEvents(p2pApiEvent, ctx);
|
||||
__atomic_fetch_add(&ctx->p2pApiPoolBase, 1, __ATOMIC_RELAXED);
|
||||
}
|
||||
while (!profilerQueueEmpty(&event->kernelLaunchEvents)) {
|
||||
profilerQueueDequeue(&event->kernelLaunchEvents);
|
||||
__atomic_fetch_add(&ctx->kernelLaunchPoolBase, 1, __ATOMIC_RELAXED);
|
||||
}
|
||||
} else {
|
||||
// else drop this event
|
||||
__atomic_fetch_sub(&ctx->groupApiPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
return ncclSuccess;
|
||||
}
|
||||
event->type = ncclProfileGroupApi;
|
||||
event->ctx = ctx;
|
||||
event->groupApiId = groupApiId;
|
||||
event->graphCaptured = eDescr->groupApi.graphCaptured;
|
||||
event->groupDepth = eDescr->groupApi.groupDepth;
|
||||
event->startTs = gettime() - startTime;
|
||||
*eHandle = event;
|
||||
} else if (eDescr->type == ncclProfileCollApi) {
|
||||
if (eDescr->parentObj == NULL) return ncclSuccess;
|
||||
struct collApi* event;
|
||||
int collApiId = __atomic_fetch_add(&ctx->collApiPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
if ((collApiId - __atomic_load_n(&ctx->collApiPoolBase, __ATOMIC_RELAXED)) < collApiPoolSize) {
|
||||
// if there are available Coll API events grab one
|
||||
event = &ctx->collApiPool[collApiId%collApiPoolSize];
|
||||
resetTaskEvents(event, ctx);
|
||||
} else {
|
||||
// else drop this event
|
||||
__atomic_fetch_sub(&ctx->collApiPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
return ncclSuccess;
|
||||
}
|
||||
event->type = ncclProfileCollApi;
|
||||
event->collApiId = collApiId;
|
||||
event->ctx = ctx;
|
||||
event->func = eDescr->collApi.func;
|
||||
event->stream = (cudaStream_t) eDescr->collApi.stream;
|
||||
event->count = eDescr->collApi.count;
|
||||
event->datatype = eDescr->collApi.datatype;
|
||||
event->root = eDescr->collApi.root;
|
||||
event->graphCaptured = eDescr->collApi.graphCaptured;
|
||||
struct groupApi* parent = (struct groupApi *) eDescr->parentObj;
|
||||
event->parent = parent;
|
||||
profilerQueueEnqueue(&parent->collApiEvents, event);
|
||||
__atomic_fetch_add(&parent->refCount, 1, __ATOMIC_RELAXED);
|
||||
*eHandle = event;
|
||||
} else if (eDescr->type == ncclProfileP2pApi) {
|
||||
if (eDescr->parentObj == NULL) return ncclSuccess;
|
||||
struct p2pApi* event;
|
||||
int p2pApiId = __atomic_fetch_add(&ctx->p2pApiPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
if ((p2pApiId - __atomic_load_n(&ctx->p2pApiPoolBase, __ATOMIC_RELAXED)) < p2pApiPoolSize) {
|
||||
// if there are available p2p API events grab one
|
||||
event = &ctx->p2pApiPool[p2pApiId%p2pApiPoolSize];
|
||||
resetTaskEvents(event, ctx);
|
||||
} else {
|
||||
// else drop this event
|
||||
__atomic_fetch_sub(&ctx->p2pApiPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
return ncclSuccess;
|
||||
}
|
||||
event->type = ncclProfileP2pApi;
|
||||
event->p2pApiId = p2pApiId;
|
||||
event->ctx = ctx;
|
||||
event->func = eDescr->p2pApi.func;
|
||||
event->stream = (cudaStream_t) eDescr->p2pApi.stream;
|
||||
event->count = eDescr->p2pApi.count;
|
||||
event->datatype = eDescr->p2pApi.datatype;
|
||||
event->graphCaptured = eDescr->p2pApi.graphCaptured;
|
||||
struct groupApi* parent = (struct groupApi *) eDescr->parentObj;
|
||||
event->parent = parent;
|
||||
profilerQueueEnqueue(&parent->p2pApiEvents, event);
|
||||
__atomic_fetch_add(&parent->refCount, 1, __ATOMIC_RELAXED);
|
||||
*eHandle = event;
|
||||
} else if (eDescr->type == ncclProfileKernelLaunch) {
|
||||
if (eDescr->parentObj == NULL) return ncclSuccess;
|
||||
struct kernelLaunch* event;
|
||||
int kernelLaunchId = __atomic_fetch_add(&ctx->kernelLaunchPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
if ((kernelLaunchId - __atomic_load_n(&ctx->kernelLaunchPoolBase, __ATOMIC_RELAXED)) < kernelLaunchPoolSize) {
|
||||
// if there are available kernel API events grab one
|
||||
event = &ctx->kernelLaunchPool[kernelLaunchId%kernelLaunchPoolSize];
|
||||
} else {
|
||||
// else drop this event
|
||||
__atomic_fetch_sub(&ctx->kernelLaunchPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
return ncclSuccess;
|
||||
}
|
||||
event->type = ncclProfileKernelLaunch;
|
||||
event->stream = (cudaStream_t) eDescr->kernelLaunch.stream;
|
||||
struct groupApi* parent = (struct groupApi *) eDescr->parentObj;
|
||||
event->parent = parent;
|
||||
profilerQueueEnqueue(&parent->kernelLaunchEvents, event);
|
||||
__atomic_fetch_add(&parent->refCount, 1, __ATOMIC_RELAXED);
|
||||
*eHandle = event;
|
||||
} else if (eDescr->type == ncclProfileGroup) {
|
||||
if (eDescr->parentObj == NULL) return ncclSuccess;
|
||||
struct group* event;
|
||||
int groupId = __atomic_fetch_add(&ctx->groupPoolIndex, 1, __ATOMIC_RELAXED);
|
||||
if ((groupId - __atomic_load_n(&ctx->groupPoolBase, __ATOMIC_RELAXED)) < groupPoolSize) {
|
||||
@@ -222,7 +372,7 @@ __hidden ncclResult_t exampleProfilerStartEvent(void* context, void** eHandle, n
|
||||
debugEvent(event, "GroupStart");
|
||||
} else if (eDescr->type == ncclProfileColl) {
|
||||
// the parent might be null if we run out of events
|
||||
struct group* parent = (struct group *)eDescr->parentObj;
|
||||
struct collApi* parent = (struct collApi *)eDescr->parentObj;
|
||||
if (parent == NULL) return ncclSuccess;
|
||||
|
||||
struct collective* event;
|
||||
@@ -253,12 +403,12 @@ __hidden ncclResult_t exampleProfilerStartEvent(void* context, void** eHandle, n
|
||||
event->proto = eDescr->coll.proto;
|
||||
*eHandle = event;
|
||||
taskEventQueueEnqueue(parent, (struct taskEventBase *)event);
|
||||
// increment the group ref counter so the event will staty open
|
||||
// increment the group ref counter so the event will stay open
|
||||
__atomic_fetch_add(&parent->refCount, 1, __ATOMIC_RELAXED);
|
||||
debugEvent(event, "CollStart");
|
||||
} else if (eDescr->type == ncclProfileP2p) {
|
||||
// the parent might be null if we run out of events
|
||||
struct group* parent = (struct group *)eDescr->parentObj;
|
||||
struct p2pApi* parent = (struct p2pApi*) eDescr->parentObj;
|
||||
if (parent == NULL) return ncclSuccess;
|
||||
|
||||
struct p2p* event;
|
||||
@@ -458,8 +608,34 @@ __hidden ncclResult_t exampleProfilerStartEvent(void* context, void** eHandle, n
|
||||
}
|
||||
|
||||
void updateEvent(void* handle) {
|
||||
uint8_t type = *(uint8_t *)handle;
|
||||
if (type == ncclProfileGroup) {
|
||||
uint64_t type = *(uint64_t *)handle;
|
||||
if (type == ncclProfileGroupApi) {
|
||||
struct groupApi* event = (struct groupApi*) handle;
|
||||
if (__atomic_sub_fetch(&event->refCount, 1, __ATOMIC_RELAXED) == 0) {
|
||||
event->stopTs = gettime() - startTime;
|
||||
__atomic_fetch_add(&event->ctx->groupApiPoolBase, 1, __ATOMIC_RELAXED);
|
||||
}
|
||||
} else if (type == ncclProfileCollApi) {
|
||||
struct collApi* event = (struct collApi*) handle;
|
||||
if (__atomic_sub_fetch(&event->refCount, 1, __ATOMIC_RELAXED) == 0) {
|
||||
event->stopTs = gettime() - startTime;
|
||||
__atomic_fetch_add(&event->ctx->collApiPoolBase, 1, __ATOMIC_RELAXED);
|
||||
}
|
||||
updateEvent(event->parent);
|
||||
return;
|
||||
} else if (type == ncclProfileP2pApi) {
|
||||
struct p2pApi* event = (struct p2pApi*) handle;
|
||||
if (__atomic_sub_fetch(&event->refCount, 1, __ATOMIC_RELAXED) == 0) {
|
||||
event->stopTs = gettime() - startTime;
|
||||
__atomic_fetch_add(&event->ctx->p2pApiPoolBase, 1, __ATOMIC_RELAXED);
|
||||
}
|
||||
updateEvent(event->parent);
|
||||
event->stopTs = gettime() - startTime;
|
||||
} else if (type == ncclProfileKernelLaunch) {
|
||||
struct kernelLaunch* event = (struct kernelLaunch*) handle;
|
||||
event->stopTs = gettime() - startTime;
|
||||
updateEvent(event->parent);
|
||||
} else if (type == ncclProfileGroup) {
|
||||
struct group* event = (struct group *)handle;
|
||||
if (__atomic_sub_fetch(&event->refCount, 1, __ATOMIC_RELAXED) == 0) {
|
||||
event->stopTs = gettime() - startTime;
|
||||
@@ -527,25 +703,35 @@ __hidden ncclResult_t exampleProfilerStopEvent(void* eHandle) {
|
||||
// the event handle might be null if we run out of events
|
||||
if (eHandle == NULL) return ncclSuccess;
|
||||
|
||||
uint8_t type = *(uint8_t *)eHandle;
|
||||
if (type == ncclProfileGroup) {
|
||||
// stopping the group event in NCCL core does not
|
||||
// mean the group has completed. It means the group
|
||||
// was submitted/enqueued so we need to keep the event open
|
||||
uint64_t type = *(uint64_t *)eHandle;
|
||||
// Stopping API events, Kernel Launch events, collective/p2p task events
|
||||
// in NCCL core do not mean that they are complete. It means that the
|
||||
// operation was enqueued so we need to keep the events open
|
||||
if (type == ncclProfileGroupApi) {
|
||||
struct groupApi* event = (struct groupApi*) eHandle;
|
||||
event->stopTs = gettime() - startTime;
|
||||
return ncclSuccess;
|
||||
} else if (type == ncclProfileCollApi) {
|
||||
struct collApi* event = (struct collApi*) eHandle;
|
||||
event->stopTs = gettime() - startTime;
|
||||
return ncclSuccess;
|
||||
} else if (type == ncclProfileP2pApi) {
|
||||
struct p2pApi* event = (struct p2pApi*) eHandle;
|
||||
event->stopTs = gettime() - startTime;
|
||||
return ncclSuccess;
|
||||
} else if (type == ncclProfileKernelLaunch) {
|
||||
struct kernelLaunch* event = (struct kernelLaunch*) eHandle;
|
||||
event->stopTs = gettime() - startTime;
|
||||
return ncclSuccess;
|
||||
} else if (type == ncclProfileGroup) {
|
||||
struct group* event = (struct group *)eHandle;
|
||||
event->stopTs = gettime() - startTime;
|
||||
return ncclSuccess;
|
||||
} else if (type == ncclProfileColl) {
|
||||
// stopping the collective event in NCCL core does not
|
||||
// mean the collective has completed. It means the collective
|
||||
// was submitted/enqueued so we need to keep the event open
|
||||
struct collective* event = (struct collective *)eHandle;
|
||||
event->base.stopTs = gettime() - startTime;
|
||||
return ncclSuccess;
|
||||
} else if (type == ncclProfileP2p) {
|
||||
// stopping the p2p event in NCCL core does not
|
||||
// mean the p2p has completed. It means the p2p
|
||||
// was submitted/enqueued so we need to keep the event open
|
||||
struct p2p* event = (struct p2p *)eHandle;
|
||||
event->base.stopTs = gettime() - startTime;
|
||||
return ncclSuccess;
|
||||
@@ -559,8 +745,15 @@ __hidden ncclResult_t exampleProfilerRecordEventState(void* eHandle, ncclProfile
|
||||
// the event handle might be null if we run out of events
|
||||
if (eHandle == NULL) return ncclSuccess;
|
||||
|
||||
uint8_t type = *(uint8_t *)eHandle;
|
||||
if (type == ncclProfileProxyOp) {
|
||||
uint64_t type = *(uint64_t *)eHandle;
|
||||
if (type == ncclProfileGroupApi) {
|
||||
struct groupApi* event = (struct groupApi*) eHandle;
|
||||
if (eState == ncclProfilerEndGroupApiStart) {
|
||||
event->endOfncclGroupStartTs = gettime() - startTime;
|
||||
} else if (eState == ncclProfilerBeginGroupApiEnd) {
|
||||
event->startOfncclGroupEndTs = gettime() - startTime;
|
||||
}
|
||||
} else if (type == ncclProfileProxyOp) {
|
||||
struct proxyOp* event = (struct proxyOp *)eHandle;
|
||||
if (eState == ncclProfilerProxyOpInProgress_v4) {
|
||||
event->progrTs = gettime() - startTime;
|
||||
@@ -592,6 +785,8 @@ __hidden ncclResult_t exampleProfilerRecordEventState(void* eHandle, ncclProfile
|
||||
case ncclProfilerProxyStepRecvGPUWait:
|
||||
event->timestamp[PROXY_STEP_RECV_GPU_WAIT] = gettime() - startTime;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
} else if (type == ncclProfileProxyCtrl) {
|
||||
struct proxyCtrl* event = (struct proxyCtrl *)eHandle;
|
||||
@@ -609,7 +804,7 @@ __hidden ncclResult_t exampleProfilerRecordEventState(void* eHandle, ncclProfile
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
ncclProfiler_t ncclProfiler_v4 = {
|
||||
ncclProfiler_t ncclProfiler_v5 = {
|
||||
"Example-profiler",
|
||||
exampleProfilerInit,
|
||||
exampleProfilerStartEvent,
|
||||
@@ -618,14 +813,15 @@ ncclProfiler_t ncclProfiler_v4 = {
|
||||
exampleProfilerFinalize,
|
||||
};
|
||||
|
||||
int exampleProfilerStart(int eActivationMask) {
|
||||
__attribute__((visibility("default"))) int exampleProfilerStart(int eActivationMask, const char* name) {
|
||||
profilerDumpFile = name;
|
||||
if (__atomic_load_n(&initialized, __ATOMIC_RELAXED)) {
|
||||
__atomic_store_n(eActivationMaskPtr, eActivationMask, __ATOMIC_RELAXED);
|
||||
}
|
||||
return ncclSuccess;
|
||||
}
|
||||
|
||||
int exampleProfilerStop(void) {
|
||||
__attribute__((visibility("default"))) int exampleProfilerStop(void) {
|
||||
if (__atomic_load_n(&initialized, __ATOMIC_RELAXED)) {
|
||||
__atomic_store_n(eActivationMaskPtr, 0, __ATOMIC_RELAXED);
|
||||
}
|
||||
@@ -7,7 +7,8 @@
|
||||
#ifndef PLUGIN_H_
|
||||
#define PLUGIN_H_
|
||||
|
||||
int exampleProfilerStart(int eActivationMask);
|
||||
int exampleProfilerStop(void);
|
||||
__attribute__((visibility("default"))) int exampleProfilerStart(int eActivationMask, const char* name);
|
||||
__attribute__((visibility("default"))) int exampleProfilerStop(void);
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
+93
-6
@@ -5,15 +5,59 @@
|
||||
************************************************************************/
|
||||
|
||||
#include <stdio.h>
|
||||
#include "err.h"
|
||||
#include "profiler.h"
|
||||
#include "event.h"
|
||||
#include "print_event.h"
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#define __hidden __attribute__ ((visibility("hidden")))
|
||||
|
||||
// FIXME: chrome tracing asynchronous events (following used) allow event nesting for events that have same id and category
|
||||
// It appears that nesting more than three events causes issues. Therefore, every event is given an increasing id and a
|
||||
// category that matches the type of event (GROUP, COLL, P2P, PROXY, NET)
|
||||
// category that matches the type of event (GROUP API, COLL API, P2P API, GROUP, COLL, P2P, PROXY, NET)
|
||||
static __thread int groupApiId;
|
||||
__hidden void printGroupApiEventHeader(FILE* fh, struct groupApi* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"GROUP_API\", \"ph\": \"b\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f, \"args\": {\"groupApiId\": %d, \"groupDepth\":%d}},\n",
|
||||
"Group API", groupApiId, getpid(), 1, event->startTs, event->groupApiId, event->groupDepth);
|
||||
}
|
||||
|
||||
__hidden void printGroupApiEventTrailer(FILE* fh, struct groupApi* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"GROUP_API\", \"ph\": \"e\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f},\n",
|
||||
"Group API", groupApiId++, getpid(), 1, event->stopTs);
|
||||
}
|
||||
|
||||
static __thread int p2pApiId;
|
||||
__hidden void printP2pApiEventHeader(FILE* fh, struct p2pApi* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"P2P_API\", \"ph\": \"b\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f, \"args\": {\"count\": %lu, \"datatype\": %s, \"GraphCaptured\":%d, \"Stream\": %p}},\n",
|
||||
event->func, p2pApiId, getpid(), 1, event->startTs, event->count, event->datatype, event->graphCaptured, event->stream);
|
||||
}
|
||||
|
||||
__hidden void printP2pApiEventTrailer(FILE* fh, struct p2pApi* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"P2P_API\", \"ph\": \"e\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f},\n",
|
||||
event->func, p2pApiId++, getpid(), 1, event->stopTs);
|
||||
}
|
||||
|
||||
static __thread int collApiId;
|
||||
__hidden void printCollApiEventHeader(FILE* fh, struct collApi* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"COLL_API\", \"ph\": \"b\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f, \"args\": {\"count\": %lu, \"datatype\": %s, \"root\": %d, \"GraphCaptured\":%d, \"Stream\": %p}},\n",
|
||||
event->func, collApiId, getpid(), 1, event->startTs, event->count, event->datatype, event->root, event->graphCaptured, event->stream);
|
||||
}
|
||||
|
||||
__hidden void printCollApiEventTrailer(FILE* fh, struct collApi* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"COLL_API\", \"ph\": \"e\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f},\n",
|
||||
event->func, collApiId++, getpid(), 1, event->stopTs);
|
||||
}
|
||||
|
||||
static __thread int kernelLaunchId;
|
||||
__hidden void printKernelLaunchEventHeader(FILE* fh, struct kernelLaunch* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"KERNEL_LAUNCH\", \"ph\": \"b\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f, \"args\": {\"groupId\": %d, \"Stream\": %p}},\n", "KernelLaunch", kernelLaunchId, getpid(), 1, event->startTs, event->kernelLaunchId, event->stream);
|
||||
}
|
||||
|
||||
__hidden void printKernelLaunchEventTrailer(FILE* fh, struct kernelLaunch* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"KERNEL_LAUNCH\", \"ph\": \"e\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f},\n", "KernelLaunch", kernelLaunchId++, getpid(), 1, event->stopTs);
|
||||
}
|
||||
|
||||
static __thread int groupId;
|
||||
__hidden void printGroupEventHeader(FILE* fh, struct group* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"GROUP\", \"ph\": \"b\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f, \"args\": {\"groupId\": %d}},\n",
|
||||
@@ -28,7 +72,7 @@ __hidden void printGroupEventTrailer(FILE* fh, struct group* event) {
|
||||
static __thread int collId;
|
||||
__hidden void printCollEventHeader(FILE* fh, struct collective* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"COLL\", \"ph\": \"b\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f, \"args\": {\"SeqNum\": %lu, \"CommHash\": %lu, \"Rank\": %d, \"Count\": %lu, \"Datatype\": \"%s\", \"Algorithm\": \"%s\", \"Protocol\": \"%s\", \"nChannels\": %d}},\n",
|
||||
event->base.func, collId, getpid(), 1, event->base.startTs, event->seqNumber, event->base.parent->ctx->commHash, event->base.rank, event->count, event->datatype, event->algo, event->proto, event->nChannels);
|
||||
event->base.func, collId, getpid(), 1, event->base.startTs, event->seqNumber, ((struct collApi*)event->base.parent)->ctx->commHash, event->base.rank, event->count, event->datatype, event->algo, event->proto, event->nChannels);
|
||||
}
|
||||
|
||||
__hidden void printCollEventTrailer(FILE* fh, struct collective* event) {
|
||||
@@ -39,7 +83,7 @@ __hidden void printCollEventTrailer(FILE* fh, struct collective* event) {
|
||||
static __thread int p2pId;
|
||||
__hidden void printP2pEventHeader(FILE* fh, struct p2p* event) {
|
||||
fprintf(fh, "{\"name\": \"%s\", \"cat\": \"P2P\", \"ph\": \"b\", \"id\": %d, \"pid\": %d, \"tid\": %d, \"ts\": %f, \"args\": {\"CommHash\": %lu, \"Rank\": %d, \"Peer\": %d, \"Count\": %lu, \"Datatype\": \"%s\", \"nChannels\": %d}},\n",
|
||||
event->base.func, p2pId, getpid(), 1, event->base.startTs, event->base.parent->ctx->commHash, event->base.rank, event->peer, event->count, event->datatype, event->nChannels);
|
||||
event->base.func, p2pId, getpid(), 1, event->base.startTs, ((struct p2pApi*)event->base.parent)->ctx->commHash, event->base.rank, event->peer, event->count, event->datatype, event->nChannels);
|
||||
}
|
||||
|
||||
__hidden void printP2pEventTrailer(FILE* fh, struct p2p* event) {
|
||||
@@ -173,7 +217,7 @@ void debugEvent(void* eHandle, const char* tag) {
|
||||
char filename[64] = { 0 };
|
||||
sprintf(filename, "EventDebug-%d", getpid());
|
||||
FILE* fh = fopen(filename, "a+");
|
||||
uint8_t type = *(uint8_t *)eHandle;
|
||||
uint64_t type = *(uint64_t *)eHandle;
|
||||
if (type == ncclProfileGroup) {
|
||||
struct group* event = (struct group *)eHandle;
|
||||
fprintf(fh, "Group event %p tag = %s {\n", event, tag);
|
||||
@@ -241,8 +285,51 @@ void debugEvent(void* eHandle, const char* tag) {
|
||||
|
||||
void printEvent(FILE* fh, void* handle) {
|
||||
if (handle == NULL || fh == NULL) return;
|
||||
uint8_t type = *(uint8_t *)handle;
|
||||
if (type == ncclProfileGroup) {
|
||||
uint64_t type = *(uint64_t *)handle;
|
||||
if (type == ncclProfileGroupApi) {
|
||||
struct groupApi* g = (struct groupApi*) handle;
|
||||
printGroupApiEventHeader(fh, g);
|
||||
struct kernelLaunch* kernelLaunchHead = profilerQueueHead(&g->kernelLaunchEvents);
|
||||
while (kernelLaunchHead != NULL) {
|
||||
printEvent(fh, kernelLaunchHead);
|
||||
kernelLaunchHead = kernelLaunchHead->next;
|
||||
}
|
||||
struct collApi* collApiHead = profilerQueueHead(&g->collApiEvents);
|
||||
while (collApiHead != NULL) {
|
||||
printEvent(fh, collApiHead);
|
||||
collApiHead = collApiHead->next;
|
||||
}
|
||||
struct p2pApi* p2pApiHead = profilerQueueHead(&g->p2pApiEvents);
|
||||
while (p2pApiHead != NULL) {
|
||||
printEvent(fh, p2pApiHead);
|
||||
p2pApiHead = p2pApiHead->next;
|
||||
}
|
||||
printGroupApiEventTrailer(fh, g);
|
||||
} else if (type == ncclProfileCollApi) {
|
||||
struct collApi* collApiEvent = (struct collApi *) handle;
|
||||
printCollApiEventHeader(fh, collApiEvent);
|
||||
struct taskEventBase* base = taskEventQueueHead(collApiEvent);
|
||||
while (base) {
|
||||
struct taskEventBase* next = base->next;
|
||||
printEvent(fh, base);
|
||||
base = next;
|
||||
}
|
||||
printCollApiEventTrailer(fh, collApiEvent);
|
||||
} else if (type == ncclProfileP2pApi) {
|
||||
struct p2pApi* p2pApiEvent = (struct p2pApi *) handle;
|
||||
printP2pApiEventHeader(fh, p2pApiEvent);
|
||||
struct taskEventBase* base = taskEventQueueHead(p2pApiEvent);
|
||||
while (base) {
|
||||
struct taskEventBase* next = base->next;
|
||||
printEvent(fh, base);
|
||||
base = next;
|
||||
}
|
||||
printP2pApiEventTrailer(fh, p2pApiEvent);
|
||||
} else if (type == ncclProfileKernelLaunch) {
|
||||
struct kernelLaunch* kernelLaunchEvent = (struct kernelLaunch *) handle;
|
||||
printKernelLaunchEventHeader(fh, kernelLaunchEvent);
|
||||
printKernelLaunchEventTrailer(fh, kernelLaunchEvent);
|
||||
} else if (type == ncclProfileGroup) {
|
||||
struct group* g = (struct group *)handle;
|
||||
printGroupEventHeader(fh, g);
|
||||
struct taskEventBase* base = taskEventQueueHead(g);
|
||||
@@ -0,0 +1,50 @@
|
||||
/*************************************************************************
|
||||
* Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* See LICENSE.txt for license information
|
||||
************************************************************************/
|
||||
#ifndef QUEUE_H
|
||||
#define QUEUE_H
|
||||
|
||||
template<typename T, T *T::*next>
|
||||
struct profilerQueue {
|
||||
T *head, *tail;
|
||||
};
|
||||
|
||||
template<typename T, T *T::*next>
|
||||
inline void profilerQueueConstruct(profilerQueue<T,next> *me) {
|
||||
me->head = nullptr;
|
||||
me->tail = nullptr;
|
||||
}
|
||||
|
||||
template<typename T, T *T::*next>
|
||||
inline bool profilerQueueEmpty(profilerQueue<T,next> *me) {
|
||||
return me->head == nullptr;
|
||||
}
|
||||
|
||||
template<typename T, T *T::*next>
|
||||
inline T* profilerQueueHead(profilerQueue<T,next> *me) {
|
||||
return me->head;
|
||||
}
|
||||
|
||||
template<typename T, T *T::*next>
|
||||
inline T* profilerQueueTail(profilerQueue<T,next> *me) {
|
||||
return me->tail;
|
||||
}
|
||||
|
||||
template<typename T, T *T::*next>
|
||||
inline void profilerQueueEnqueue(profilerQueue<T,next> *me, T *x) {
|
||||
x->*next = nullptr;
|
||||
(me->head ? me->tail->*next : me->head) = x;
|
||||
me->tail = x;
|
||||
}
|
||||
|
||||
template<typename T, T *T::*next>
|
||||
inline T* profilerQueueDequeue(profilerQueue<T,next> *me) {
|
||||
T *ans = me->head;
|
||||
me->head = ans->*next;
|
||||
if (me->head == nullptr) me->tail = nullptr;
|
||||
return ans;
|
||||
}
|
||||
|
||||
#endif
|
||||
新增問題並參考
封鎖使用者