Support for tracing mutex locking (#52)
* Parallel overhead example with locks
* Support tracing mutex locking + more
- support wrapping pthread_mutex_lock
- support wrapping pthread_mutex_unlock
- support wrapping pthread_mutex_trylock
- get_perfetto_combined_traces setting
- OMNITRACE_TRACE_THREAD_LOCKS option
- ThreadState
- critical trace includes queue id
- enabled/disabled settings in timemory
- fix OMNITRACE_TIMEMORY_COMPONENTS
- fix reading config
- fix setting categories
- applied ThreadState::Internal in various places
- utility::get_filled_array
- utility::get_reserved_vector
- utility::get_thread_index
- fork_gotcha messages about forks
- split out some pthread_gotcha functionality into pthread_create_gotcha
- handle queue id in roctracer callbacks
* Update timemory and PTL submodules
* Misc CMake updates
- Includes fix to omnitrace-static-lib{gcc,stdcxx}
* Misc cleanup to pthread_mutex_gotcha and backtrace
* Fix to duplicate field in module_function json
* Improvement to debug messages
* omnitrace-dl and common improvements
- tweak to delimit
- common::ignore message
- common::join quoting of strings
- omnitrace_set_env ignores if inited and active
- omnitrace_set_mpi ignores if inited and active
* nsync for transpose example
* Fix to thread_deleter<void> functor invoke
* Fix thread state and HIP stream enums
Este cometimento está contido em:
cometido por
GitHub
ascendente
bab90baf0b
cometimento
b208047741
@@ -7,7 +7,13 @@ find_package(Threads REQUIRED)
|
||||
add_executable(parallel-overhead parallel-overhead.cpp)
|
||||
target_link_libraries(parallel-overhead Threads::Threads)
|
||||
|
||||
add_executable(parallel-overhead-locks parallel-overhead.cpp)
|
||||
target_link_libraries(parallel-overhead-locks Threads::Threads)
|
||||
target_compile_definitions(parallel-overhead-locks PRIVATE USE_LOCKS=1)
|
||||
|
||||
if(NOT CMAKE_PROJECT_NAME STREQUAL PROJECT_NAME)
|
||||
set_target_properties(parallel-overhead PROPERTIES RUNTIME_OUTPUT_DIRECTORY
|
||||
${CMAKE_BINARY_DIR})
|
||||
set_target_properties(parallel-overhead-locks PROPERTIES RUNTIME_OUTPUT_DIRECTORY
|
||||
${CMAKE_BINARY_DIR})
|
||||
endif()
|
||||
|
||||
@@ -1,14 +1,23 @@
|
||||
|
||||
#include <atomic>
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <string>
|
||||
#include <thread>
|
||||
#include <vector>
|
||||
|
||||
#if defined(USE_LOCKS)
|
||||
# include <mutex>
|
||||
using auto_lock_t = std::unique_lock<std::mutex>;
|
||||
long total = 0;
|
||||
std::mutex mtx{};
|
||||
#else
|
||||
# include <atomic>
|
||||
std::atomic<long> total{ 0 };
|
||||
#endif
|
||||
|
||||
long
|
||||
fib(long n) __attribute__((noinline));
|
||||
|
||||
void
|
||||
run(size_t nitr, long) __attribute__((noinline));
|
||||
|
||||
@@ -21,10 +30,19 @@ fib(long n)
|
||||
void
|
||||
run(size_t nitr, long n)
|
||||
{
|
||||
#if defined(USE_LOCKS)
|
||||
for(size_t i = 0; i < nitr; ++i)
|
||||
{
|
||||
auto _v = fib(n);
|
||||
auto_lock_t _lk{ mtx };
|
||||
total += _v;
|
||||
}
|
||||
#else
|
||||
long local = 0;
|
||||
for(size_t i = 0; i < nitr; ++i)
|
||||
local += fib(n);
|
||||
total += local;
|
||||
#endif
|
||||
}
|
||||
|
||||
int
|
||||
@@ -42,7 +60,7 @@ main(int argc, char** argv)
|
||||
if(argc > 2) nthread = atol(argv[2]);
|
||||
if(argc > 3) nitr = atol(argv[3]);
|
||||
|
||||
printf("[%s] Threads: %zu\n[%s] Iterations: %zu\n[%s] fibonacci(%li)...\n",
|
||||
printf("\n[%s] Threads: %zu\n[%s] Iterations: %zu\n[%s] fibonacci(%li)...\n",
|
||||
_name.c_str(), nthread, _name.c_str(), nitr, _name.c_str(), nfib);
|
||||
|
||||
std::vector<std::thread> threads{};
|
||||
@@ -53,13 +71,16 @@ main(int argc, char** argv)
|
||||
threads.emplace_back(&run, _nitr, nfib);
|
||||
}
|
||||
|
||||
#if !defined(USE_LOCKS)
|
||||
auto _nitr = std::max<size_t>(nitr - 0.25 * nitr, 1);
|
||||
run(_nitr, nfib - 0.1 * nfib);
|
||||
#endif
|
||||
|
||||
for(auto& itr : threads)
|
||||
itr.join();
|
||||
|
||||
printf("[%s] fibonacci(%li) x %lu = %li\n", _name.c_str(), nfib, nthread,
|
||||
total.load());
|
||||
static_cast<long>(total));
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -95,10 +95,12 @@ transpose_a(int* in, int* out, int M, int N)
|
||||
void
|
||||
run(int rank, int tid, hipStream_t stream, int argc, char** argv)
|
||||
{
|
||||
size_t nitr = 500;
|
||||
unsigned int M = 4960 * 2;
|
||||
unsigned int N = 4960 * 2;
|
||||
size_t nitr = 500;
|
||||
size_t nsync = 10;
|
||||
unsigned int M = 4960 * 2;
|
||||
unsigned int N = 4960 * 2;
|
||||
if(argc > 2) nitr = atoll(argv[2]);
|
||||
if(argc > 3) nsync = atoll(argv[3]);
|
||||
|
||||
auto_lock_t _lk{ print_lock };
|
||||
std::cout << "[" << rank << "][" << tid << "] M: " << M << " N: " << N << std::endl;
|
||||
@@ -126,10 +128,11 @@ run(int rank, int tid, hipStream_t stream, int argc, char** argv)
|
||||
dim3 block(32, 32, 1); // transpose_a
|
||||
|
||||
auto t1 = std::chrono::high_resolution_clock::now();
|
||||
for(size_t i = 0; i < nitr; i++)
|
||||
for(size_t i = 0; i < nitr; ++i)
|
||||
{
|
||||
transpose_a<<<grid, block, 0, stream>>>(in, out, M, N);
|
||||
check_hip_error();
|
||||
if(i % nsync == (nsync - 1)) HIP_API_CALL(hipStreamSynchronize(stream));
|
||||
}
|
||||
auto t2 = std::chrono::high_resolution_clock::now();
|
||||
HIP_API_CALL(hipStreamSynchronize(stream));
|
||||
@@ -179,15 +182,18 @@ do_a2a(int rank)
|
||||
int
|
||||
main(int argc, char** argv)
|
||||
{
|
||||
int rank = 0;
|
||||
int size = 1;
|
||||
int nthreads = 2;
|
||||
int nitr = 5000;
|
||||
int rank = 0;
|
||||
int size = 1;
|
||||
int nthreads = 2;
|
||||
int nitr = 5000;
|
||||
size_t nsync = 10;
|
||||
if(argc > 1) nthreads = atoi(argv[1]);
|
||||
if(argc > 2) nitr = atoi(argv[2]);
|
||||
if(argc > 3) nsync = atoll(argv[3]);
|
||||
|
||||
printf("[transpose] Number of threads: %i\n", nthreads);
|
||||
printf("[transpose] Number of iterations: %i\n", nitr);
|
||||
printf("[transpose] Syncing every %zu iterations\n", nsync);
|
||||
|
||||
#if defined(USE_MPI)
|
||||
MPI_Init(&argc, &argv);
|
||||
|
||||
Criar uma nova questão referindo esta
Bloquear um utilizador