Merge pull request #31 from BKP/ipc_bringup_fine_unit_09-26-24
Add IPC Simple Buffer Fine-grained Unit Tests
Этот коммит содержится в:
@@ -90,6 +90,7 @@ target_sources(
|
||||
free_list_gtest.cpp
|
||||
#context_ipc_gtest.cpp
|
||||
ipc_impl_simple_coarse_gtest.cpp
|
||||
ipc_impl_simple_fine_gtest.cpp
|
||||
)
|
||||
|
||||
###############################################################################
|
||||
|
||||
@@ -33,6 +33,7 @@ TEST_F(IPCImplSimpleCoarseTestFixture, MPI_num_pes) {
|
||||
}
|
||||
|
||||
TEST_F(IPCImplSimpleCoarseTestFixture, IPC_bases) {
|
||||
ASSERT_NE(ipc_impl_.ipc_bases, nullptr);
|
||||
for(int i{0}; i < mpi_.num_pes(); i++) {
|
||||
ASSERT_NE(ipc_impl_.ipc_bases[i], nullptr);
|
||||
}
|
||||
|
||||
@@ -218,8 +218,6 @@ class IPCImplSimpleCoarseTestFixture : public ::testing::Test {
|
||||
protected:
|
||||
std::vector<int> golden_;
|
||||
|
||||
std::vector<int> output_;
|
||||
|
||||
HEAP_T heap_mem_ {};
|
||||
|
||||
MPI_T mpi_ {heap_mem_.get_ptr(), heap_mem_.get_size()};
|
||||
|
||||
Разница между файлами не показана из-за своего большого размера
Загрузить разницу
@@ -0,0 +1,344 @@
|
||||
/******************************************************************************
|
||||
* Copyright (c) 2024 Advanced Micro Devices, Inc. All rights reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
* of this software and associated documentation files (the "Software"), to
|
||||
* deal in the Software without restriction, including without limitation the
|
||||
* rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
|
||||
* sell copies of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
* AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||
* FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
|
||||
* IN THE SOFTWARE.
|
||||
*****************************************************************************/
|
||||
|
||||
#ifndef ROCSHMEM_IPC_IMPL_SIMPLE_FINE_GTEST_HPP
|
||||
#define ROCSHMEM_IPC_IMPL_SIMPLE_FINE_GTEST_HPP
|
||||
|
||||
#include "gtest/gtest.h"
|
||||
|
||||
#include <numeric>
|
||||
#include <mpi.h>
|
||||
|
||||
#include "../src/atomic.hpp"
|
||||
#include "../src/ipc_policy.hpp"
|
||||
#include "../src/memory/notifier.hpp"
|
||||
#include "../src/memory/symmetric_heap.hpp"
|
||||
#include "../src/util.hpp"
|
||||
|
||||
namespace rocshmem {
|
||||
|
||||
enum TestType {
|
||||
READ = 0,
|
||||
WRITE = 1
|
||||
};
|
||||
|
||||
const uint32_t SIGNAL_OFFSET {67108864};
|
||||
|
||||
__device__
|
||||
void
|
||||
validator(bool *error, int *golden, int *dest, size_t bytes) {
|
||||
size_t elements {bytes / sizeof(int)};
|
||||
for (int i {get_flat_id()}; i < elements; i += get_flat_grid_size()) {
|
||||
if (golden[i] != dest[i]) {
|
||||
printf("golden[%d] %d != dest[%d] %d\n", i, golden[i], i, dest[i]);
|
||||
*error = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template <typename NotifierT>
|
||||
__global__
|
||||
void
|
||||
kernel_put_with_signal_validator(bool *error, int *golden, int *dest, size_t bytes, NotifierT *notifier) {
|
||||
detail::atomic::rocshmem_memory_orders orders{};
|
||||
if (!get_flat_id()) {
|
||||
while (detail::atomic::load<int, detail::atomic::memory_scope_system>(dest + SIGNAL_OFFSET, orders) == 0) {
|
||||
;
|
||||
}
|
||||
}
|
||||
notifier->sync();
|
||||
validator(error, golden, dest, bytes);
|
||||
}
|
||||
|
||||
template <typename NotifierT>
|
||||
__global__
|
||||
void
|
||||
kernel_simple_fine_copy(IpcImpl *ipc_impl, bool *error, int *golden, int *src, int *dest, size_t bytes, TestType test, NotifierT *notifier) {
|
||||
if (!get_flat_id()) {
|
||||
ipc_impl->ipcCopy(dest, src, bytes);
|
||||
ipc_impl->ipcFence();
|
||||
if (test == WRITE) {
|
||||
ipc_impl->ipcAMOFetchAdd(dest + SIGNAL_OFFSET, 1);
|
||||
}
|
||||
}
|
||||
if (test == READ) {
|
||||
notifier->sync();
|
||||
validator(error, golden, dest, bytes);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename NotifierT>
|
||||
__global__
|
||||
void
|
||||
kernel_simple_fine_copy_wg(IpcImpl *ipc_impl, bool *error, int *golden, int *src, int *dest, size_t bytes, TestType test, NotifierT *notifier) {
|
||||
if (!blockIdx.x) {
|
||||
ipc_impl->ipcCopy_wg(dest, src, bytes);
|
||||
ipc_impl->ipcFence();
|
||||
if (test == WRITE) {
|
||||
if (!threadIdx.x) {
|
||||
ipc_impl->ipcAMOFetchAdd(dest + SIGNAL_OFFSET, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (test == READ) {
|
||||
notifier->sync();
|
||||
validator(error, golden, dest, bytes);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename NotifierT>
|
||||
__global__
|
||||
void
|
||||
kernel_simple_fine_copy_wave(IpcImpl *ipc_impl, bool *error, int *golden, int *src, int *dest, size_t bytes, TestType test, NotifierT *notifier) {
|
||||
if (!blockIdx.x && threadIdx.x < 64) {
|
||||
ipc_impl->ipcCopy_wave(dest, src, bytes);
|
||||
ipc_impl->ipcFence();
|
||||
if (test == WRITE) {
|
||||
if (!threadIdx.x) {
|
||||
ipc_impl->ipcAMOFetchAdd(dest + SIGNAL_OFFSET, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
if (test == READ) {
|
||||
notifier->sync();
|
||||
validator(error, golden, dest, bytes);
|
||||
}
|
||||
}
|
||||
|
||||
class IPCImplSimpleFineTestFixture : public ::testing::Test {
|
||||
using HEAP_T = HeapMemory<HIPDefaultFinegrainedAllocator>;
|
||||
using MPI_T = RemoteHeapInfo<CommunicatorMPI>;
|
||||
using NotifierT = Notifier<detail::atomic::memory_scope_agent>;
|
||||
using NotifierProxyT = NotifierProxy<HIPAllocator, detail::atomic::memory_scope_agent>;
|
||||
using FN_T1 = void (*)(IpcImpl*, bool*, int*, int*, int*, size_t, TestType, NotifierT*);
|
||||
using FN_T2 = void (*)(bool*, int*, int*, size_t, NotifierT*);
|
||||
|
||||
public:
|
||||
IPCImplSimpleFineTestFixture() {
|
||||
ipc_impl_.ipcHostInit(mpi_.my_pe(), mpi_.get_heap_bases() , MPI_COMM_WORLD);
|
||||
|
||||
assert(ipc_impl_dptr_ == nullptr);
|
||||
hip_allocator_.allocate((void**)&ipc_impl_dptr_, sizeof(IpcImpl));
|
||||
CHECK_HIP(hipMemcpy(ipc_impl_dptr_, &ipc_impl_, sizeof(IpcImpl), hipMemcpyHostToDevice));
|
||||
|
||||
assert(error_dptr_ == nullptr);
|
||||
hip_allocator_.allocate((void**)&error_dptr_, sizeof(bool));
|
||||
*error_dptr_ = false;
|
||||
}
|
||||
|
||||
~IPCImplSimpleFineTestFixture() {
|
||||
if (ipc_impl_dptr_) {
|
||||
hip_allocator_.deallocate(ipc_impl_dptr_);
|
||||
}
|
||||
if (error_dptr_) {
|
||||
hip_allocator_.deallocate(error_dptr_);
|
||||
}
|
||||
if (golden_dptr_) {
|
||||
hip_allocator_.deallocate(golden_dptr_);
|
||||
}
|
||||
|
||||
ipc_impl_.ipcHostStop();
|
||||
}
|
||||
|
||||
void launch(FN_T1 f, const dim3 grid, const dim3 block, int* src, int* dest, size_t bytes, TestType test) {
|
||||
f<<<grid, block>>>(ipc_impl_dptr_, error_dptr_, golden_dptr_, src, dest, bytes, test, notifier_.get());
|
||||
CHECK_HIP(hipStreamSynchronize(nullptr));
|
||||
}
|
||||
|
||||
void launch(FN_T2 f, const dim3 grid, const dim3 block, int* dest, size_t bytes) {
|
||||
f<<<grid, block>>>(error_dptr_, golden_dptr_, dest, bytes, notifier_.get());
|
||||
CHECK_HIP(hipStreamSynchronize(nullptr));
|
||||
}
|
||||
|
||||
void write(const dim3 grid, const dim3 block, size_t elems) {
|
||||
iota_golden(elems);
|
||||
initialize_signal(WRITE);
|
||||
initialize_src_buffer(WRITE);
|
||||
copy(WRITE, grid, block);
|
||||
check_device_validation_errors(WRITE);
|
||||
}
|
||||
|
||||
void write_wg(const dim3 grid, const dim3 block, size_t elems) {
|
||||
iota_golden(elems);
|
||||
initialize_signal(WRITE);
|
||||
initialize_src_buffer(WRITE);
|
||||
copy_wg(WRITE, grid, block);
|
||||
check_device_validation_errors(WRITE);
|
||||
}
|
||||
|
||||
void write_wave(const dim3 grid, const dim3 block, size_t elems) {
|
||||
iota_golden(elems);
|
||||
initialize_signal(WRITE);
|
||||
initialize_src_buffer(WRITE);
|
||||
copy_wave(WRITE, grid, block);
|
||||
check_device_validation_errors(WRITE);
|
||||
}
|
||||
|
||||
void read(const dim3 grid, const dim3 block, size_t elems) {
|
||||
iota_golden(elems);
|
||||
initialize_signal(READ);
|
||||
initialize_src_buffer(READ);
|
||||
copy(READ, grid, block);
|
||||
check_device_validation_errors(READ);
|
||||
}
|
||||
|
||||
void read_wg(const dim3 grid, const dim3 block, size_t elems) {
|
||||
iota_golden(elems);
|
||||
initialize_signal(READ);
|
||||
initialize_src_buffer(READ);
|
||||
copy_wg(READ, grid, block);
|
||||
check_device_validation_errors(READ);
|
||||
}
|
||||
|
||||
void read_wave(const dim3 grid, const dim3 block, size_t elems) {
|
||||
iota_golden(elems);
|
||||
initialize_signal(READ);
|
||||
initialize_src_buffer(READ);
|
||||
copy_wave(READ, grid, block);
|
||||
check_device_validation_errors(READ);
|
||||
}
|
||||
|
||||
void iota_golden(size_t elems) {
|
||||
golden_.resize(elems);
|
||||
std::iota(golden_.begin(), golden_.end(), 0);
|
||||
|
||||
assert(golden_dptr_ == nullptr);
|
||||
size_t golden_dptr_bytes {golden_.size() * sizeof(int)};
|
||||
hip_allocator_.allocate((void**)&golden_dptr_, golden_dptr_bytes);
|
||||
CHECK_HIP(hipMemcpy(golden_dptr_, golden_.data(), golden_dptr_bytes, hipMemcpyHostToDevice));
|
||||
}
|
||||
|
||||
void validate_golden(size_t elems) {
|
||||
ASSERT_EQ(golden_.size(), elems);
|
||||
for (int i{0}; i < golden_.size(); i++) {
|
||||
ASSERT_EQ(golden_[i], i);
|
||||
}
|
||||
}
|
||||
|
||||
void initialize_signal(TestType test) {
|
||||
bool is_write_test = test;
|
||||
if (is_write_test && mpi_.my_pe() == 0) {
|
||||
int *dest = reinterpret_cast<int*>(ipc_impl_.ipc_bases[1]);
|
||||
*(dest + SIGNAL_OFFSET) = 0;
|
||||
}
|
||||
}
|
||||
|
||||
void initialize_src_buffer(TestType test) {
|
||||
if (!pe_initializes_src_buffer(test)) {
|
||||
return;
|
||||
}
|
||||
size_t bytes = golden_.size() * sizeof(int);
|
||||
auto dev_src = reinterpret_cast<int*>(ipc_impl_.ipc_bases[mpi_.my_pe()]);
|
||||
CHECK_HIP(hipMemcpy(dev_src, golden_.data(), bytes, hipMemcpyHostToDevice));
|
||||
}
|
||||
|
||||
bool pe_initializes_src_buffer(TestType test) {
|
||||
bool is_write_test = test;
|
||||
bool is_read_test = !test;
|
||||
return (is_write_test && mpi_.my_pe() == 0) ||
|
||||
(is_read_test && mpi_.my_pe() == 1);
|
||||
}
|
||||
|
||||
void execute(TestType test, FN_T1 fn, const dim3 grid, const dim3 block) {
|
||||
size_t bytes = golden_.size() * sizeof(int);
|
||||
if (mpi_.my_pe()) {
|
||||
mpi_.barrier();
|
||||
if (test == WRITE) {
|
||||
int *dest = reinterpret_cast<int*>(ipc_impl_.ipc_bases[1]);
|
||||
FN_T2 val_fn = kernel_put_with_signal_validator;
|
||||
launch(val_fn, grid, block, dest, bytes);
|
||||
}
|
||||
mpi_.barrier();
|
||||
return;
|
||||
}
|
||||
int *src{nullptr};
|
||||
int *dest{nullptr};
|
||||
if (test == WRITE) {
|
||||
src = reinterpret_cast<int*>(ipc_impl_.ipc_bases[0]);
|
||||
dest = reinterpret_cast<int*>(ipc_impl_.ipc_bases[1]);
|
||||
} else {
|
||||
src = reinterpret_cast<int*>(ipc_impl_.ipc_bases[1]);
|
||||
dest = reinterpret_cast<int*>(ipc_impl_.ipc_bases[0]);
|
||||
}
|
||||
mpi_.barrier();
|
||||
launch(fn, grid, block, src, dest, bytes, test);
|
||||
mpi_.barrier();
|
||||
}
|
||||
|
||||
void copy(TestType test, dim3 grid, dim3 block) {
|
||||
execute(test, kernel_simple_fine_copy, grid, block);
|
||||
}
|
||||
|
||||
void copy_wg(TestType test, dim3 grid, dim3 block) {
|
||||
execute(test, kernel_simple_fine_copy_wg, grid, block);
|
||||
}
|
||||
|
||||
void copy_wave(TestType test, dim3 grid, dim3 block) {
|
||||
execute(test, kernel_simple_fine_copy_wave, grid, block);
|
||||
}
|
||||
|
||||
void check_device_validation_errors(TestType test) {
|
||||
if (!pe_validates_dest_buffer(test)) {
|
||||
return;
|
||||
}
|
||||
ASSERT_EQ(*error_dptr_, false);
|
||||
}
|
||||
|
||||
void validate_dest_buffer(TestType test) {
|
||||
if (!pe_validates_dest_buffer(test)) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto dev_dest = reinterpret_cast<int*>(ipc_impl_.ipc_bases[mpi_.my_pe()]);
|
||||
for (int i{0}; i < golden_.size(); i++) {
|
||||
ASSERT_EQ(golden_[i], dev_dest[i]);
|
||||
}
|
||||
}
|
||||
|
||||
bool pe_validates_dest_buffer(TestType test) {
|
||||
return !pe_initializes_src_buffer(test);
|
||||
}
|
||||
|
||||
protected:
|
||||
HIPDefaultFinegrainedAllocator hip_allocator_ {};
|
||||
|
||||
NotifierProxyT notifier_ {};
|
||||
|
||||
HEAP_T heap_mem_ {};
|
||||
|
||||
MPI_T mpi_ {heap_mem_.get_ptr(), heap_mem_.get_size()};
|
||||
|
||||
std::vector<int> golden_;
|
||||
|
||||
int *golden_dptr_ {nullptr};
|
||||
|
||||
IpcImpl ipc_impl_ {};
|
||||
|
||||
IpcImpl *ipc_impl_dptr_ {nullptr};
|
||||
|
||||
bool *error_dptr_ {nullptr};
|
||||
};
|
||||
|
||||
} // namespace rocshmem
|
||||
|
||||
#endif // ROCSHMEM_IPC_IMPL_SIMPLE_FINE_GTEST_HPP
|
||||
@@ -28,30 +28,98 @@ using namespace rocshmem;
|
||||
******************************* Fixture Tests *******************************
|
||||
*****************************************************************************/
|
||||
|
||||
TEST_F(NotifierTestFixture, run_all_threads_once_1_1) {
|
||||
TEST_F(NotifierBlockTestFixture, run_all_threads_once_1_1) {
|
||||
run_all_threads_once(1, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierTestFixture, run_all_threads_once_2_1) {
|
||||
TEST_F(NotifierBlockTestFixture, run_all_threads_once_2_1) {
|
||||
run_all_threads_once(2, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierTestFixture, run_all_threads_once_64_1) {
|
||||
TEST_F(NotifierBlockTestFixture, run_all_threads_once_64_1) {
|
||||
run_all_threads_once(64, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierTestFixture, run_all_threads_once_128_1) {
|
||||
TEST_F(NotifierBlockTestFixture, run_all_threads_once_128_1) {
|
||||
run_all_threads_once(128, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierTestFixture, run_all_threads_once_256_1) {
|
||||
TEST_F(NotifierBlockTestFixture, run_all_threads_once_256_1) {
|
||||
run_all_threads_once(256, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierTestFixture, run_all_threads_once_512_1) {
|
||||
TEST_F(NotifierBlockTestFixture, run_all_threads_once_512_1) {
|
||||
run_all_threads_once(512, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierTestFixture, run_all_threads_once_1024_1) {
|
||||
TEST_F(NotifierBlockTestFixture, run_all_threads_once_1024_1) {
|
||||
run_all_threads_once(1024, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1_1) {
|
||||
run_all_threads_once(1, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_2_1) {
|
||||
run_all_threads_once(2, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_64_1) {
|
||||
run_all_threads_once(64, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_128_1) {
|
||||
run_all_threads_once(128, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_256_1) {
|
||||
run_all_threads_once(256, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_512_1) {
|
||||
run_all_threads_once(512, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1024_1) {
|
||||
run_all_threads_once(1024, 1);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1_2) {
|
||||
run_all_threads_once(1, 2);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1024_2) {
|
||||
run_all_threads_once(1024, 2);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1_4) {
|
||||
run_all_threads_once(1, 4);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1024_4) {
|
||||
run_all_threads_once(1024, 4);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1_8) {
|
||||
run_all_threads_once(1, 8);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1024_8) {
|
||||
run_all_threads_once(1024, 8);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1_32) {
|
||||
run_all_threads_once(1, 32);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1024_32) {
|
||||
run_all_threads_once(1024, 32);
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1_38) {
|
||||
run_all_threads_once(1, 38); // MI300 CPX
|
||||
}
|
||||
|
||||
TEST_F(NotifierAgentTestFixture, run_all_threads_once_1024_38) {
|
||||
run_all_threads_once(1024, 38); // MI300 CPX
|
||||
}
|
||||
|
||||
@@ -44,66 +44,45 @@ static const uint64_t NOTIFIER_OFFSET {0x100B00};
|
||||
inline __device__
|
||||
void
|
||||
write_to_memory(uint8_t* raw_memory) {
|
||||
auto thread_idx {get_flat_block_id()};
|
||||
auto thread_idx {get_flat_id()};
|
||||
raw_memory[thread_idx] = THREAD_VALUE;
|
||||
__threadfence();
|
||||
}
|
||||
|
||||
template <typename NotifierT>
|
||||
__global__
|
||||
void
|
||||
all_threads_once(uint8_t* raw_memory,
|
||||
Notifier* notifier) {
|
||||
notifier->write(NOTIFIER_OFFSET);
|
||||
uint64_t offset_u64 {notifier->read()};
|
||||
notifier->done();
|
||||
|
||||
NotifierT * notifier) {
|
||||
if (!get_flat_id()) {
|
||||
notifier->store(NOTIFIER_OFFSET);
|
||||
notifier->fence();
|
||||
}
|
||||
notifier->sync();
|
||||
uint64_t offset_u64 {notifier->load()};
|
||||
uint64_t raw_memory_u64 {reinterpret_cast<uint64_t>(raw_memory)};
|
||||
uint64_t address_u64 {raw_memory_u64 + offset_u64};
|
||||
uint8_t* address {reinterpret_cast<uint8_t*>(address_u64)};
|
||||
write_to_memory(address);
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
class NotifierTestFixture : public ::testing::Test {
|
||||
using NotifierProxyT = NotifierProxy<HIPAllocator>;
|
||||
|
||||
class NotifierBase : public ::testing::Test {
|
||||
public:
|
||||
NotifierTestFixture() {
|
||||
NotifierBase() {
|
||||
assert(raw_memory_ == nullptr);
|
||||
hip_allocator_.allocate((void**)&raw_memory_, GIBIBYTE_);
|
||||
assert(raw_memory_);
|
||||
}
|
||||
|
||||
~NotifierTestFixture() {
|
||||
~NotifierBase() {
|
||||
if (raw_memory_) {
|
||||
hip_allocator_.deallocate(raw_memory_);
|
||||
}
|
||||
}
|
||||
|
||||
void
|
||||
run_all_threads_once(uint32_t x_block_dim,
|
||||
uint32_t x_grid_dim) {
|
||||
const dim3 hip_blocksize(x_block_dim, 1, 1);
|
||||
const dim3 hip_gridsize(x_grid_dim, 1, 1);
|
||||
|
||||
hipLaunchKernelGGL(all_threads_once,
|
||||
hip_gridsize,
|
||||
hip_blocksize,
|
||||
0,
|
||||
nullptr,
|
||||
raw_memory_,
|
||||
notifier_.get());
|
||||
|
||||
hipError_t return_code = hipStreamSynchronize(nullptr);
|
||||
if (return_code != hipSuccess) {
|
||||
printf("Failed in stream synchronize\n");
|
||||
assert(return_code == hipSuccess);
|
||||
}
|
||||
|
||||
size_t number_threads {x_block_dim * x_grid_dim};
|
||||
|
||||
verify(size_t number_threads) {
|
||||
uint8_t* offset_addr {compute_offset_addr()};
|
||||
|
||||
for (size_t i {0}; i < number_threads; i++) {
|
||||
ASSERT_EQ(offset_addr[i], THREAD_VALUE);
|
||||
}
|
||||
@@ -136,12 +115,51 @@ class NotifierTestFixture : public ::testing::Test {
|
||||
*/
|
||||
uint8_t *raw_memory_ {nullptr};
|
||||
|
||||
};
|
||||
|
||||
class NotifierBlockTestFixture : public NotifierBase {
|
||||
using NotifierT = Notifier<detail::atomic::memory_scope_workgroup>;
|
||||
using NotifierProxyT = NotifierProxy<HIPAllocator, detail::atomic::memory_scope_workgroup>;
|
||||
|
||||
public:
|
||||
void
|
||||
run_all_threads_once(uint32_t x_block_dim,
|
||||
uint32_t x_grid_dim) {
|
||||
new (notifier_.get()) NotifierT();
|
||||
const dim3 block(x_block_dim, 1, 1);
|
||||
const dim3 grid(x_grid_dim, 1, 1);
|
||||
all_threads_once<NotifierT><<<grid, block>>>(raw_memory_, notifier_.get());
|
||||
CHECK_HIP(hipStreamSynchronize(nullptr));
|
||||
verify(x_block_dim * x_grid_dim);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Used to broadcast base offset for writing.
|
||||
*/
|
||||
NotifierProxyT notifier_ {};
|
||||
};
|
||||
|
||||
class NotifierAgentTestFixture : public NotifierBase {
|
||||
using NotifierT = Notifier<detail::atomic::memory_scope_agent>;
|
||||
using NotifierProxyT = NotifierProxy<HIPAllocator, detail::atomic::memory_scope_agent>;
|
||||
|
||||
public:
|
||||
void
|
||||
run_all_threads_once(uint32_t x_block_dim,
|
||||
uint32_t x_grid_dim) {
|
||||
new (notifier_.get()) NotifierT();
|
||||
const dim3 block(x_block_dim, 1, 1);
|
||||
const dim3 grid(x_grid_dim, 1, 1);
|
||||
all_threads_once<NotifierT><<<grid, block>>>(raw_memory_, notifier_.get());
|
||||
CHECK_HIP(hipStreamSynchronize(nullptr));
|
||||
verify(x_block_dim * x_grid_dim);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Used to broadcast base offset for writing.
|
||||
*/
|
||||
NotifierProxyT notifier_ {};
|
||||
};
|
||||
|
||||
} // namespace rocshmem
|
||||
|
||||
|
||||
Ссылка в новой задаче
Block a user