Remove MPI compile-time dependency (#264)

* use dlsym for MPI functions

to allow compiling without MPI support, convert the usage of MPI functions and symbols to be based on a dlopen/dlsym based mechanism. Turns out this cannot be done entirely vendor neutral, slightly different solutions might be required for Open MPI, MPICH and the new MPI ABI.

* checkpoint

more work to be done.

* checkpoint 2

* checkpoint 3

* checkpoint 4

examples compile and link correctly

* checkpoitn 5 (I think)

* Checkpoitn 6

* dyld-mpi: adapt GDA

* dyldmpi: tests that depend on MPI need to link with it themselves

* do not ../mpi_instance.h

* dyldmpi: make the symetricHeapTestFixture compile

* dyldmpi: Change cmakery, compiles and run gda w/o external MPI

* Make it also compile in external MPI mode

* dyldmpi: ipc unit tests compile but do not link

* dyldmpi: new approach, if external mpi required, link with mpi,
otherwise use ompi5 abi

* C-style comments in cmakelist..

* dyldmpi: examples: do not fail compiling if MPI not found at build time,
instead do not compile the MPI required examples

* more updates to CMake logic

* convert RO backend

and a few other cleanups

* update some unit tests

to work with the dlopen MPI environment correctly.

---------

Co-authored-by: Aurelien Bouteiller <abouteil@amd.com>

[ROCm/rocshmem commit: e4c427a736]
This commit is contained in:
Edgar Gabriel
2025-10-01 08:06:56 -05:00
committed by GitHub
szülő 4f955324ac
commit 53fa35b980
50 fájl változott, egészen pontosan 712 új sor hozzáadva és 294 régi sor törölve
+13 -14
Fájl megtekintése
@@ -24,8 +24,6 @@
#include "ipc_policy.hpp"
#include <mpi.h>
#include "rocshmem/rocshmem_config.h" // NOLINT(build/include_subdir)
#include "backend_bc.hpp"
#include "context_incl.hpp"
@@ -39,20 +37,20 @@ __host__ void IpcOnImpl::ipcHostInit(int my_pe, const HEAP_BASES_T &heap_bases,
* Create an MPI communicator that deals only with local processes.
*/
MPI_Comm shmcomm;
MPI_Comm_split_type(thread_comm, MPI_COMM_TYPE_SHARED, 0, MPI_INFO_NULL,
&shmcomm);
mpilib_ftable_.Comm_split_type(thread_comm, MPI_COMM_TYPE_SHARED, 0, MPI_INFO_NULL,
&shmcomm);
/*
* Figure out how many local process there are.
*/
int Shm_size;
MPI_Comm_size(shmcomm, &Shm_size);
mpilib_ftable_.Comm_size(shmcomm, &Shm_size);
shm_size = Shm_size;
/*
* Figure out how this process' rank among local processes.
*/
MPI_Comm_rank(shmcomm, &shm_rank);
mpilib_ftable_.Comm_rank(shmcomm, &shm_rank);
/*
* Allocate a host-side c-array to hold the IPC handles.
@@ -73,8 +71,8 @@ __host__ void IpcOnImpl::ipcHostInit(int my_pe, const HEAP_BASES_T &heap_bases,
* Do an all-to-all exchange with each local processing element to
* share the symmetric heap IPC handles.
*/
MPI_Allgather(MPI_IN_PLACE, sizeof(hipIpcMemHandle_t), MPI_CHAR,
vec_ipc_handle, sizeof(hipIpcMemHandle_t), MPI_CHAR, shmcomm);
mpilib_ftable_.Allgather(MPI_IN_PLACE, sizeof(hipIpcMemHandle_t), MPI_CHAR,
vec_ipc_handle, sizeof(hipIpcMemHandle_t), MPI_CHAR, shmcomm);
/*
* Allocate device-side array to hold the IPC symmetric heap base
@@ -114,16 +112,17 @@ __host__ void IpcOnImpl::ipcHostInit(int my_pe, const HEAP_BASES_T &heap_bases,
CHECK_HIP(hipMalloc(reinterpret_cast<void**>(&pes_with_ipc_avail), shm_size * sizeof(int)));
MPI_Group thread_grp, shm_grp;
MPI_Comm_group(thread_comm, &thread_grp);
MPI_Comm_group(shmcomm, &shm_grp);
MPI_Group thread_grp;
MPI_Group shm_grp;
mpilib_ftable_.Comm_group(thread_comm, &thread_grp);
mpilib_ftable_.Comm_group(shmcomm, &shm_grp);
int *seqranks = new int[shm_size];
for(int i = 0; i < shm_size; i++)
seqranks[i] = i;
MPI_Group_translate_ranks(shm_grp, shm_size, seqranks, thread_grp, pes_with_ipc_avail);
mpilib_ftable_.Group_translate_ranks(shm_grp, shm_size, seqranks, thread_grp, pes_with_ipc_avail);
delete [] seqranks;
MPI_Group_free(&shm_grp);
MPI_Group_free(&thread_grp);
mpilib_ftable_.Group_free(&shm_grp);
mpilib_ftable_.Group_free(&thread_grp);
}
}