Add windows build support into ROCr (#912)

Make sure ROCR can be compiled under windows. Extra setup for the windows build environment is required. The change should not have any functional changes under Linux.
Этот коммит содержится в:
German Andryeyev
2025-09-19 10:10:17 -04:00
коммит произвёл GitHub
родитель 96a0d16eda
Коммит 913743d433
40 изменённых файлов: 5476 добавлений и 280 удалений
+18 -1
Просмотреть файл
@@ -117,6 +117,11 @@ set_target_properties(hsakmt PROPERTIES
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/archive"
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/lib"
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/runtime")
if (WIN32)
set_target_properties(hsakmt PROPERTIES
CXX_STANDARD 20
CXX_STANDARD_REQUIRED ON)
endif()
if (BUILD_THUNK_VIRTIO)
add_rocm_subdir(libhsakmt/src/virtio "${THUNK_VIRTIO_DEFINITIONS}")
@@ -128,6 +133,11 @@ if (BUILD_ROCR)
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/archive"
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/lib"
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/runtime")
if (WIN32)
set_target_properties(hsa-runtime64 PROPERTIES
CXX_STANDARD 20
CXX_STANDARD_REQUIRED ON)
endif()
if (BUILD_SHARED_LIBS)
add_dependencies(hsa-runtime64 hsakmt)
@@ -187,6 +197,7 @@ if (DEFINED ENV{ROCM_LIBPATCH_VERSION})
message("Using CPACK_PACKAGE_VERSION ${CPACK_PACKAGE_VERSION}")
endif()
if (UNIX)
# Debian package specific variables
set(CPACK_DEBIAN_BINARY_PACKAGE_NAME "hsa-rocr")
set(CPACK_DEBIAN_DEV_PACKAGE_NAME "hsa-rocr-dev")
@@ -223,6 +234,10 @@ set(CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS "libdrm-amdgpu-dev | libdrm-dev, rocm-core
set(CPACK_DEBIAN_ASAN_PACKAGE_RECOMMENDS "libdrm-amdgpu-dev")
set(CPACK_DEBIAN_BINARY_PACKAGE_RECOMMENDS "libdrm-amdgpu-amdgpu1")
else()
set(CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS "hsakmt-roct")
set(CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS "hsakmt-roct")
endif()
if (ROCM_DEP_ROCMCORE)
string(APPEND CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS ", rocm-core")
string(APPEND CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS ", rocm-core-asan")
@@ -244,6 +259,7 @@ set(CPACK_RPM_DEV_PACKAGE_OBSOLETES "hsakmt-roct,hsakmt-roct-devel,hsakmt-roct-d
set(CPACK_RPM_DEV_PACKAGE_NAME "hsa-rocr-devel")
set(CPACK_RPM_ASAN_PACKAGE_NAME "hsa-rocr-asan")
if (UNIX)
if (DEFINED ENV{CPACK_RPM_PACKAGE_RELEASE})
set(CPACK_RPM_PACKAGE_RELEASE $ENV{CPACK_RPM_PACKAGE_RELEASE})
else()
@@ -254,6 +270,7 @@ string(APPEND CPACK_RPM_PACKAGE_RELEASE "%{?dist}")
set(CPACK_RPM_FILE_NAME "RPM-DEFAULT")
message("CPACK_RPM_PACKAGE_RELEASE: ${CPACK_RPM_PACKAGE_RELEASE}")
set(CPACK_RPM_PACKAGE_LICENSE "NCSA")
endif()
## Process the Rpm install/remove scripts to update the CPACK variables
configure_file("${CMAKE_CURRENT_SOURCE_DIR}/RPM/Binary/post.in" RPM/Binary/post @ONLY)
@@ -289,7 +306,7 @@ endif()
if (ROCM_DEP_ROCMCORE)
string(APPEND CPACK_RPM_BINARY_PACKAGE_REQUIRES " rocm-core")
string(APPEND CPACK_RPM_ASAN_PACKAGE_REQUIRES " rocm-core-asan")
else()
elseif (UNIX)
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_PACKAGE_REQUIRES ${CPACK_RPM_PACKAGE_REQUIRES})
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_DEBIAN_PACKAGE_DEPENDS ${CPACK_DEBIAN_PACKAGE_DEPENDS})
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_DEV_PACKAGE_REQUIRES ${CPACK_RPM_DEV_PACKAGE_REQUIRES})
+12 -2
Просмотреть файл
@@ -1,3 +1,11 @@
################################################################################
##
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
##
## SPDX-License-Identifier: MIT
##
################################################################################
cmake_minimum_required ( VERSION 3.5.0 )
# Set ext runtime module name and project name.
@@ -23,8 +31,10 @@ include ( utils )
## Compiler preproc definitions.
#add_definitions ( -D__linux__ )
add_definitions ( -DUNIX_OS )
add_definitions ( -DLINUX )
if(UNIX)
add_definitions ( -DUNIX_OS )
add_definitions ( -DLINUX )
endif()
add_definitions ( -D__AMD64__ )
add_definitions ( -D__x86_64__ )
add_definitions ( -DAMD_INTERNAL_BUILD )
+118 -65
Просмотреть файл
@@ -48,7 +48,11 @@ cmake_minimum_required ( VERSION 3.7 )
unset ( hsa-runtime64_LIB_DEPENDS CACHE )
set(CMAKE_VERBOSE_MAKEFILE ON)
set(CMAKE_CXX_STANDARD 17)
if (UNIX)
set(CMAKE_CXX_STANDARD 17)
else()
set(CMAKE_CXX_STANDARD 20)
endif()
## Set core runtime module name and project name.
set ( CORE_RUNTIME_NAME "hsa-runtime64" )
@@ -89,35 +93,46 @@ if(NOT LibElf_FOUND)
find_package(LibElf REQUIRED)
endif()
pkg_check_modules(drm REQUIRED IMPORTED_TARGET libdrm)
## Create the rocr target.
add_library( ${CORE_RUNTIME_TARGET} "" )
if (UNIX)
pkg_check_modules(drm REQUIRED IMPORTED_TARGET libdrm)
else()
target_include_directories(${CORE_RUNTIME_TARGET} PRIVATE ${LIBELF_INCLUDE_DIR})
if (${BUILD_SHARED_LIBS})
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE oclelf)
endif()
endif()
## Enforce uniform output file naming.
set_property(TARGET ${CORE_RUNTIME_TARGET} PROPERTY OUTPUT_NAME ${CORE_RUNTIME_NAME} )
## Compiler preproc definitions.
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" __linux__ HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}" )
## Check for memfd_create syscall
include(CheckSymbolExists)
CHECK_SYMBOL_EXISTS ( "__NR_memfd_create" "sys/syscall.h" HAVE_MEMFD_CREATE )
if ( HAVE_MEMFD_CREATE )
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_MEMFD_CREATE )
endif()
## Check for _GNU_SOURCE pthread extensions
set(CMAKE_REQUIRED_DEFINITIONS -D_GNU_SOURCE)
CHECK_SYMBOL_EXISTS ( "pthread_attr_setaffinity_np" "pthread.h" HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
CHECK_SYMBOL_EXISTS ( "pthread_rwlockattr_setkind_np" "pthread.h" HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
unset(CMAKE_REQUIRED_DEFINITIONS)
if ( HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
endif()
if ( HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
if (UNIX)
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" __linux__ HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}" )
## Check for memfd_create syscall
include(CheckSymbolExists)
CHECK_SYMBOL_EXISTS ( "__NR_memfd_create" "sys/syscall.h" HAVE_MEMFD_CREATE )
if ( HAVE_MEMFD_CREATE )
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_MEMFD_CREATE )
endif()
## Check for _GNU_SOURCE pthread extensions
set(CMAKE_REQUIRED_DEFINITIONS -D_GNU_SOURCE)
CHECK_SYMBOL_EXISTS ( "pthread_attr_setaffinity_np" "pthread.h" HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
CHECK_SYMBOL_EXISTS ( "pthread_rwlockattr_setkind_np" "pthread.h" HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
unset(CMAKE_REQUIRED_DEFINITIONS)
if ( HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
endif()
if ( HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
endif()
else()
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" AMD_LIBELF=1 HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}")
endif()
## Set include directories for ROCr runtime
@@ -133,16 +148,18 @@ target_include_directories( ${CORE_RUNTIME_TARGET}
## ------------------------- Linux Compiler and Linker options -------------------------
set ( HSA_CXX_FLAGS ${HSA_COMMON_CXX_FLAGS} -fexceptions -fno-rtti -fvisibility=hidden -Wno-error=missing-braces -Wno-error=sign-compare -Wno-sign-compare -Wno-write-strings -Wno-conversion-null -fno-math-errno -fno-threadsafe-statics -fmerge-all-constants -fms-extensions -Wno-error=comment -Wno-comment -Wno-error=pointer-arith -Wno-pointer-arith -Wno-error=unused-variable -Wno-error=unused-function )
if (UNIX)
set ( HSA_CXX_FLAGS ${HSA_COMMON_CXX_FLAGS} -fexceptions -fno-rtti -fvisibility=hidden -Wno-error=missing-braces -Wno-error=sign-compare -Wno-sign-compare -Wno-write-strings -Wno-conversion-null -fno-math-errno -fno-threadsafe-statics -fmerge-all-constants -fms-extensions -Wno-error=comment -Wno-comment -Wno-error=pointer-arith -Wno-pointer-arith -Wno-error=unused-variable -Wno-error=unused-function )
## Extra x86 specific settings
if ( CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64" )
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -mmwaitx )
## Extra x86 specific settings
if ( CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64" )
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -mmwaitx )
endif()
## Extra image settings - audit!
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-deprecated-declarations )
endif()
## Extra image settings - audit!
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-deprecated-declarations )
if ( CMAKE_COMPILER_IS_GNUCXX )
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-error=maybe-uninitialized -Wno-error=unused-but-set-variable)
endif ()
@@ -153,9 +170,14 @@ if ( CMAKE_CXX_COMPILER_ID MATCHES "Clang")
endif()
endif()
set ( DRVDEF "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.so.def" )
set ( LNKSCR "hsacore.so.link" )
set ( HSA_SHARED_LINK_FLAGS "-Wl,-Bdynamic -Wl,-z,noexecstack -Wl,${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR} -Wl,--version-script=${DRVDEF} -Wl,--enable-new-dtags" )
if (UNIX)
set ( LNKSCR "hsacore.so.link" )
set ( DRVDEF "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.so.def" )
set(HSA_SHARED_LINK_FLAGS "-Wl,-Bdynamic -Wl,-z,noexecstack -Wl,${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR} -Wl,--version-script=${DRVDEF} -Wl,--enable-new-dtags")
else()
target_sources(${CORE_RUNTIME_TARGET} PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.dll.def")
set(HSA_SHARED_LINK_FLAGS "")
endif()
target_compile_options(${CORE_RUNTIME_TARGET} PRIVATE ${HSA_CXX_FLAGS})
#target_link_options not available prior to CMake 3.13
@@ -165,13 +187,9 @@ set_property(TARGET ${CORE_RUNTIME_TARGET} PROPERTY LINK_FLAGS ${HSA_SHARED_LINK
## Source files.
set ( SRCS core/driver/driver.cpp
core/driver/kfd/amd_kfd_driver.cpp
core/driver/xdna/amd_xdna_driver.cpp
core/util/lnx/os_linux.cpp
core/util/small_heap.cpp
core/util/timer.cpp
core/util/flag.cpp
core/runtime/amd_aie_agent.cpp
core/runtime/amd_aie_aql_queue.cpp
core/runtime/amd_blit_kernel.cpp
core/runtime/amd_blit_sdma.cpp
core/runtime/amd_cpu_agent.cpp
@@ -208,12 +226,22 @@ set ( SRCS core/driver/driver.cpp
libamdhsacode/amd_hsa_code.cpp
libamdhsacode/amd_core_dump.cpp )
if(UNIX)
set(SRC_OS core/util/lnx/os_linux.cpp)
set(SRC_XDNA core/driver/xdna/amd_xdna_driver.cpp
core/runtime/amd_aie_agent.cpp
core/runtime/amd_aie_aql_queue.cpp)
else()
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE NOMINMAX)
set(SRC_OS core/util/win/os_win.cpp)
endif()
if ( BUILD_THUNK_VIRTIO )
list(APPEND SRCS core/driver/virtio/amd_kfd_virtio_driver.cpp)
target_compile_definitions(hsa-runtime64 PRIVATE HSAKMT_VIRTIO_ENABLED=1)
endif()
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${SRCS} )
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${SRCS} ${SRC_OS} ${SRC_XDNA} )
## Depend on trap handler target.
add_subdirectory( ${CMAKE_CURRENT_SOURCE_DIR}/core/runtime/trap_handler )
@@ -233,10 +261,15 @@ if (${PC_SAMPLING_SUPPORT})
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${PCS_SRCS} )
endif()
if ( NOT DEFINED IMAGE_SUPPORT AND CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64|loongarch64" )
set ( IMAGE_SUPPORT ON )
endif()
if (UNIX)
if ( NOT DEFINED IMAGE_SUPPORT AND CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64|loongarch64" )
set ( IMAGE_SUPPORT ON )
endif()
set ( IMAGE_SUPPORT ${IMAGE_SUPPORT} CACHE BOOL "Build with image support (default: ON for x86, OFF elsewise)." )
else()
# Force IMAGE_SUPPORT to be OFF
set(IMAGE_SUPPORT OFF CACHE BOOL "Build with image support (forced to OFF)" FORCE)
endif()
## Optional image module defintions.
if(${IMAGE_SUPPORT})
@@ -305,30 +338,45 @@ if(${IMAGE_SUPPORT})
endif()
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE elf::elf dl pthread rt )
if (UNIX)
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE elf::elf dl pthread rt )
else()
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE Ws2_32 )
target_link_directories(${CORE_RUNTIME_TARGET} PRIVATE ${DXCORE_LIB_PATH})
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE dxcore)
endif()
# For static package rocprofiler-register dependency is not required
# Link to hsakmt target for shared library builds
# Link to hsakmt-staticdrm target for static library builds
if( BUILD_SHARED_LIBS )
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt PkgConfig::drm)
if( BUILD_THUNK_VIRTIO )
message(STATUS "Building with virtio support")
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt_virtio)
endif()
find_package(rocprofiler-register)
if(rocprofiler-register_FOUND)
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HSA_ROCPROFILER_REGISTER=1
HSA_VERSION_MAJOR=${VERSION_MAJOR}
HSA_VERSION_MINOR=${VERSION_MINOR}
HSA_VERSION_PATCH=${VERSION_PATCH})
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE rocprofiler-register::rocprofiler-register)
set(HSA_DEP_ROCPROFILER_REGISTER ON CACHE INTERNAL "")
if (UNIX)
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt PkgConfig::drm)
if( BUILD_THUNK_VIRTIO )
message(STATUS "Building with virtio support")
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt_virtio)
endif()
find_package(rocprofiler-register)
if(rocprofiler-register_FOUND)
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HSA_ROCPROFILER_REGISTER=1
HSA_VERSION_MAJOR=${VERSION_MAJOR}
HSA_VERSION_MINOR=${VERSION_MINOR}
HSA_VERSION_PATCH=${VERSION_PATCH})
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE rocprofiler-register::rocprofiler-register)
set(HSA_DEP_ROCPROFILER_REGISTER ON CACHE INTERNAL "")
else()
set(HSA_DEP_ROCPROFILER_REGISTER OFF CACHE INTERNAL "")
endif() # end rocprofiler-register_FOUND
else()
set(HSA_DEP_ROCPROFILER_REGISTER OFF CACHE INTERNAL "")
endif() # end rocprofiler-register_FOUND
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt)
endif()
else()
include_directories(${drm_INCLUDE_DIRS})
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
if (UNIX)
include_directories(${drm_INCLUDE_DIRS})
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
else()
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
endif()
endif()#end BUILD_SHARED_LIBS
## Set the VERSION and SOVERSION values
@@ -350,8 +398,11 @@ if( NOT ${BUILD_SHARED_LIBS} )
## Add external link requirements.
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE hsakmt-staticdrm::hsakmt-staticdrm )
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE elf::elf dl pthread rt )
if (UNIX)
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE elf::elf dl pthread rt )
else()
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE rt )
endif()
install ( TARGETS ${CORE_RUNTIME_NAME} EXPORT ${CORE_RUNTIME_NAME}Targets )
endif()
@@ -404,10 +455,12 @@ install(FILES ${CMAKE_CURRENT_BINARY_DIR}/${CORE_RUNTIME_NAME}-config.cmake ${CM
# Install build files needed only when using a static build.
if( NOT ${BUILD_SHARED_LIBS} )
# libelf find package module
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/FindLibElf.cmake ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/COPYING-CMAKE-SCRIPTS
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
COMPONENT dev)
if (UNIX)
# libelf find package module
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/FindLibElf.cmake ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/COPYING-CMAKE-SCRIPTS
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
COMPONENT dev)
endif()
# Linker script (defines function aliases)
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR}
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
+72 -51
Просмотреть файл
@@ -17,57 +17,78 @@ if (LIBELF_FOUND)
return()
endif (LIBELF_FOUND)
find_path (LIBELF_INCLUDE_DIRS
NAMES
libelf.h
PATHS
/usr/include
/usr/include/libelf
/usr/local/include
/usr/local/include/libelf
/opt/local/include
/opt/local/include/libelf
ENV CPATH)
find_library (LIBELF_LIBRARIES
NAMES
elf
PATHS
/usr/lib
/usr/lib64
/usr/local/lib
/usr/local/lib64
/opt/local/lib
/opt/local/lib64
ENV LIBRARY_PATH
ENV LD_LIBRARY_PATH)
include (FindPackageHandleStandardArgs)
# handle the QUIETLY and REQUIRED arguments and set LIBELF_FOUND to TRUE if all listed variables are TRUE
FIND_PACKAGE_HANDLE_STANDARD_ARGS(LibElf DEFAULT_MSG
LIBELF_LIBRARIES
LIBELF_INCLUDE_DIRS)
SET(CMAKE_REQUIRED_LIBRARIES elf)
if (CMAKE_CXX_COMPILER_LOADED)
INCLUDE(CheckCXXSourceCompiles)
CHECK_CXX_SOURCE_COMPILES("#include <libelf.h>
int main() {
Elf *e = (Elf*)0;
size_t sz;
elf_getshdrstrndx(e, &sz);
return 0;
}" ELF_GETSHDRSTRNDX)
if (UNIX)
find_path (LIBELF_INCLUDE_DIRS
NAMES
libelf.h
PATHS
/usr/include
/usr/include/libelf
/usr/local/include
/usr/local/include/libelf
/opt/local/include
/opt/local/include/libelf
ENV CPATH)
find_library (LIBELF_LIBRARIES
NAMES
elf
PATHS
/usr/lib
/usr/lib64
/usr/local/lib
/usr/local/lib64
/opt/local/lib
/opt/local/lib64
ENV LIBRARY_PATH
ENV LD_LIBRARY_PATH)
include (FindPackageHandleStandardArgs)
# handle the QUIETLY and REQUIRED arguments and set LIBELF_FOUND to TRUE if all listed variables are TRUE
FIND_PACKAGE_HANDLE_STANDARD_ARGS(LibElf DEFAULT_MSG
LIBELF_LIBRARIES
LIBELF_INCLUDE_DIRS)
SET(CMAKE_REQUIRED_LIBRARIES elf)
if (CMAKE_CXX_COMPILER_LOADED)
INCLUDE(CheckCXXSourceCompiles)
CHECK_CXX_SOURCE_COMPILES("#include <libelf.h>
int main() {
Elf *e = (Elf*)0;
size_t sz;
elf_getshdrstrndx(e, &sz);
return 0;
}" ELF_GETSHDRSTRNDX)
else()
set ( ELF_GETSHDRSTRNDX "TRUE" )
endif(CMAKE_CXX_COMPILER_LOADED)
mark_as_advanced(LIBELF_INCLUDE_DIRS LIBELF_LIBRARIES ELF_GETSHDRSTRNDX)
if(LIBELF_FOUND)
add_library(elf::elf UNKNOWN IMPORTED)
set_property(TARGET elf::elf PROPERTY IMPORTED_LOCATION ${LIBELF_LIBRARIES})
set_property(TARGET elf::elf PROPERTY INTERFACE_INCLUDE_DIRECTORIES ${LIBELF_INCLUDE_DIRS})
endif()
else()
set ( ELF_GETSHDRSTRNDX "TRUE" )
endif(CMAKE_CXX_COMPILER_LOADED)
find_path(ROCR_LIBELF_INCLUDE_DIR libelf.h
HINTS
${AMD_LIBELF_PATH}
PATHS
${CMAKE_SOURCE_DIR}/hsail-compiler/lib/loaders/elf/utils/libelf
${CMAKE_SOURCE_DIR}/../hsail-compiler/lib/loaders/elf/utils/libelf
${CMAKE_SOURCE_DIR}/../../hsail-compiler/lib/loaders/elf/utils/libelf
NO_DEFAULT_PATH)
mark_as_advanced(LIBELF_INCLUDE_DIRS LIBELF_LIBRARIES ELF_GETSHDRSTRNDX)
if(LIBELF_FOUND)
add_library(elf::elf UNKNOWN IMPORTED)
set_property(TARGET elf::elf PROPERTY IMPORTED_LOCATION ${LIBELF_LIBRARIES})
set_property(TARGET elf::elf PROPERTY INTERFACE_INCLUDE_DIRECTORIES ${LIBELF_INCLUDE_DIRS})
message("=> LibElf paths:" ${CMAKE_CURRENT_BINARY_DIR} ${ROCR_LIBELF_INCLUDE_DIR})
if (${BUILD_SHARED_LIBS})
mark_as_advanced(ROCR_LIBELF_INCLUDE_DIR)
add_subdirectory("${ROCR_LIBELF_INCLUDE_DIR}" ${CMAKE_CURRENT_BINARY_DIR}/libelf)
endif()
set(USE_AMD_LIBELF "yes" CACHE FORCE "")
set(AMD_ELFTOOLCHAIN_DIR ${ROCR_LIBELF_INCLUDE_DIR}/../..;${ROCR_LIBELF_INCLUDE_DIR}/../common/win32;${ROCR_LIBELF_INCLUDE_DIR}/../common)
set(ROCR_LIBELF_INCLUDE_DIR ${ROCR_LIBELF_INCLUDE_DIR};${AMD_ELFTOOLCHAIN_DIR})
set(LIBELF_INCLUDE_DIR ${ROCR_LIBELF_INCLUDE_DIR})
endif()
+37 -2
Просмотреть файл
@@ -45,9 +45,11 @@
#include <memory>
#include <string>
#if defined(__linux__)
#include <amdgpu_drm.h>
#include <link.h>
#include <sys/ioctl.h>
#endif
#include "hsakmt/hsakmt.h"
@@ -55,11 +57,16 @@
#include "core/inc/amd_memory_region.h"
#include "core/inc/runtime.h"
#if defined(_WIN32)
#include "loader/executable.hpp"
#endif
extern r_debug _amdgpu_r_debug;
namespace rocr {
namespace AMD {
#if defined(__linux__)
static_assert(
(sizeof(core::ShareableHandle::handle) >= sizeof(amdgpu_bo_handle)) &&
(alignof(core::ShareableHandle::handle) >= alignof(amdgpu_bo_handle)),
@@ -82,6 +89,7 @@ __forceinline uint64_t drm_perm(hsa_access_permission_t perm) {
}
} // namespace
#endif
KfdDriver::KfdDriver(std::string devnode_name)
: core::Driver(core::DriverType::KFD, std::move(devnode_name)) {}
@@ -425,6 +433,7 @@ hsa_status_t KfdDriver::ExportDMABuf(void *mem, size_t size, int *dmabuf_fd,
hsa_status_t KfdDriver::ImportDMABuf(int dmabuf_fd, core::Agent &agent,
core::ShareableHandle &handle) {
#if defined(__linux__)
auto &gpu_agent = static_cast<GpuAgent &>(agent);
amdgpu_bo_import_result res;
auto ret = DRM_CALL(amdgpu_bo_import(
@@ -433,12 +442,16 @@ hsa_status_t KfdDriver::ImportDMABuf(int dmabuf_fd, core::Agent &agent,
return HSA_STATUS_ERROR;
handle.handle = reinterpret_cast<uint64_t>(res.buf_handle);
#else
assert(!"Unimplemented!");
#endif
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdDriver::Map(core::ShareableHandle handle, void *mem,
size_t offset, size_t size,
hsa_access_permission_t perms) {
#if defined(__linux__)
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
if (!ldrm_bo)
return HSA_STATUS_ERROR;
@@ -446,12 +459,15 @@ hsa_status_t KfdDriver::Map(core::ShareableHandle handle, void *mem,
if (DRM_CALL(amdgpu_bo_va_op(ldrm_bo, offset, size, reinterpret_cast<uint64_t>(mem),
drm_perm(perms), AMDGPU_VA_OP_MAP)) != 0)
return HSA_STATUS_ERROR;
#else
assert(!"Unimplemented!");
#endif
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdDriver::Unmap(core::ShareableHandle handle, void *mem,
size_t offset, size_t size) {
#if defined(__linux__)
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
if (!ldrm_bo)
return HSA_STATUS_ERROR;
@@ -459,11 +475,14 @@ hsa_status_t KfdDriver::Unmap(core::ShareableHandle handle, void *mem,
if (DRM_CALL(amdgpu_bo_va_op(ldrm_bo, offset, size, reinterpret_cast<uint64_t>(mem), 0,
AMDGPU_VA_OP_UNMAP)) != 0)
return HSA_STATUS_ERROR;
#else
assert(!"Unimplemented!");
#endif
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdDriver::ReleaseShareableHandle(core::ShareableHandle &handle) {
#if defined(__linux__)
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
if (!ldrm_bo)
return HSA_STATUS_ERROR;
@@ -473,6 +492,9 @@ hsa_status_t KfdDriver::ReleaseShareableHandle(core::ShareableHandle &handle) {
return HSA_STATUS_ERROR;
handle = {};
#else
assert(!"Unimplemented!");
#endif
return HSA_STATUS_SUCCESS;
}
@@ -650,6 +672,7 @@ hsa_status_t KfdDriver::DeregisterMemory(void* ptr) const {
hsa_status_t KfdDriver::MakeMemoryResident(const void* mem, size_t size, uint64_t* alternate_va,
const HsaMemMapFlags* mem_flags, uint32_t num_nodes,
const uint32_t* nodes) const {
#if defined(__linux__)
if (mem_flags == nullptr && nodes == nullptr) {
if (HSAKMT_CALL(hsaKmtMapMemoryToGPU(const_cast<void*>(mem), size, alternate_va)) !=
HSAKMT_STATUS_SUCCESS) {
@@ -663,7 +686,19 @@ hsa_status_t KfdDriver::MakeMemoryResident(const void* mem, size_t size, uint64_
debug_print("Invalid memory flags ptr:%p nodes ptr:%p\n", mem_flags, nodes);
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
#else
assert(num_nodes > 0);
assert(nodes != NULL);
*alternate_va = 0;
const HSAKMT_STATUS status =
HSAKMT_CALL(hsaKmtMapMemoryToGPUNodes(const_cast<void*>(mem), size, alternate_va, *mem_flags,
num_nodes, const_cast<uint32_t*>(nodes)));
if (status != HSAKMT_STATUS_SUCCESS) {
return HSA_STATUS_ERROR;
}
#endif
return HSA_STATUS_SUCCESS;
}
+3 -4
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2024-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -43,11 +43,10 @@
#ifndef HSA_RUNTME_CORE_INC_AMD_AVAILABLE_DRIVERS_H_
#define HSA_RUNTME_CORE_INC_AMD_AVAILABLE_DRIVERS_H_
#ifdef __linux__
#include "core/inc/amd_kfd_driver.h"
#include "core/inc/amd_xdna_driver.h"
#ifdef __linux__
#include "core/inc/amd_xdna_driver.h"
#endif
#endif // header guard
+1
Просмотреть файл
@@ -116,6 +116,7 @@ class BlitKernel : public core::Blit {
virtual bool GangLeader() const override { return false; }
const uint16_t kInvalidPacketHeader = HSA_PACKET_TYPE_INVALID;
private:
union KernelArgs {
struct __ALIGNED__(16) {
+1
Просмотреть файл
@@ -482,6 +482,7 @@ public:
/// @brief Finds the handle of executable to which @p device_address
/// belongs. Return NULL handle if device address is invalid.
#undef FindExecutable
virtual hsa_executable_t FindExecutable(uint64_t device_address) = 0;
/// @brief Returns host address given @p device_address. If @p device_address
+7 -3
Просмотреть файл
@@ -51,11 +51,12 @@
#include <tuple>
#include <utility>
#include <thread>
#include <sys/un.h>
#if defined(__linux__)
#include <sys/un.h>
#include <xf86drm.h>
#include <amdgpu.h>
#else
#include <hsakmt/drm/amdgpu.h>
#endif
#include "core/inc/hsa_ext_interface.h"
@@ -232,6 +233,7 @@ class Runtime {
/// @param [in] size Copy size in bytes.
///
/// @retval ::HSA_STATUS_SUCCESS if memory copy is successful and completed.
#undef CopyMemory
hsa_status_t CopyMemory(void* dst, const void* src, size_t size);
/// @brief Non-blocking memory copy from src to dst.
@@ -302,6 +304,7 @@ class Runtime {
/// @param [in] count Number of uint32_t element to be set.
///
/// @retval ::HSA_STATUS_SUCCESS if memory fill is successful and completed.
#undef FillMemory
hsa_status_t FillMemory(void* ptr, uint32_t value, size_t count);
/// @brief Set agents as the whitelist to access ptr.
@@ -517,7 +520,8 @@ class Runtime {
static bool IsGPUDriver(DriverType driver_type) {
return driver_type == core::DriverType::KFD
#ifdef HSAKMT_VIRTIO_ENABLED
#if defined(HSAKMT_VIRTIO_ENABLED) && defined(__linux__)
|| driver_type == core::DriverType::KFD_VIRTIO
#endif
;
+4
Просмотреть файл
@@ -44,7 +44,11 @@
#define HSA_RUNTIME_CORE_INC_THUNK_LOADER_H
#include <string>
#if defined(__linux__)
#include <amdgpu.h>
#else
#include "hsakmt/drm/amdgpu.h"
#endif
#include "hsakmt/hsakmttypes.h"
class DtifPlatform;
+4 -1
Просмотреть файл
@@ -50,8 +50,11 @@
#include <unistd.h>
#endif
#include <algorithm>
#ifdef _WIN32
#define WIN32_NO_STATUS
#include <Windows.h>
#undef WIN32_NO_STATUS
#endif
#include <stdio.h>
@@ -967,7 +970,7 @@ void AqlQueue::HandleInsufficientScratch(hsa_signal_value_t& error_code,
maxGroupsPerEngine < 16 &&
lanes_per_group * maxGroupsPerEngine < 256) {
uint64_t groups_per_interleave = (256 + lanes_per_group - 1) / lanes_per_group;
maxGroupsPerEngine = Min(groups_per_interleave, 16ul);
maxGroupsPerEngine = Min(groups_per_interleave, uint64_t(16ul));
}
// Populate all engines at max group occupancy, then clip down to device limits.
+6 -2
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -903,9 +903,13 @@ void BlitKernel::PopulateQueue(uint64_t index, uint64_t code_handle, void* args,
// Ensure the packet body is written as header may get reordered when writing over PCIE
_mm_sfence();
}
#if defined(__linux__)
__atomic_store_n(&(queue_buffer[index & queue_bitmask_].full_header),
kDispatchPacketHeader | packet.setup << 16, __ATOMIC_RELEASE);
#else
std::atomic_ref<uint32_t> atomic_header(queue_buffer[index & queue_bitmask_].full_header);
atomic_header.store(kDispatchPacketHeader | packet.setup << 16, std::memory_order_release);
#endif
LogPrint(HSA_AMD_LOG_FLAG_AQL,
"HWq=%p, id=%lu, Dispatch Header = "
"0x%x (type=%d, barrier=%d, acquire=%d, release=%d), "
+7 -6
Просмотреть файл
@@ -47,6 +47,7 @@
#include <cmath>
#include <cstring>
#include <limits>
#include <core/util/utils.h>
#include "core/inc/amd_gpu_agent.h"
#include "core/inc/amd_memory_region.h"
@@ -855,7 +856,7 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
// width | 16 ensures that we don't return a higher element than is supported and avoids
// issues with 0.
auto maxAlignedElement = [](size_t width) {
return __builtin_ctz(width | 16);
return rocr::os::Ctz(width | 16);
};
// GFX12 or later use a different packet format that is incompatible (fields changed in size and location).
@@ -872,7 +873,7 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
// Find maximum element that describes the pitch and slice.
// Pitch and slice must both be represented in units of elements. No element larger than this
// may be used in any tile as the pitches would not be exactly represented.
int max_ele = Min(maxAlignedElement(src->pitch), maxAlignedElement(dst->pitch));
auto max_ele = Min(maxAlignedElement(src->pitch), maxAlignedElement(dst->pitch));
if (range->z != 1) // Only need to consider slice if HW will copy along Z.
max_ele = Min(max_ele, maxAlignedElement(src->slice), maxAlignedElement(dst->slice));
@@ -895,8 +896,8 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
src and dst base has already been checked for DWORD alignment so we only need to consider the
offset here.
*/
int min_ele = Min(max_ele, maxAlignedElement(range->x), maxAlignedElement(src_offset->x % 4),
maxAlignedElement(dst_offset->x % 4));
auto min_ele = Min(max_ele, maxAlignedElement(range->x), maxAlignedElement(src_offset->x % 4),
maxAlignedElement(dst_offset->x % 4));
// Check that pitch and slice can be represented in the tile with the smallest element
if ((src->pitch >> min_ele) > max_pitch || (dst->pitch >> min_ele) > max_pitch)
@@ -916,8 +917,8 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
// Get largest element which describes the start of this tile after its base address has
// been aligned. Base addresses must be DWORD (4 byte) aligned.
int aligned_ele = Min(maxAlignedElement((src_offset->x + x) % 4),
maxAlignedElement((dst_offset->x + x) % 4), max_ele);
auto aligned_ele = Min(maxAlignedElement((src_offset->x + x) % 4),
maxAlignedElement((dst_offset->x + x) % 4), max_ele);
// Get largest permissible element which exactly covers width
int element = Min(maxAlignedElement(width), aligned_ele);
+3 -3
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2023, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -1026,8 +1026,8 @@ hsa_status_t GpuAgent::DmaCopy(void* dst, core::Agent& dst_agent,
std::vector<core::Signal*>& dep_signals,
core::Signal& out_signal) {
// Recommended SDMA engine copies only have gang factor 1
uint32_t rec_sdma_eng = ffs(rec_sdma_eng_id_peers_info_[dst_agent.public_handle().handle]);
uint32_t rec_sdma_eng =
rocr::os::Ffs(rec_sdma_eng_id_peers_info_[dst_agent.public_handle().handle]);
if (rec_sdma_eng)
return DmaCopyOnEngine(dst, dst_agent, src, src_agent, size,
dep_signals, out_signal, rec_sdma_eng, false);
+9 -21
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -44,12 +44,16 @@
#include "core/inc/runtime.h"
#include <assert.h>
#if defined(__linux__)
#include <link.h>
#include <linux/limits.h>
#include <sys/mman.h>
#include <stdlib.h>
#include <unistd.h>
#else
#include <cstdint>
#endif
#include <stdlib.h>
#include <cstring>
#include <fstream>
#include <iomanip>
@@ -92,7 +96,7 @@ std::string EncodePathname(const char *file_path) {
}
std::string GetUriFromMemoryAddress(const void *memory, size_t size) {
pid_t pid = getpid();
int pid = getpid();
std::ostringstream uri_stream;
uri_stream << "memory://" << pid
<< "#offset=0x" << std::hex << (uintptr_t)memory << std::dec
@@ -313,23 +317,7 @@ hsa_status_t CodeObjectReaderImpl::SetFile(
code_object_size = _code_object_size;
is_mmap = true;
#else
if (__lseek__(_code_object_file_descriptor, 0, SEEK_SET) == (off_t)-1) {
return HSA_STATUS_ERROR_INVALID_FILE;
}
std::unique_ptr<unsigned char> memory(new unsigned char[_code_object_size]);
if (!memory) {
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
if (__read__(_code_object_file_descriptor, mmap_memory,
_code_object_size) != _code_object_size) {
return HSA_STATUS_ERROR_INVALID_FILE;
}
mmap_memory = memory.release();
mmap_size = _code_object_size;
code_object_memory = memory;
code_object_size = _code_object_size;
//@todo May need an implementation in Windows
#endif // !defined(_WIN32) && !defined(_WIN64)
uri = GetUriFromFile(_code_object_file_descriptor, _code_object_offset,
+8 -3
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -40,6 +40,11 @@
//
////////////////////////////////////////////////////////////////////////////////
#if defined(__linux__)
#include <unistd.h>
#else
#include <cstdint>
#endif
#include "core/inc/amd_memory_region.h"
#include <algorithm>
@@ -48,15 +53,15 @@
#include "core/inc/amd_cpu_agent.h"
#include "core/inc/amd_gpu_agent.h"
#include "core/util/utils.h"
#include "core/util/os.h"
#include "core/inc/exceptions.h"
#include <unistd.h>
namespace rocr {
namespace AMD {
// Tracks aggregate size of system memory available on platform
size_t MemoryRegion::max_sysmem_alloc_size_ = 0;
const size_t MemoryRegion::kPageSize_ = sysconf(_SC_PAGESIZE);
const size_t MemoryRegion::kPageSize_ = os::PageSize();
MemoryRegion::MemoryRegion(bool fine_grain, bool kernarg, bool full_profile,
bool extended_scope_fine_grain, bool user_visible, core::Agent* owner,
+13 -8
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -58,8 +58,6 @@
#include <unordered_map>
#include <vector>
#include <link.h>
#include "core/inc/amd_aie_agent.h"
#include "core/inc/amd_available_drivers.h"
#include "core/inc/amd_cpu_agent.h"
@@ -72,7 +70,12 @@
#include "core/inc/amd_virtio_driver.h"
#endif
extern r_debug _amdgpu_r_debug;
#if defined(__linux__)
#include <link.h>
#else
#include "loader/executable.hpp"
#endif
extern r_debug _amdgpu_r_debug_r;
namespace rocr {
namespace AMD {
@@ -81,17 +84,17 @@ namespace {
const std::array<std::function<hsa_status_t(std::unique_ptr<core::Driver>&)>,
#if _WIN32
0
1
#elif __linux__
static_cast<size_t>(core::DriverType::NUM_DRIVER_TYPES)
#endif
>
discover_driver_funcs = {
KfdDriver::DiscoverDriver
#ifdef __linux__
KfdDriver::DiscoverDriver,
XdnaDriver::DiscoverDriver,
, XdnaDriver::DiscoverDriver
#ifdef HSAKMT_VIRTIO_ENABLED
KfdVirtioDriver::DiscoverDriver,
, KfdVirtioDriver::DiscoverDriver
#endif
#endif
};
@@ -181,8 +184,10 @@ GpuAgent* DiscoverGpu(HSAuint32 node_id, HsaNodeProperties& node_prop, bool xnac
}
void DiscoverAie(uint32_t node_id, HsaNodeProperties& node_prop) {
#if defined(__linux__)
AieAgent* aie = new AieAgent(node_id, node_prop);
core::Runtime::runtime_singleton_->RegisterAgent(aie, true);
#endif
}
void RegisterLinkInfo(const std::unique_ptr<core::Driver>& driver, uint32_t node_id,
+10 -2
Просмотреть файл
@@ -3,7 +3,7 @@
## The University of Illinois/NCSA
## Open Source License (NCSA)
##
## Copyright (c) 2014-2023, Advanced Micro Devices, Inc. All rights reserved.
## Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
##
## Developed by:
##
@@ -148,11 +148,19 @@ function(generate_bytecodeStrm HeaderFILE)
COPYONLY)
# Add a custom command to generate the header file
if (UNIX)
add_custom_command(OUTPUT ${HeaderFILE}.h
COMMAND ${CMAKE_CURRENT_BINARY_DIR}/create_blit_shader_header.sh ${ARG_LIST} ${HSACO_TARG_LIST}
COMMENT "Collating blit shaders..."
DEPENDS ${HSACO_TARG_LIST} ${CMAKE_CURRENT_BINARY_DIR}/create_blit_shader_header.sh)
else()
find_package(Python3 COMPONENTS Interpreter REQUIRED)
add_custom_command(
OUTPUT ${HeaderFILE}.h
COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/create_blit_shader_header.py ${ARG_LIST} ${HSACO_TARG_LIST}
COMMENT "Collating blit shaders..."
DEPENDS ${HSACO_TARG_LIST} create_blit_shader_header.py)
endif()
# Add a custom target that depends on the header file
add_custom_target(${HeaderFILE} DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/${HeaderFILE}.h)
@@ -0,0 +1,71 @@
################################################################################
##
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
##
## SPDX-License-Identifier: MIT
##
################################################################################
import sys
def GetSize(fileobject):
fileobject.seek(0,2) # move the cursor to the end of the file
size = fileobject.tell()
return size
def DumpFile(header, input_name):
try:
with open(input_name, "rb") as binary_file:
# Read the entire content of the file as bytes
binary_data = binary_file.read()
file_size = GetSize(binary_file)
#print(f"Binary size: {file_size}")
# Reset file pointer
binary_file.seek(0)
parts = input_name.split('.')
file_name = parts[0]
content = f"unsigned char {file_name}""[] = {\n "
header.write(content)
line = 0
count = 0
for byte_value in binary_data:
count += 1
padded_hex = '{:02x}'.format(byte_value)
if (count != file_size):
header.write(f"0x{padded_hex},")
else:
header.write(f"0x{padded_hex}")
line += 1
if (line == 12):
header.write(f"\n ")
line = 0
else:
header.write(f" ")
header.write("\n};\nunsigned int "f"{file_name}_len = {file_size};\n")
except FileNotFoundError:
print(f"Error: The file {input_name} was not found.")
except Exception as e:
print(f"An error occurred: {e}")
if len(sys.argv) > 1:
header_name = sys.argv[1];
with open(header_name, 'w') as header:
header.write("//==============================================================================\n")
header.write("// This file is automatically generated during build process, don't modify it\n")
header.write("//==============================================================================\n\n")
header.write("namespace rocr {\n")
header.write("namespace AMD {\n\n")
for i, arg in enumerate(sys.argv):
if (i > 1):
#print(f"File {i}: {arg}\n")
DumpFile(header, arg)
header.write("} // namespace AMD\n")
header.write("} // namespace rocr\n\n")
else:
print("Empty arguments!")
+2
Просмотреть файл
@@ -835,12 +835,14 @@ hsa_status_t hsa_amd_agent_iterate_memory_pools(
reinterpret_cast<hsa_status_t (*)(hsa_region_t memory_pool,
void *data)>(callback),
data);
#if defined(__linux__)
case core::Agent::kAmdAieDevice:
return reinterpret_cast<const AMD::AieAgent *>(agent)->VisitRegion(
false,
reinterpret_cast<hsa_status_t (*)(hsa_region_t memory_pool,
void *data)>(callback),
data);
#endif
case core::Agent::kAmdGpuDevice:
return reinterpret_cast<const AMD::GpuAgentInt *>(agent)->VisitRegion(
false,
+3 -2
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -42,8 +42,9 @@
#include "core/inc/hsa_ven_amd_loader_impl.h"
#include "core/inc/amd_hsa_loader.hpp"
#include "core/inc/runtime.h"
#include "core/inc/amd_gpu_agent.h"
#include "core/inc/amd_hsa_loader.hpp"
namespace rocr {
+73 -18
Просмотреть файл
@@ -48,13 +48,19 @@
#include <string>
#include <vector>
#include <list>
#if defined(__linux__)
#include <link.h>
#include <dlfcn.h>
#include <amdgpu_drm.h>
#include <sys/mman.h>
#include <sys/socket.h>
#include <sys/un.h>
#else
#define debug_warning(__VA_ARGS__)
#endif
#include <iostream>
#include <thread>
#include <chrono>
#include "core/inc/runtime.h"
#include "core/inc/hsa_table_interface.h"
@@ -97,8 +103,12 @@
ROCPROFILER_REGISTER_DEFINE_IMPORT(hsa, ROCP_REG_VERSION)
#endif
#if defined(__linux__)
const char rocrbuildid[] __attribute__((used)) = "ROCR BUILD ID: " STRING(ROCR_BUILD_ID);
#else
#include "loader/executable.hpp"
const char rocrbuildid[] = "ROCR BUILD ID: " STRING(ROCR_BUILD_ID);
#endif
extern r_debug _amdgpu_r_debug;
namespace rocr {
@@ -591,7 +601,7 @@ hsa_status_t Runtime::CopyMemoryOnEngine(void* dst, core::Agent* dst_agent, cons
core::Agent* copy_agent = (src_gpu) ? src_agent : dst_agent;
// engine_id is single bitset unique.
int engine_offset = ffs(engine_id);
int engine_offset = rocr::os::Ffs(engine_id);
if (!engine_id || !!((engine_id >> engine_offset))) {
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
@@ -1178,6 +1188,7 @@ hsa_status_t Runtime::SetPtrInfoData(const void* ptr, void* userptr) {
// Send the dmabuf_fd to from process via Unix socket
static int SendDmaBufFd(int socket, int dmabuf_fd) {
#if defined(__linux__)
char iov_buf[1];
struct msghdr msg = {0};
char buf[CMSG_SPACE(sizeof(dmabuf_fd))];
@@ -1205,10 +1216,15 @@ static int SendDmaBufFd(int socket, int dmabuf_fd) {
ssize_t sent = sendmsg(socket, &msg, 0);
return (sent < 0) ? -1 : 0;
#else
assert(!"Unimplemented!");
return 0;
#endif
}
// Receive the dmabuf_fd to from process via Unix socket
static int ReceiveDmaBufFd(int socket) {
#if defined(__linux__)
struct msghdr msg = {0};
// The struct iovec is needed, even if it points to minimal data
@@ -1233,6 +1249,10 @@ static int ReceiveDmaBufFd(int socket) {
memcpy(&fd, CMSG_DATA(cmsg), sizeof(fd));
return fd;
#else
assert(!"Unimplemented!");
return 0;
#endif
}
#define IPC_SOCK_SERVER_DMABUF_FD_HANDLE_LENGTH 64
@@ -1363,6 +1383,7 @@ hsa_status_t Runtime::IPCCreate(void* ptr, size_t len, hsa_amd_ipc_memory_t* han
close(dmabuf_fd);
ScopedAcquire<KernelMutex> lock(&ipc_sock_server_lock_);
#if defined(__linux__)
if (!ipc_sock_server_conns_.size()) { // create new runtime socket server
struct sockaddr_un address;
ipc_sock_server_fd_ = socket(AF_UNIX, SOCK_STREAM, 0);
@@ -1393,7 +1414,9 @@ hsa_status_t Runtime::IPCCreate(void* ptr, size_t len, hsa_amd_ipc_memory_t* han
// as the attach life cycle is unknown.
os::CreateThread(AsyncIPCSockServerConnLoop, NULL);
}
#else
assert(!"Unimplemented! Do we really need this?");
#endif
ipc_sock_server_conns_[reinterpret_cast<uint64_t>(ptr)] = len;
// TODO: fragment block discard for better memory performance causes memory violations
@@ -1406,7 +1429,6 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
amdgpu_bo_import_result *res,
unsigned int numNodes, HSAuint32 *nodes,
void **importAddress, HSAuint64 *importSize) {
struct sockaddr_un address;
int dmabuf_fd = -1, socket_fd = socket(AF_UNIX, SOCK_STREAM, 0);
assert(socket_fd > -1 && "DMA buffer could not be imported for IPC!");
if (socket_fd == -1) return -1;
@@ -1420,13 +1442,15 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
if (status) return -1;
char buf[IPC_SOCK_SERVER_DMABUF_FD_HANDLE_LENGTH];
memset(&address, 0, sizeof(struct sockaddr_un));
memset(buf, 0, sizeof(buf));
int timeoutLimitMs = 10000, timeoutMs = 0, timeoutIntervalMs = 1;
#if defined(__linux__)
struct sockaddr_un address;
memset(&address, 0, sizeof(struct sockaddr_un));
address.sun_family = AF_UNIX;
snprintf(address.sun_path, IPC_SOCK_SERVER_NAME_LENGTH, "xhsa%i", conn_handle);
address.sun_path[0] = 0; // first NULL char creates unlisted abstract socket
int timeoutLimitMs = 10000, timeoutMs = 0, timeoutIntervalMs = 1;
while (timeoutMs < timeoutLimitMs) {
if (connect(socket_fd, (struct sockaddr *) &address, sizeof(struct sockaddr_un))) {
timeoutMs += timeoutIntervalMs;
@@ -1435,7 +1459,9 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
break;
}
}
#else
assert(!"Unimplmented!");
#endif
MAKE_SCOPE_GUARD([&]() { close(socket_fd); });
if (timeoutMs >= timeoutLimitMs) return -1;
@@ -1545,6 +1571,7 @@ hsa_status_t Runtime::IPCAttach(const hsa_amd_ipc_memory_t* handle, size_t len,
dmaBufFDHandle = (dmaBufFDHandleHi << 32) | dmaBufFDHandleLo;
}
#if defined(__linux__)
if (num_agents == 0) {
amdgpu_bo_import_result res;
bool isDmabufSysMem = ipc_dmabuf_supported_ && importHandle.handle[3];
@@ -1575,6 +1602,9 @@ hsa_status_t Runtime::IPCAttach(const hsa_amd_ipc_memory_t* handle, size_t len,
*mapped_ptr = importAddress;
return HSA_STATUS_SUCCESS;
}
#else
assert(!"Unimplemented!");
#endif
HSAuint32* nodes = nullptr;
if (num_agents > tinyArraySize)
@@ -1602,6 +1632,7 @@ hsa_status_t Runtime::IPCDetach(void* ptr) {
const auto& it = allocation_map_.find(ptr);
if (it != allocation_map_.end()) {
if (it->second.region != nullptr) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
#if defined(__linux__)
if (it->second.ldrm_bo) {
if (DRM_CALL(amdgpu_bo_va_op(it->second.ldrm_bo, 0, it->second.size,
reinterpret_cast<uint64_t>(ptr), 0, AMDGPU_VA_OP_UNMAP)))
@@ -1610,6 +1641,9 @@ hsa_status_t Runtime::IPCDetach(void* ptr) {
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
ldrmImportCleaned = true;
}
#else
assert(!"Unimplemented!");
#endif
allocation_map_.erase(it);
lock.Release(); // Can't hold memory lock when using pointer info.
@@ -2325,6 +2359,7 @@ int fn_amdgpu_device_get_fd_nosupport(HsaAMDGPUDeviceHandle device_handle) {
int Runtime::GetAmdgpuDeviceArgs(Agent *agent, ShareableHandle handle,
int *drm_fd, uint64_t *cpu_addr) {
#if defined(__linux__)
int renderFd = fn_amdgpu_device_get_fd(static_cast<AMD::GpuAgent*>(agent)->libDrmDev());
if (renderFd < 0) return HSA_STATUS_ERROR;
@@ -2343,6 +2378,9 @@ int Runtime::GetAmdgpuDeviceArgs(Agent *agent, ShareableHandle handle,
*drm_fd = renderFd;
*cpu_addr = args.out.addr_ptr;
#else
assert(!"Unimplemented!");
#endif
return HSA_STATUS_SUCCESS;
}
@@ -2353,6 +2391,7 @@ void Runtime::CheckVirtualMemApiSupport() {
if (kfd_version.KernelInterfaceMajorVersion > 1 ||
(kfd_version.KernelInterfaceMajorVersion == 1 &&
kfd_version.KernelInterfaceMinorVersion >= 15)) {
#if defined(__linux__)
char* error;
fn_amdgpu_device_get_fd =
@@ -2365,6 +2404,9 @@ void Runtime::CheckVirtualMemApiSupport() {
} else {
virtual_mem_api_supported_ = true;
}
#else
virtual_mem_api_supported_ = false;
#endif
}
}
@@ -2379,7 +2421,7 @@ void Runtime::InitIPCDmaBufSupport() {
GetSystemInfo(HSA_AMD_SYSTEM_INFO_DMABUF_SUPPORTED, &dmabuf_supported);
if (!dmabuf_supported) return;
#if defined(__linux__)
char* error;
fn_amdgpu_device_get_fd =
(int (*)(HsaAMDGPUDeviceHandle device_handle))dlsym(
@@ -2391,6 +2433,9 @@ void Runtime::InitIPCDmaBufSupport() {
} else {
ipc_dmabuf_supported_ = !flag().enable_ipc_mode_legacy();
}
#else
ipc_dmabuf_supported_ = false;
#endif
}
void Runtime::LoadTools() {
@@ -3237,15 +3282,14 @@ hsa_status_t Runtime::VMemoryAddressReserve(void** va, size_t size, uint64_t add
void* addr = (void*)address;
HsaMemFlags memFlags = {};
if (!alignment)
alignment = sysconf(_SC_PAGE_SIZE);
if (!alignment) alignment = rocr::os::PageSize();
ScopedAcquire<KernelSharedMutex> lock(&memory_lock_);
if (flags & HSA_AMD_VMEM_ADDRESS_NO_REGISTER) {
size_t requested = size + alignment - sysconf(_SC_PAGE_SIZE);
auto mem = mmap(addr, requested, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE | MAP_NORESERVE, -1, 0);
if (mem == MAP_FAILED)
size_t requested = size + alignment - rocr::os::PageSize();
auto mem = rocr::os::ReserveMemory(addr, requested, alignment, rocr::os::MEM_PROT_RW);
if (mem == nullptr)
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
auto aligned = AlignUp(mem, alignment);
@@ -3253,8 +3297,10 @@ hsa_status_t Runtime::VMemoryAddressReserve(void** va, size_t size, uint64_t add
// Hint to enable THP for large host allocations which can help in performance gain
constexpr size_t kLargePageSize = 2*1024*1024;
if (size >= kLargePageSize) {
#if defined(__linux__)
if (madvise(aligned, size, MADV_HUGEPAGE))
debug_warning(false && "madvise with MADV_HUGEPAGE failed");
#endif
}
reserved_address_map_[aligned] = AddressHandle(mem, size, false);
@@ -3292,10 +3338,11 @@ hsa_status_t Runtime::VMemoryAddressFree(void* va, size_t size) {
if (it->second.use_count > 0) return HSA_STATUS_ERROR_RESOURCE_FREE;
if (it->second.registered) {
if (HSAKMT_CALL(hsaKmtFreeMemory(it->second.os_addr, size)) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
} else {
if (munmap(it->second.os_addr, size)) return HSA_STATUS_ERROR;
if (HSAKMT_CALL(hsaKmtFreeMemory(it->second.os_addr, size)) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
}
else if (!rocr::os::ReleaseMemory(it->second.os_addr, size))
return HSA_STATUS_ERROR;
reserved_address_map_.erase(it);
return HSA_STATUS_SUCCESS;
@@ -3488,10 +3535,10 @@ hsa_status_t Runtime::VMemoryHandleUnmap(void* va, size_t size) {
}
Runtime::MappedHandleAllowedAgent::MappedHandleAllowedAgent(
MappedHandle *mappedHandle, Agent *targetAgent, void *va, size_t size,
MappedHandle* _mappedHandle, Agent *targetAgent, void *va, size_t size,
hsa_access_permission_t perms)
: va(va), size(size), targetAgent(targetAgent), permissions(perms),
mappedHandle(mappedHandle) {
mappedHandle(_mappedHandle) {
// CPU agents have access as the memory is already mapped to the host.
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) return;
@@ -3527,6 +3574,7 @@ Runtime::MappedHandleAllowedAgent::~MappedHandleAllowedAgent() {
hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permission_t perms) {
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) {
#if defined(__linux__)
void* mapped_ptr =
mmap(va, size, PermissionsToMmapFlags(perms), MAP_SHARED | MAP_FIXED, mappedHandle->drm_fd,
reinterpret_cast<uint64_t>(mappedHandle->drm_cpu_addr));
@@ -3537,6 +3585,9 @@ hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permissi
shareable_handle, va, mappedHandle->offset, size, perms);
if (status != HSA_STATUS_SUCCESS)
return status;
#else
assert(!"Unimplemented!");
#endif
}
permissions = perms;
return HSA_STATUS_SUCCESS;
@@ -3544,8 +3595,12 @@ hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permissi
hsa_status_t Runtime::MappedHandleAllowedAgent::RemoveAccess() {
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) {
#if defined(__linux__)
if (munmap(va, size) != 0)
return HSA_STATUS_ERROR;
#else
assert(!"Unimplemented!");
#endif
return HSA_STATUS_SUCCESS;
} else {
return targetAgent->driver().Unmap(
+12 -3
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -51,6 +51,9 @@
#include "core/util/timer.h"
#include "core/inc/runtime.h"
#if defined(_WIN32)
#include "malloc.h"
#endif
namespace rocr {
namespace core {
@@ -234,8 +237,11 @@ uint32_t Signal::WaitMultiple(uint32_t signal_count, const hsa_signal_t* hsa_sig
MAKE_SCOPE_GUARD([&]() {
if (signal_count > small_size) delete[] evts;
});
#if defined(__linux__)
uint64_t event_age[unique_evts];
#else
auto event_age = reinterpret_cast<uint64_t*>(_alloca(unique_evts * sizeof(unique_evts)));
#endif
memset(event_age, 0, unique_evts * sizeof(uint64_t));
if (core::Runtime::runtime_singleton_->KfdVersion().supports_event_age)
for (uint32_t i = 0; i < unique_evts; i++)
@@ -367,8 +373,11 @@ uint32_t Signal::WaitAnyExceptions(uint32_t signal_count, const hsa_signal_t* hs
std::sort(evts, evts + signal_count);
HsaEvent** end = std::unique(evts, evts + signal_count);
unique_evts = uint32_t(end - evts);
#if defined(__linux__)
uint64_t event_age[unique_evts];
#else
auto event_age = reinterpret_cast<uint64_t*>(_alloca(unique_evts * sizeof(unique_evts)));
#endif
memset(event_age, 0, unique_evts * sizeof(uint64_t));
if (core::Runtime::runtime_singleton_->KfdVersion().supports_event_age)
for (uint32_t i = 0; i < unique_evts; i++)
+18 -1
Просмотреть файл
@@ -44,8 +44,17 @@
#include <stdint.h>
#include <algorithm>
#if defined(__linux__)
#include <sys/eventfd.h>
#include <poll.h>
#else
struct pollfd {
int fd;
short int events;
short int revents;
};
#define POLLIN 0x001 // from poll.h...
#endif
#include "core/util/utils.h"
#include "core/inc/runtime.h"
@@ -171,7 +180,12 @@ void SvmProfileControl::PollSmi() {
};
while (!exit) {
#if defined(__linux__)
int ready = poll(&files[0], files.size(), -1);
#else
int ready = 0;
assert(!"Unimplemented!");
#endif
if (ready < 1) {
assert(false && "poll failed!");
return;
@@ -345,9 +359,10 @@ void SvmProfileControl::PollSmi() {
}
SvmProfileControl::SvmProfileControl() : event(-1), exit(false) {
#if defined(__linux__)
event = eventfd(0, EFD_CLOEXEC);
if (event == -1) return;
#endif
poll_smi_thread_ = os::CreateThread(PollSmiRun, (void*)this);
if (poll_smi_thread_ == NULL) {
assert(false && "Poll SMI thread creation error.");
@@ -356,10 +371,12 @@ SvmProfileControl::SvmProfileControl() : event(-1), exit(false) {
}
SvmProfileControl::~SvmProfileControl() {
#if defined(__linux__)
if (event != -1) {
eventfd_write(event, 1);
close(event);
}
#endif
if (poll_smi_thread_ != NULL) {
exit = true;
os::WaitForThread(poll_smi_thread_);
+25 -10
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -43,9 +43,12 @@
#include "core/inc/thunk_loader.h"
#include "core/inc/runtime.h"
#include <dlfcn.h>
#include <core/util/os.h>
#include <iostream>
#if defined(__linux__)
#include <dlfcn.h>
#include <fcntl.h>
#endif
namespace rocr {
namespace core {
@@ -57,6 +60,7 @@ namespace core {
return "libdtif.so";
}
#if defined(__linux__)
if (core::Runtime::runtime_singleton_->flag().enable_dxg_detection()) {
int fd = open("/dev/dxg", O_RDWR);
if (fd >= 0) {
@@ -65,6 +69,9 @@ namespace core {
return "librocdxg.so";
}
}
#else
is_dxg_ = true;
#endif
return "";
}
@@ -74,10 +81,10 @@ namespace core {
library_name(whoami()),
is_loaded_(false) {
if (!library_name.empty()) {
dlerror(); // Clear any existing error messages
thunk_handle = dlopen(library_name.c_str(), RTLD_LAZY);
rocr::os::DlError(); // Clear any existing error messages
thunk_handle = rocr::os::LoadLib(library_name.c_str());
if (thunk_handle == NULL) {
fprintf(stderr, "Cannot load %s, failed:%s\n", library_name.c_str(), dlerror());
fprintf(stderr, "Cannot load %s, failed:%s\n", library_name.c_str(), rocr::os::DlError());
} else {
debug_print("Load %s successully!\n", library_name.c_str());
}
@@ -88,8 +95,8 @@ namespace core {
ThunkLoader::~ThunkLoader() {
if (IsSharedLibraryLoaded()
&& (thunk_handle != NULL)) {
if (dlclose(thunk_handle) != 0) {
fprintf(stderr, "Cannot unload %s, failed:%s\n", library_name.c_str(), dlerror());
if (!rocr::os::CloseLib(thunk_handle)) {
fprintf(stderr, "Cannot unload %s, failed:%s\n", library_name.c_str(), rocr::os::DlError());
} else {
debug_print("Unload %s successully!\n", library_name.c_str());
}
@@ -98,6 +105,7 @@ namespace core {
void ThunkLoader::LoadThunkApiTable() {
if (IsSharedLibraryLoaded()) {
#if defined(__linux__)
dlerror(); // Clear any existing error messages
HSAKMT_PFN(hsaKmtOpenKFD) = (HSAKMT_DEF(hsaKmtOpenKFD)*)dlsym(thunk_handle, "hsaKmtOpenKFD");
@@ -402,12 +410,12 @@ namespace core {
DRM_PFN(drmCommandWriteRead) = (DRM_DEF(drmCommandWriteRead)*)dlsym(thunk_handle, "drmCommandWriteRead");
if (DRM_PFN(drmCommandWriteRead) == NULL) goto ERROR;
debug_print("Load all DTIF APIs OK!\n");
return;
ERROR:
fprintf(stderr, "dlsym failed: %s\n", dlerror());
#endif
} else {
HSAKMT_PFN(hsaKmtOpenKFD) = (HSAKMT_DEF(hsaKmtOpenKFD)*)(&hsaKmtOpenKFD);
HSAKMT_PFN(hsaKmtCloseKFD) = (HSAKMT_DEF(hsaKmtCloseKFD)*)(&hsaKmtCloseKFD);
@@ -499,6 +507,9 @@ ERROR:
HSAKMT_PFN(hsaKmtPcSamplingStart) = (HSAKMT_DEF(hsaKmtPcSamplingStart)*)(&hsaKmtPcSamplingStart);
HSAKMT_PFN(hsaKmtPcSamplingStop) = (HSAKMT_DEF(hsaKmtPcSamplingStop)*)(&hsaKmtPcSamplingStop);
HSAKMT_PFN(hsaKmtPcSamplingSupport) = (HSAKMT_DEF(hsaKmtPcSamplingSupport)*)(&hsaKmtPcSamplingSupport);
#if defined(_WIN32)
HSAKMT_PFN(hsaKmtQueueRingDoorbell) = (HSAKMT_DEF(hsaKmtQueueRingDoorbell)*)(&hsaKmtQueueRingDoorbell);
#endif
HSAKMT_PFN(hsaKmtModelEnabled) = (HSAKMT_DEF(hsaKmtModelEnabled)*)(&hsaKmtModelEnabled);
DRM_PFN(amdgpu_device_initialize) = (DRM_DEF(amdgpu_device_initialize)*)(&amdgpu_device_initialize);
@@ -509,7 +520,9 @@ ERROR:
DRM_PFN(amdgpu_bo_export) = (DRM_DEF(amdgpu_bo_export)*)(&amdgpu_bo_export);
DRM_PFN(amdgpu_bo_import) = (DRM_DEF(amdgpu_bo_import)*)(&amdgpu_bo_import);
DRM_PFN(amdgpu_bo_va_op) = (DRM_DEF(amdgpu_bo_va_op)*)(&amdgpu_bo_va_op);
#if defined(__linux__)
DRM_PFN(drmCommandWriteRead) = (DRM_DEF(drmCommandWriteRead)*)(&drmCommandWriteRead);
#endif
}
}
@@ -517,7 +530,8 @@ ERROR:
if (!IsDTIF())
return true;
DtifCreateFunc* pfnDtifCreate = (DtifCreateFunc*)dlsym(thunk_handle, "DtifCreate");
DtifCreateFunc* pfnDtifCreate =
(DtifCreateFunc*)rocr::os::GetExportAddress(thunk_handle, "DtifCreate");
if (pfnDtifCreate != NULL) {
if (pfnDtifCreate("HSA") != NULL) {
debug_print("DtifCreate OK!\n");
@@ -537,7 +551,8 @@ ERROR:
if (thunk_handle == NULL)
return false;
DtifDestroyFunc* pfnDtifDestroy = (DtifDestroyFunc*)dlsym(thunk_handle, "DtifDestroy");
DtifDestroyFunc* pfnDtifDestroy =
(DtifDestroyFunc*)rocr::os::GetExportAddress(thunk_handle, "DtifDestroy");
if (pfnDtifDestroy != NULL) {
pfnDtifDestroy();
debug_print("DtifDestroy OK!\n");
+12 -1
Просмотреть файл
@@ -3,7 +3,7 @@
## The University of Illinois/NCSA
## Open Source License (NCSA)
##
## Copyright (c) 2022, Advanced Micro Devices, Inc. All rights reserved.
## Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
##
## Developed by:
##
@@ -139,10 +139,21 @@ function(generate_bytecodeStrm HeaderFILE)
## Add a custom command that generates amd_trap_handler_v2.h
## This depends on all the generated code object files and the C++ generator script.
if (UNIX)
add_custom_command(OUTPUT ${HeaderFILE}.h
COMMAND ${CMAKE_CURRENT_SOURCE_DIR}/create_trap_handler_header.sh ${ARG_LIST}
COMMENT "Collating trap handlers..."
DEPENDS ${HSACO_TARG_LIST} create_trap_handler_header.sh )
else()
find_package(Python3 COMPONENTS Interpreter REQUIRED)
add_custom_command(
OUTPUT ${HeaderFILE}.h
COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/create_trap_handler_header.py ${ARG_LIST}
COMMENT "Collating blit shaders..."
DEPENDS ${HSACO_TARG_LIST} create_trap_handler_header.py)
endif()
## Export a target that builds (and depends on) amd_trap_handler_v2.h
add_custom_target( ${HeaderFILE} DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/${HeaderFILE}.h )
@@ -0,0 +1,71 @@
################################################################################
##
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
##
## SPDX-License-Identifier: MIT
##
################################################################################
import sys
def GetSize(fileobject):
fileobject.seek(0,2) # move the cursor to the end of the file
size = fileobject.tell()
return size
def DumpFile(header, input_name):
try:
with open(input_name, "rb") as binary_file:
# Read the entire content of the file as bytes
binary_data = binary_file.read()
file_size = GetSize(binary_file)
#print(f"Binary size: {file_size}")
# Reset file pointer
binary_file.seek(0)
parts = input_name.split('.')
file_name = parts[0]
content = f"unsigned char {file_name}""[] = {\n "
header.write(content)
line = 0
count = 0
for byte_value in binary_data:
count += 1
padded_hex = '{:02x}'.format(byte_value)
if (count != file_size):
header.write(f"0x{padded_hex},")
else:
header.write(f"0x{padded_hex}")
line += 1
if (line == 12):
header.write(f"\n ")
line = 0
else:
header.write(f" ")
header.write("\n};\nunsigned int "f"{file_name}_len = {file_size};\n")
except FileNotFoundError:
print(f"Error: The file {input_name} was not found.")
except Exception as e:
print(f"An error occurred: {e}")
if len(sys.argv) > 1:
header_name = sys.argv[1];
with open(header_name, 'w') as header:
header.write("//==============================================================================\n")
header.write("// This file is automatically generated during build process, don't modify it\n")
header.write("//==============================================================================\n\n")
header.write("namespace rocr {\n")
header.write("namespace AMD {\n\n")
for i, arg in enumerate(sys.argv):
if (i > 1):
#print(f"File {i}: {arg}\n")
DumpFile(header, arg)
header.write("} // namespace AMD\n")
header.write("} // namespace rocr\n\n")
else:
print("Empty arguments!")
+132 -1
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -48,6 +48,133 @@
#ifndef HSA_RUNTIME_CORE_UTIL_ATOMIC_HELPERS_H_
#define HSA_RUNTIME_CORE_UTIL_ATOMIC_HELPERS_H_
#if defined(_WIN32)
#define WIN32_NO_STATUS
#include <Windows.h>
#undef WIN32_NO_STATUS
template <class T>
void __atomic_load(const T* object, typename std::remove_volatile<T>::type* ret, int arg) {
if constexpr (sizeof(T) == 8) {
*ret = InterlockedOr64(
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
0);
} else {
*ret = InterlockedOr(
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
0);
}
}
template <class T>
void __atomic_store(const T* object, typename std::remove_volatile<T>::type* val, int arg) {
if constexpr (sizeof(T) == 8) {
InterlockedExchange64(
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
*val);
} else {
InterlockedExchange(
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
*val);
}
}
template <class T>
typename std::remove_volatile<T>::type __atomic_fetch_or(
const T* object, typename std::remove_volatile<T>::type val, int arg) {
if constexpr (sizeof(T) == 8) {
return InterlockedOr64(
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
val);
} else {
return InterlockedOr(
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
val);
}
}
template <class T>
typename std::remove_volatile<T>::type __atomic_fetch_and(
const T* object, typename std::remove_volatile<T>::type val, int arg) {
if constexpr (sizeof(T) == 8) {
return InterlockedAnd64(
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
val);
} else {
return InterlockedAnd(
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
val);
}
}
template <class T>
typename std::remove_volatile<T>::type __atomic_fetch_xor(
const T* object, typename std::remove_volatile<T>::type val, int arg) {
if constexpr (sizeof(T) == 8) {
return InterlockedXor64(
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
val);
} else {
return InterlockedXor(
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
val);
}
}
template <class T>
typename std::remove_volatile<T>::type __atomic_fetch_add(
const T* object, typename std::remove_volatile<T>::type val, int arg) {
if constexpr (sizeof(T) == 8) {
return InterlockedExchangeAdd64(
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
val);
} else {
return InterlockedExchangeAdd(
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
val);
}
}
template <class T>
typename std::remove_volatile<T>::type __atomic_fetch_sub(
const T* object, typename std::remove_volatile<T>::type val, int arg) {
if constexpr (sizeof(T) == 8) {
return InterlockedExchangeAdd64(
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
val * (-1));
} else {
return InterlockedExchangeAdd(
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
val * (-1));
}
}
template <class T>
void __atomic_compare_exchange(
T* object, typename std::remove_volatile<T>::type* expected,
typename std::remove_volatile<T>::type* val, int arg0, int arg1, int arg2) {
if constexpr (sizeof(T) == 8) {
InterlockedCompareExchange64(reinterpret_cast<volatile LONG64*>(object),
*val, *expected);
} else {
InterlockedCompareExchange(reinterpret_cast<volatile LONG*>(object),
*val, *expected);
}
}
template <class T>
void __atomic_exchange(T* object, typename std::remove_volatile<T>::type* val,
typename std::remove_volatile<T>::type* ret, int arg0) {
if constexpr (sizeof(T) == 8) {
*ret = InterlockedExchange64(reinterpret_cast<volatile LONG64*>(object), *val);
} else {
*ret = InterlockedExchange(reinterpret_cast<volatile LONG*>(object), *val);
}
}
#define __ATOMIC_RELAXED 0
#endif
#include <atomic>
//ALWAYS_CONSERVATIVE will very likely overfence your code.
@@ -145,14 +272,18 @@ static __forceinline void Fence(std::memory_order order=std::memory_order_seq_cs
template <class T>
static __forceinline void BasicCheck(const T* ptr) {
#if defined(__linux__)
constexpr bool value = __atomic_always_lock_free(sizeof(T), 0);
static_assert(value, "Atomic type may not be compatible with peripheral atomics.");
#endif
};
template <class T>
static __forceinline void BasicCheck(const volatile T* ptr) {
#if defined(__linux__)
constexpr bool value = __atomic_always_lock_free(sizeof(T), 0);
static_assert(value, "Atomic type may not be compatible with peripheral atomics.");
#endif
};
/// @brief: Load value of type T atomically with specified memory order.
+122 -3
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -61,6 +61,7 @@
#include <utility>
#include <semaphore.h>
#include "core/inc/runtime.h"
#include <sys/mman.h>
#if defined(__i386__) || defined(__x86_64__)
#include <cpuid.h>
#endif
@@ -294,7 +295,7 @@ void* GetExportAddress(LibHandle lib, std::string export_name) {
return NULL;
}
void CloseLib(LibHandle lib) { dlclose(*(void**)&lib); }
bool CloseLib(LibHandle lib) { return (dlclose(*(void**)&lib) == 0) ? true : false; }
/*
* @brief Look for a symbol called "HSA_AMD_TOOL_PRIORITY" across all loaded
@@ -579,7 +580,7 @@ int WaitForOsEvent(EventHandle event, unsigned int milli_seconds) {
}
int ret_code = 0;
if (!eventDescrp->state) {
if (milli_seconds == 0) {
ret_code = 1;
@@ -816,6 +817,124 @@ bool ParseCpuID(cpuid_t* cpuinfo) {
#endif
}
uint64_t TimeNanos() {
struct timespec tp;
::clock_gettime(CLOCK_MONOTONIC, &tp);
return (uint64_t)tp.tv_sec * (1000ULL * 1000ULL * 1000ULL) + (uint64_t)tp.tv_nsec;
}
static inline int MemProtToOsProt(MemProt prot) {
switch (prot) {
case MEM_PROT_NONE:
return PROT_NONE;
case MEM_PROT_READ:
return PROT_READ;
case MEM_PROT_RW:
return PROT_READ | PROT_WRITE;
case MEM_PROT_RWX:
return PROT_READ | PROT_WRITE | PROT_EXEC;
default:
break;
}
return -1;
}
size_t PageSize() {
static size_t g_page_size_ = 0; //!< The default os page size
if (g_page_size_ == 0) {
g_page_size_ = (size_t)::sysconf(_SC_PAGESIZE);
}
return g_page_size_;
}
void* ReserveMemory(void* start, size_t size, size_t alignment, MemProt prot) {
size = AlignUp(size, PageSize());
// check for invalid input size
if (size == 0) {
return NULL;
}
alignment = std::max(PageSize(), AlignUp(alignment, PageSize()));
assert(IsPowerOfTwo(alignment) && "not a power of 2");
size_t requested = size + alignment - PageSize();
address mem = (address)::mmap(start, requested, MemProtToOsProt(prot),
MAP_PRIVATE | MAP_NORESERVE | MAP_ANONYMOUS, 0, 0);
// check for out of memory
if (mem == MAP_FAILED) return NULL;
address aligned = AlignUp(mem, alignment);
// return the unused leading pages to the free state
if (&aligned[0] != &mem[0]) {
assert(&aligned[0] > &mem[0] && "check this code");
if (::munmap(&mem[0], &aligned[0] - &mem[0]) != 0) {
assert(!"::munmap failed");
}
}
// return the unused trailing pages to the free state
if (&aligned[size] != &mem[requested]) {
assert(&aligned[size] < &mem[requested] && "check this code");
if (::munmap(&aligned[size], &mem[requested] - &aligned[size]) != 0) {
assert(!"::munmap failed");
}
}
// Hint to enable THP for large host allocations which can help in performance gain
constexpr size_t kLargePageSize = 2 * 1024 * 1024;
if (size >= kLargePageSize) {
int status = madvise(aligned, size, MADV_HUGEPAGE);
if (status) {
LogPrint(HSA_AMD_LOG_FLAG_INFO,
"madvise with advice MADV_HUGEPAGE"
" starting at address %p and page size 0x%zx, returned %d, errno: %s",
aligned, size, status, strerror(errno));
}
}
return aligned;
}
bool ReleaseMemory(void* addr, size_t size) {
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
size = AlignUp(size, PageSize());
return 0 == ::munmap(addr, size);
}
bool CommitMemory(void* addr, size_t size, MemProt prot) {
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
size = AlignUp(size, PageSize());
return ::mmap(addr, size, MemProtToOsProt(prot), MAP_PRIVATE | MAP_FIXED | MAP_ANONYMOUS, -1,
0) != MAP_FAILED;
}
bool UncommitMemory(void* addr, size_t size) {
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
size = AlignUp(size, PageSize());
return ::mmap(addr, size, PROT_NONE, MAP_PRIVATE | MAP_FIXED | MAP_NORESERVE | MAP_ANONYMOUS, -1,
0) != MAP_FAILED;
}
uint64_t HostTotalPhysicalMemory() {
static uint64_t totalPhys = 0;
if (totalPhys != 0) {
return totalPhys;
}
totalPhys = sysconf(_SC_PAGESIZE) * sysconf(_SC_PHYS_PAGES);
return totalPhys;
}
int Ffs(int i) { return ffs(i); }
int Ctz(uint64_t i) { return __builtin_ctz(i); }
char* DlError() { return dlerror(); }
} // namespace os
} // namespace rocr
+39 -4
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -91,7 +91,7 @@ void* GetExportAddress(LibHandle lib, std::string export_name);
/// @brief: Unloads the dynamic library.
/// @param: lib(Input), library handle which will be unloaded.
void CloseLib(LibHandle lib);
bool CloseLib(LibHandle lib);
/// @brief: Lists loaded tool libraries that contain
/// symbol HSA_AMD_TOOL_PRIORITY
@@ -106,6 +106,7 @@ std::string GetLibraryName(LibHandle lib);
/// @brief: Creates a Semaphore, will return NULL if failed.
/// @param: void.
/// @return: Semaphore.
#undef CreateSemaphore
Semaphore CreateSemaphore();
/// @brief: Waits for the semaphore. This is a blocking wait.
@@ -127,6 +128,7 @@ void DestroySemaphore(Semaphore sem);
/// @brief: Creates a mutex, will return NULL if failed.
/// @param: void.
/// @return: Mutex.
#undef CreateMutex
Mutex CreateMutex();
/// @brief: Tries to acquire the mutex once, if successed, return true.
@@ -319,15 +321,48 @@ uint64_t ReadSystemClock();
/// @brief read the system clock frequency
uint64_t SystemClockFrequency();
typedef struct cpuid_s {
struct cpuid_t {
char ManufacturerID[13]; // 12 char, NULL terminated
bool mwaitx;
} cpuid_t;
};
/// @brief parse CPUID
/// @param: cpuinfo struct to be filled
bool ParseCpuID(cpuid_t* cpuinfo);
//! Return the default os page size.
size_t PageSize();
/// @brief CPU time in nanoseconds
/// @param: None
uint64_t TimeNanos();
using address = char*;
enum MemProt { MEM_PROT_NONE = 0, MEM_PROT_READ, MEM_PROT_RW, MEM_PROT_RWX };
/// @brief Reserves a chunk of memory (priv | anon | noreserve)
/// @param:
void* ReserveMemory(void* start, size_t size, size_t alignment = 0,
MemProt prot = MEM_PROT_NONE);
/// Release a chunk of memory reserved with reserveMemory.
bool ReleaseMemory(void* addr, size_t size);
/// Commit a chunk of memory previously reserved with reserveMemory.
bool CommitMemory(void* addr, size_t size, MemProt prot = MEM_PROT_NONE);
/// Uncommit a chunk of memory previously committed with commitMemory.
bool UncommitMemory(void* addr, size_t size);
uint64_t HostTotalPhysicalMemory();
/// Find First Set for any OS
int Ffs(int i);
/// Find the count of leading zeros
int Ctz(uint64_t i);
/// Shared library or DLL load error
char* DlError();
} // namespace os
} // namespace rocr
+30 -1
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -196,6 +196,35 @@ template <typename Allocator> class SimpleHeap {
return reinterpret_cast<void*>(base);
}
/* Return block-base the ptr belongs to if the ptr is a valid ptr which is allocated
* from this simpleheap and the block-base is allocated from block_allocator_*/
void* block_base(void* ptr) {
if (ptr == nullptr)
return nullptr;
uintptr_t base = reinterpret_cast<uintptr_t>(ptr);
// Find fragment and validate.
auto frag_map_it = block_list_.upper_bound(base);
if (frag_map_it == block_list_.begin())
return nullptr;
frag_map_it--;
auto& frag_map = frag_map_it->second;
auto fragment = frag_map.find(base);
if (fragment == frag_map.end() || isFree(fragment->second))
return nullptr;
return reinterpret_cast<void*>(frag_map_it->first);
}
void reset() {
free_list_.clear();
block_list_.clear();
block_cache_.clear();
in_use_size_ = 0;
cache_size_ = 0;
}
bool free(void* ptr) {
if (ptr == nullptr) return true;
+48 -6
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -49,13 +49,16 @@
#include "stddef.h"
#include "stdlib.h"
#include "stdarg.h"
#if defined(__linux__)
#include "unistd.h"
#endif
#include <assert.h>
#include <iostream>
#include <string>
#include <algorithm>
#include <sstream>
#include <thread>
#include <locale>
namespace rocr {
extern FILE* log_file;
@@ -64,6 +67,14 @@ extern uint8_t log_flags[8];
typedef unsigned int uint;
typedef uint64_t uint64;
// 2MB huge page size
#define GPU_HUGE_PAGE_SIZE (2 << 20)
// 4KB page size
#define DEFAULT_GPU_PAGE_SIZE (1 << 12)
void log_printf(const char* file, int line, const char* format, ...);
#if defined(__GNUC__)
#if defined(__i386__) || defined(__x86_64__)
#include <x86intrin.h>
@@ -75,8 +86,6 @@ typedef uint64_t uint64;
#define __stdcall // __attribute__((__stdcall__))
#define __ALIGNED__(x) __attribute__((aligned(x)))
void log_printf(const char* file, int line, const char* format, ...);
static __forceinline void* _aligned_malloc(size_t size, size_t alignment) {
#ifdef _ISOC11_SOURCE
return aligned_alloc(alignment, size);
@@ -114,6 +123,7 @@ static __forceinline unsigned long long int strtoull(const char* str,
do { \
} while (false)
#else
#if defined(__linux__)
#define debug_warning_n(exp, limit) \
do { \
static std::atomic<int> count(0); \
@@ -123,6 +133,18 @@ static __forceinline unsigned long long int strtoull(const char* str,
count++; \
} \
} while (false)
#else
#define debug_warning_n(exp, limit) \
do { \
static std::atomic<int> count(0); \
if (!(exp) && (limit == 0 || count < limit)) { \
fprintf(stderr, "Warning: " STRING(exp) " in %s, " __FILE__ ":" STRING(__LINE__) "\n" \
); \
count++; \
} \
} while (false)
#endif
#endif
#define debug_warning(exp) debug_warning_n((exp), 0)
@@ -369,10 +391,15 @@ inline void FlushCpuCache(const void* base, size_t offset, size_t len) {
static long cacheline_size = 0;
if (!cacheline_size) {
#ifdef _SC_LEVEL1_DCACHE_LINESIZE
long sz = sysconf(_SC_LEVEL1_DCACHE_LINESIZE);
long sz = 64;
#if defined(__linux__)
#ifdef _SC_LEVEL1_DCACHE_LINESIZE
sz = sysconf(_SC_LEVEL1_DCACHE_LINESIZE);
#else
sz = 0;
#endif
#else
long sz = 0;
//@todo abstract GetLogicalProcessorInformation call
#endif
if (sz <= 0) return;
cacheline_size = sz;
@@ -421,6 +448,17 @@ inline uint32_t PtrHigh64Shift40(const void* p) {
return (uint32_t)((ptr & 0xFFFFFF0000000000ULL) >> 40);
}
static inline uint8_t Ptr48High8(const void* p) {
uintptr_t ptr = reinterpret_cast<uintptr_t>(p);
return (uint8_t)((ptr & 0xFF0000000000ULL) >> 40);
}
static inline uint32_t Ptr48Low32(const void* p) {
uintptr_t ptr = reinterpret_cast<uintptr_t>(p);
assert((ptr & 0xFFFFFFFFFF00ULL) == ptr);
return (uint32_t)((ptr & 0xFFFFFFFFFFULL) >> 8);
}
inline uint32_t PtrLow32(const void* p) {
return static_cast<uint32_t>(reinterpret_cast<uintptr_t>(p));
}
@@ -433,6 +471,10 @@ inline uint32_t PtrHigh32(const void* p) {
return ptr;
}
inline uint32_t HighPart(uint64_t value) { return (value & 0xFFFFFFFF00000000) >> 32; }
inline uint32_t LowPart(uint64_t value) { return (value & 0x00000000FFFFFFFF); }
/// @brief: Concatenates two numbers of type InType to a number of type OutType
/// @param: hi(Input), To be placed in the upper bits of the output
/// @param: lo(Input), To be placed in the lower bits of the output
+196 -39
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -41,7 +41,6 @@
////////////////////////////////////////////////////////////////////////////////
#ifdef _WIN32 // Are we compiling for windows?
#define NOMINMAX
#include "core/util/os.h"
@@ -49,10 +48,13 @@
#include <process.h>
#include <string>
#include <windows.h>
#include <ntstatus.h>
#include <psapi.h>
#include <emmintrin.h>
#include <pmmintrin.h>
#include <xmmintrin.h>
#include <shared_mutex>
#undef Yield
#undef CreateMutex
@@ -82,37 +84,37 @@ void* GetExportAddress(LibHandle lib, std::string export_name) {
return GetProcAddress(*(HMODULE*)&lib, export_name.c_str());
}
void CloseLib(LibHandle lib) { FreeLibrary(*(::HMODULE*)&lib); }
bool CloseLib(LibHandle lib) { return FreeLibrary(*(::HMODULE*)&lib); }
std::vector<LibHandle> GetLoadedLibs() {
// Use EnumProcessModulesEx
static_assert(false, "Not implemented.");
assert(!"Not implemented.");
return std::vector<LibHandle>{};
}
std::string GetLibraryName(LibHandle lib) {
static_assert(false, "Not implemented.");
assert(!"Not implemented.");
return std::string{};
}
Semaphore CreateSemaphore() {
sem = static_cast<void*>(CreateSemaphore(NULL, 0, LONG_MAX, NULL));
assert(sem != NULL && "CreateSemaphore failed");
auto sem = static_cast<void*>(CreateSemaphoreA(nullptr, 0, LONG_MAX, nullptr));
assert(sem != nullptr && "CreateSemaphore failed");
return *(Semaphore*)&sem;
}
bool WaitSemaphore(Semaphore sem) {
return WaitForSingleObject(*(::HANDLE*)&lock, INFINITE) == WAIT_OBJECT_0;
return WaitForSingleObject(sem, INFINITE) == WAIT_OBJECT_0;
}
void PostSemaphore(Semaphore sem) {
ReleaseSemaphore(static_cast<HANDLE>(*sem), 1, NULL);
ReleaseSemaphore(sem, 1, nullptr);
}
void DestroySemaphore(Semaphore sem) {
if (!CloseHandle(static_cast<HANDLE>(*sem))) {
if (!CloseHandle(sem)) {
assert("CloseHandle() failed");
}
*sem = NULL;
}
Mutex CreateMutex() { return CreateEvent(NULL, false, true, NULL); }
@@ -259,48 +261,37 @@ uint64_t AccurateClockFrequency() {
}
SharedMutex CreateSharedMutex() {
assert(false && "Not implemented.");
abort();
return nullptr;
return reinterpret_cast<SharedMutex>(new std::shared_mutex());
}
bool TryAcquireSharedMutex(SharedMutex lock) {
assert(false && "Not implemented.");
abort();
return false;
return reinterpret_cast<std::shared_mutex*>(lock)->try_lock();
}
bool AcquireSharedMutex(SharedMutex lock) {
assert(false && "Not implemented.");
abort();
return false;
reinterpret_cast<std::shared_mutex*>(lock)->lock();
return true;
}
void ReleaseSharedMutex(SharedMutex lock) {
assert(false && "Not implemented.");
abort();
reinterpret_cast<std::shared_mutex*>(lock)->unlock();
}
bool TrySharedAcquireSharedMutex(SharedMutex lock) {
assert(false && "Not implemented.");
abort();
return false;
return reinterpret_cast<std::shared_mutex*>(lock)->try_lock_shared();
}
bool SharedAcquireSharedMutex(SharedMutex lock) {
assert(false && "Not implemented.");
abort();
return false;
reinterpret_cast<std::shared_mutex*>(lock)->lock_shared();
return true;
}
void SharedReleaseSharedMutex(SharedMutex lock) {
assert(false && "Not implemented.");
abort();
reinterpret_cast<std::shared_mutex*>(lock)->unlock_shared();
}
void DestroySharedMutex(SharedMutex lock) {
assert(false && "Not implemented.");
abort();
delete reinterpret_cast<std::shared_mutex*>(lock);
}
uint64_t ReadSystemClock() {
@@ -310,17 +301,183 @@ uint64_t ReadSystemClock() {
}
uint64_t SystemClockFrequency() {
assert(false && "Not implemented.");
abort();
return 0;
LARGE_INTEGER frequency;
QueryPerformanceFrequency(&frequency);
return frequency.QuadPart;
}
bool ParseCpuID(cpuid_t* cpuinfo) {
assert(false && "Not implemented.");
abort();
return false;
int regs[4] = {};
int info{};
__cpuid(regs, info);
memset(cpuinfo->ManufacturerID, 0, sizeof(cpuinfo->ManufacturerID));
*reinterpret_cast<int*>(cpuinfo->ManufacturerID) = regs[1];
*reinterpret_cast<int*>(cpuinfo->ManufacturerID + 4) = regs[3];
*reinterpret_cast<int*>(cpuinfo->ManufacturerID + 8) = regs[2];
// @todo fill the rest of CPU info
return true;
}
bool IsEnvVarSet(std::string env_var_name) {
char* buff = NULL;
buff = getenv(env_var_name.c_str());
return (buff != NULL);
}
std::vector<LibHandle> GetLoadedToolsLib() {
std::vector<LibHandle> ret;
std::vector<std::string> names;
HMODULE hMods[1024];
HANDLE hProcess = GetCurrentProcess();
DWORD cbNeeded;
unsigned int i;
if (EnumProcessModules(hProcess, hMods, sizeof(hMods), &cbNeeded)) {
for (i = 0; i < (cbNeeded / sizeof(HMODULE)); i++) {
TCHAR szModName[MAX_PATH];
// Get the full path to the module's file.
if (GetModuleFileNameEx(hProcess, hMods[i], szModName, sizeof(szModName) / sizeof(TCHAR))) {
// Print the module name and handle value.
names.push_back(szModName);
}
}
}
if (!names.empty()) {
for (auto& name : names) ret.push_back(LoadLib(name));
}
return ret;
}
int GetProcessId() { return ::_getpid(); }
uint64_t TimeNanos() {
static double PerformanceFrequency = 0.f;
if (PerformanceFrequency == 0) {
LARGE_INTEGER frequency;
QueryPerformanceFrequency(&frequency);
PerformanceFrequency = (double)frequency.QuadPart;
}
LARGE_INTEGER current;
QueryPerformanceCounter(&current);
return (uint64_t)((double)current.QuadPart / PerformanceFrequency * 1e9);
}
static inline int memProtToOsProt(MemProt prot) {
switch (prot) {
case MEM_PROT_NONE:
return PAGE_NOACCESS;
case MEM_PROT_READ:
return PAGE_READONLY;
case MEM_PROT_RW:
return PAGE_READWRITE;
case MEM_PROT_RWX:
return PAGE_EXECUTE_READWRITE;
default:
break;
}
return -1;
}
static size_t g_page_size_ = 0; //!< The default os page size
static int processorCount_; //!< The number of active processors
static size_t allocationGranularity_;
//! Return the default os page size.
size_t PageSize() {
if (g_page_size_ == 0) {
SYSTEM_INFO si{};
::GetSystemInfo(&si);
g_page_size_ = si.dwPageSize;
}
return g_page_size_;
}
void* ReserveMemory(void* start, size_t size, size_t alignment, MemProt prot) {
size = AlignUp(size, PageSize());
if (allocationGranularity_ == 0) {
SYSTEM_INFO si;
::GetSystemInfo(&si);
g_page_size_ = si.dwPageSize;
allocationGranularity_ = (size_t)si.dwAllocationGranularity;
}
alignment = std::max(allocationGranularity_, AlignUp(alignment, allocationGranularity_));
assert(IsPowerOfTwo(alignment) && "not a power of 2");
size_t requested = size + alignment - allocationGranularity_;
address mem, aligned;
do {
mem = reinterpret_cast<address>(VirtualAlloc(start, requested, MEM_RESERVE, memProtToOsProt(prot)));
// check for out of memory.
if (mem == NULL) return NULL;
aligned = AlignUp(mem, alignment);
// check for already aligned memory.
if (aligned == mem && size == requested) {
return mem;
}
// try to reserve the aligned address.
if (VirtualFree(mem, 0, MEM_RELEASE) == 0) {
assert(!"VirtualFree failed");
}
mem = (address)VirtualAlloc(aligned, size, MEM_RESERVE, memProtToOsProt(prot));
assert((mem == NULL || mem == aligned) && "VirtualAlloc failed");
} while (mem != aligned);
return mem;
}
bool ReleaseMemory(void* addr, size_t size) { return VirtualFree(addr, 0, MEM_RELEASE) != 0; }
bool CommitMemory(void* addr, size_t size, MemProt prot) {
return VirtualAlloc(addr, size, MEM_COMMIT, memProtToOsProt(prot)) != NULL;
}
bool UncommitMemory(void* addr, size_t size) { return VirtualFree(addr, size, MEM_DECOMMIT) != 0; }
uint64_t HostTotalPhysicalMemory() {
static uint64_t totalPhys = 0;
if (totalPhys != 0) {
return totalPhys;
}
MEMORYSTATUSEX mstatus;
mstatus.dwLength = sizeof(mstatus);
::GlobalMemoryStatusEx(&mstatus);
totalPhys = mstatus.ullTotalPhys;
return totalPhys;
}
int Ffs(int i) {
int res = 0;
unsigned long index;
if (_BitScanForward(&index, i) != 0) {
res = index + 1;
}
return res;
}
int Ctz(uint64_t i) {
unsigned long index;
if (_BitScanReverse64(&index, i)) {
return sizeof(i) * 8 - 1 - index;
} else {
return sizeof(i) * 8;
}
}
char* DlError() { return nullptr; }
} // namespace os
} // namespace rocr
+3 -2
Просмотреть файл
@@ -51,8 +51,9 @@ if( NOT _is_hsa_runtime_dynamic )
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} "${CMAKE_CURRENT_LIST_DIR}")
find_dependency(hsakmt 1.0)
find_dependency(LibElf)
if (UNIX)
find_dependency(LibElf)
endif()
endif()
include( "${CMAKE_CURRENT_LIST_DIR}/@CORE_RUNTIME_NAME@Targets.cmake" )
+226
Просмотреть файл
@@ -0,0 +1,226 @@
EXPORTS
hsa_init
hsa_shut_down
hsa_system_get_info
hsa_extension_get_name
hsa_system_extension_supported
hsa_system_major_extension_supported
hsa_system_get_extension_table
hsa_system_get_major_extension_table
hsa_iterate_agents
hsa_agent_get_info
hsa_agent_get_exception_policies
hsa_cache_get_info
hsa_agent_iterate_caches
hsa_agent_extension_supported
hsa_agent_major_extension_supported
hsa_queue_create
hsa_soft_queue_create
hsa_queue_destroy
hsa_queue_inactivate
hsa_queue_load_read_index_scacquire
hsa_queue_load_read_index_relaxed
hsa_queue_load_write_index_scacquire
hsa_queue_load_write_index_relaxed
hsa_queue_store_write_index_relaxed
hsa_queue_store_write_index_screlease
hsa_queue_cas_write_index_scacq_screl
hsa_queue_cas_write_index_scacquire
hsa_queue_cas_write_index_relaxed
hsa_queue_cas_write_index_screlease
hsa_queue_add_write_index_scacq_screl
hsa_queue_add_write_index_scacquire
hsa_queue_add_write_index_relaxed
hsa_queue_add_write_index_screlease
hsa_queue_store_read_index_relaxed
hsa_queue_store_read_index_screlease
hsa_agent_iterate_regions
hsa_region_get_info
hsa_memory_register
hsa_memory_deregister
hsa_memory_allocate
hsa_memory_free
hsa_memory_copy
hsa_memory_assign_agent
hsa_signal_create
hsa_signal_destroy
hsa_signal_load_relaxed
hsa_signal_load_scacquire
hsa_signal_store_relaxed
hsa_signal_store_screlease
hsa_signal_silent_store_relaxed
hsa_signal_silent_store_screlease
hsa_signal_wait_relaxed
hsa_signal_wait_scacquire
hsa_signal_group_create
hsa_signal_group_destroy
hsa_signal_group_wait_any_scacquire
hsa_signal_group_wait_any_relaxed
hsa_signal_and_relaxed
hsa_signal_and_scacquire
hsa_signal_and_screlease
hsa_signal_and_scacq_screl
hsa_signal_or_relaxed
hsa_signal_or_scacquire
hsa_signal_or_screlease
hsa_signal_or_scacq_screl
hsa_signal_xor_relaxed
hsa_signal_xor_scacquire
hsa_signal_xor_screlease
hsa_signal_xor_scacq_screl
hsa_signal_exchange_relaxed
hsa_signal_exchange_scacquire
hsa_signal_exchange_screlease
hsa_signal_exchange_scacq_screl
hsa_signal_add_relaxed
hsa_signal_add_scacquire
hsa_signal_add_screlease
hsa_signal_add_scacq_screl
hsa_signal_subtract_relaxed
hsa_signal_subtract_scacquire
hsa_signal_subtract_screlease
hsa_signal_subtract_scacq_screl
hsa_signal_cas_relaxed
hsa_signal_cas_scacquire
hsa_signal_cas_screlease
hsa_signal_cas_scacq_screl
hsa_isa_from_name
hsa_agent_iterate_isas
hsa_isa_get_info
hsa_isa_get_info_alt
hsa_isa_get_exception_policies
hsa_isa_get_round_method
hsa_wavefront_get_info
hsa_isa_iterate_wavefronts
hsa_isa_compatible
hsa_code_object_serialize
hsa_code_object_deserialize
hsa_code_object_destroy
hsa_code_object_get_info
hsa_code_object_get_symbol
hsa_code_object_get_symbol_from_name
hsa_code_symbol_get_info
hsa_code_object_iterate_symbols
hsa_code_object_reader_create_from_file
hsa_code_object_reader_create_from_memory
hsa_code_object_reader_destroy
hsa_executable_create
hsa_executable_create_alt
hsa_executable_destroy
hsa_executable_load_code_object
hsa_executable_load_program_code_object
hsa_executable_load_agent_code_object
hsa_executable_freeze
hsa_executable_get_info
hsa_executable_global_variable_define
hsa_executable_agent_global_variable_define
hsa_executable_readonly_variable_define
hsa_executable_validate
hsa_executable_validate_alt
hsa_executable_get_symbol
hsa_executable_get_symbol_by_name
hsa_executable_symbol_get_info
hsa_executable_iterate_symbols
hsa_executable_iterate_agent_symbols
hsa_executable_iterate_program_symbols
hsa_status_string
hsa_ext_program_create
hsa_ext_program_destroy
hsa_ext_program_add_module
hsa_ext_program_iterate_modules
hsa_ext_program_get_info
hsa_ext_program_finalize
hsa_amd_coherency_get_type
hsa_amd_coherency_set_type
hsa_amd_profiling_set_profiler_enabled
hsa_amd_profiling_get_dispatch_time
hsa_amd_profiling_async_copy_enable
hsa_amd_profiling_get_async_copy_time
hsa_amd_profiling_convert_tick_to_system_domain
hsa_amd_signal_create
hsa_amd_signal_wait_any
hsa_amd_signal_async_handler
hsa_amd_async_function
hsa_amd_image_get_info_max_dim
hsa_amd_queue_cu_set_mask
hsa_amd_queue_cu_get_mask
hsa_amd_memory_fill
hsa_amd_memory_async_copy
hsa_amd_memory_async_copy_on_engine
hsa_amd_memory_copy_engine_status
hsa_amd_memory_get_preferred_copy_engine
hsa_amd_memory_async_copy_rect
hsa_amd_memory_lock
hsa_amd_memory_lock_to_pool
hsa_amd_memory_unlock
hsa_amd_agent_iterate_memory_pools
hsa_amd_agent_memory_pool_get_info
hsa_amd_agents_allow_access
hsa_amd_memory_pool_get_info
hsa_amd_memory_pool_allocate
hsa_amd_memory_pool_free
hsa_amd_memory_pool_can_migrate
hsa_amd_memory_migrate
hsa_amd_interop_map_buffer
hsa_amd_interop_unmap_buffer
hsa_amd_image_create
hsa_ext_image_get_capability
hsa_ext_image_data_get_info
hsa_ext_image_create
hsa_ext_image_import
hsa_ext_image_export
hsa_ext_image_copy
hsa_ext_image_clear
hsa_ext_image_destroy
hsa_ext_sampler_create
hsa_ext_sampler_create_v2
hsa_ext_sampler_destroy
hsa_ext_image_get_capability_with_layout
hsa_ext_image_data_get_info_with_layout
hsa_ext_image_create_with_layout
hsa_amd_pointer_info
hsa_amd_pointer_info_set_userdata
hsa_amd_ipc_memory_create
hsa_amd_ipc_memory_attach
hsa_amd_ipc_memory_detach
hsa_amd_ipc_signal_create
hsa_amd_ipc_signal_attach
hsa_amd_register_system_event_handler
hsa_amd_queue_set_priority
hsa_amd_register_deallocation_callback
hsa_amd_deregister_deallocation_callback
hsa_amd_signal_value_pointer
_amdgpu_r_debug
hsa_amd_svm_attributes_set
hsa_amd_svm_attributes_get
hsa_amd_svm_prefetch_async
hsa_amd_spm_acquire
hsa_amd_spm_release
hsa_amd_spm_set_dest_buffer
hsa_amd_portable_export_dmabuf
hsa_amd_portable_close_dmabuf
hsa_amd_vmem_address_reserve
hsa_amd_vmem_address_reserve_align
hsa_amd_vmem_address_free
hsa_amd_vmem_handle_create
hsa_amd_vmem_handle_release
hsa_amd_vmem_map
hsa_amd_vmem_unmap
hsa_amd_vmem_set_access
hsa_amd_vmem_get_access
hsa_amd_vmem_export_shareable_handle
hsa_amd_vmem_import_shareable_handle
hsa_amd_vmem_retain_alloc_handle
hsa_amd_vmem_get_alloc_properties_from_handle
hsa_amd_agent_set_async_scratch_limit
hsa_ven_amd_pcs_iterate_configuration
hsa_ven_amd_pcs_create
hsa_ven_amd_pcs_create_from_id
hsa_ven_amd_pcs_destroy
hsa_ven_amd_pcs_start
hsa_ven_amd_pcs_stop
hsa_ven_amd_pcs_flush
hsa_amd_queue_get_info
hsa_amd_enable_logging
hsa_amd_signal_wait_all
hsa_amd_portable_export_dmabuf_v2
+23 -6
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2023, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2023-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -40,10 +40,16 @@
//
////////////////////////////////////////////////////////////////////////////////
#if defined(__linux__)
#include <unistd.h>
#include <elf.h>
#include <fcntl.h>
#include <sys/resource.h>
#include <elf.h>
#else
#include <cstdint>
#include <stdio.h>
#include <win32/elf.h>
#endif
#include <fcntl.h>
#include <cstring>
#include <vector>
#include <sstream>
@@ -270,11 +276,14 @@ struct LoadSegmentBuilder : public SegmentBuilder {
if (fd_ == -1) return HSA_STATUS_ERROR;
size_t done = 0;
ssize_t read;
size_t read;
do {
#if defined(__linux__)
read = pread(fd_, static_cast<char *>(buf) + done, buf_size - done,
offset + done);
#else
assert(!"Unimplemented!");
#endif
if (read == -1 && errno != EINTR) {
perror("Failed to read GPU memory");
return HSA_STATUS_ERROR;
@@ -305,6 +314,7 @@ hsa_status_t build_core_dump(const std::string& filename, const SegmentsInfo& se
debug_print("Core file size over limit\n");
return HSA_STATUS_SUCCESS;
}
#if defined(__linux__)
int fd = open(filename.c_str(), O_WRONLY | O_CREAT | O_EXCL, S_IRUSR | S_IWUSR);
if (fd == -1) {
perror("Failed to create GPU coredump");
@@ -423,6 +433,9 @@ hsa_status_t build_core_dump(const std::string& filename, const SegmentsInfo& se
}
printf("GPU core dump created: %s\n", filename.c_str());
close(fd);
#else
assert(!"Unimplemented!");
#endif
return HSA_STATUS_SUCCESS;
}
} // namespace impl
@@ -431,7 +444,7 @@ hsa_status_t dump_gpu_core() {
impl::NoteSegmentBuilder nbuilder;
impl::LoadSegmentBuilder lbuilder;
impl::SegmentsInfo segments;
#if defined(__linux__)
struct rlimit rlimit;
if (getrlimit(RLIMIT_CORE, &rlimit)) {
@@ -452,6 +465,10 @@ hsa_status_t dump_gpu_core() {
std::stringstream st;
st << PREFIX_FILE_NAME << "." << getpid();
return build_core_dump(st.str(), segments, rlimit.rlim_cur);
#else
assert(!"Unimplemented!");
return HSA_STATUS_SUCCESS;
#endif
}
} // namespace coredump
} // namespace amd
Разница между файлами не показана из-за своего большого размера Загрузить разницу
+13 -4
Просмотреть файл
@@ -3,7 +3,7 @@
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
@@ -40,12 +40,14 @@
//
////////////////////////////////////////////////////////////////////////////////
#include "executable.hpp"
#include <libelf.h>
#include <limits.h>
#if defined(__linux__)
#include <link.h>
#include <unistd.h>
#else
#include <cstdint>
#endif
#include <algorithm>
#include <cstddef>
@@ -61,6 +63,8 @@
#include "amd_options.hpp"
#include "core/util/utils.h"
#include "executable.hpp"
#include "AMDHSAKernelDescriptor.h"
using namespace rocr::amd::hsa;
@@ -88,7 +92,12 @@ static __forceinline link_map*& r_debug_tail() {
namespace rocr {
// Having a side effect prevents call site optimization that allows removal of a noinline function call
// with no side effect.
__attribute__((noinline)) void _loader_debug_state() {
#if defined(__linux__)
__attribute__((noinline))
#else
__declspec(noinline)
#endif
void _loader_debug_state() {
static volatile int function_needs_a_side_effect = 0;
function_needs_a_side_effect ^= 1;
}
+71 -1
Просмотреть файл
@@ -48,7 +48,9 @@
#include <cstdint>
#include <iostream>
#include <libelf.h>
#if defined(__linux__)
#include <link.h>
#endif
#include <list>
#include <string>
#include <unordered_map>
@@ -62,6 +64,74 @@
#include "inc/amd_hsa_kernel_code.h"
#include "amd_hsa_locks.hpp"
#if defined(_WIN32) || defined(_WIN64)
// r_version history:
// 1: Initial debug protocol
// 2: New trap handler ABI. The reason for halting a wave is recorded in ttmp11[8:7].
// 3: New trap handler ABI. A wave halted at S_ENDPGM rewinds its PC by 8 bytes, and sets
// ttmp11[9]=1. 4: New trap handler ABI. Save the trap id in ttmp11[16:9] 5: New trap handler ABI.
// Save the PC in ttmp11[22:7] ttmp6[31:0], and park the wave if stopped 6: New trap handler ABI.
// ttmp6[25:0] contains dispatch index modulo queue size 7: New trap handler ABI. Send interrupts as
// a bitmask, coalescing concurrent exceptions. 8: New trap handler ABI. for gfx942: Initialize
// ttmp[4:5] if ttmp11[31] == 0. 9: New trap handler ABI. For gfx11: Save PC in ttmp11[22:7]
// ttmp6[31:0], and park the wave if stopped. 10: New trap handler ABI. Set status.skip_export when
// halting the wave.
// For gfx942, set ttmp6[31] = 0 if ttmp11[31] == 0.
#if _WIN64
#define __WORDSIZE 64
#else
#define __WORDSIZE 32
#endif
#define __ELF_NATIVE_CLASS __WORDSIZE
/* We use this macro to refer to ELF types independent of the native wordsize.
`ElfW(TYPE)' is used in place of `Elf32_TYPE' or `Elf64_TYPE'. */
#define _ElfW_1(e, w, t) e##w##t
#define _ElfW(e, w, t) _ElfW_1(e, w, _##t)
#define ElfW(type) _ElfW(Elf, __ELF_NATIVE_CLASS, type)
/* Structure describing a loaded shared object. The `l_next' and `l_prev'
members form a chain of all the shared objects loaded at startup.
These data structures exist in space used by the run-time dynamic linker;
modifying them may have disastrous results. */
struct link_map {
/* These first few members are part of the protocol with the debugger.
This is the same format used in SVR4. */
ElfW(Addr) l_addr; /* Difference between the address in the ELF
file and the addresses in memory. */
char* l_name; /* Absolute file name object was found in. */
ElfW(Dyn) * l_ld; /* Dynamic section of the shared object. */
struct link_map *l_next, *l_prev; /* Chain of loaded objects. */
};
struct r_debug {
/* Version number for this protocol. It should be greater than 0. */
int r_version;
struct link_map* r_map; /* Head of the chain of loaded objects. */
/* This is the address of a function internal to the run-time linker,
that will always be called when the linker begins to map in a
library or unmap it, and again when the mapping change is complete.
The debugger can set a breakpoint at this address if it wants to
notice shared object mapping changes. */
ElfW(Addr) r_brk;
enum RT {
/* This state value describes the mapping change taking place when
the `r_brk' address is called. */
RT_CONSISTENT, /* Mapping change is complete. */
RT_ADD, /* Beginning to add a new object. */
RT_DELETE /* Beginning to remove an object mapping. */
} r_state;
ElfW(Addr) r_ldbase; /* Base address the linker is loaded at. */
};
#endif
namespace rocr {
namespace amd {
namespace hsa {
@@ -604,7 +674,7 @@ public:
hsa_status_t QuerySegmentDescriptors(
hsa_ven_amd_loader_segment_descriptor_t *segment_descriptors,
size_t *num_segment_descriptors) override;
#undef FindExecutable
hsa_executable_t FindExecutable(uint64_t device_address) override;
uint64_t FindHostAddress(uint64_t device_address) override;
+4 -4
Просмотреть файл
@@ -306,7 +306,7 @@ hsa_status_t PcsRuntime::PcSamplingCreateInternal(
hsa_status_t PcsRuntime::PcSamplingDestroy(hsa_ven_amd_pcs_t handle) {
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
if (pcSamplingSessionIt == pc_sampling_.end()) {
debug_warning(false && "Cannot find PcSampling session");
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
@@ -320,7 +320,7 @@ hsa_status_t PcsRuntime::PcSamplingDestroy(hsa_ven_amd_pcs_t handle) {
hsa_status_t PcsRuntime::PcSamplingStart(hsa_ven_amd_pcs_t handle) {
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
if (pcSamplingSessionIt == pc_sampling_.end()) {
debug_warning(false && "Cannot find PcSampling session");
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
@@ -332,7 +332,7 @@ hsa_status_t PcsRuntime::PcSamplingStart(hsa_ven_amd_pcs_t handle) {
hsa_status_t PcsRuntime::PcSamplingStop(hsa_ven_amd_pcs_t handle) {
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
if (pcSamplingSessionIt == pc_sampling_.end()) {
debug_warning(false && "Cannot find PcSampling session");
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
@@ -344,7 +344,7 @@ hsa_status_t PcsRuntime::PcSamplingStop(hsa_ven_amd_pcs_t handle) {
hsa_status_t PcsRuntime::PcSamplingFlush(hsa_ven_amd_pcs_t handle) {
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
if (pcSamplingSessionIt == pc_sampling_.end()) {
debug_warning(false && "Cannot find PcSampling session");
return HSA_STATUS_ERROR_INVALID_ARGUMENT;