Add windows build support into ROCr (#912)
Make sure ROCR can be compiled under windows. Extra setup for the windows build environment is required. The change should not have any functional changes under Linux.
This commit is contained in:
zatwierdzone przez
GitHub
rodzic
96a0d16eda
commit
913743d433
@@ -117,6 +117,11 @@ set_target_properties(hsakmt PROPERTIES
|
||||
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/archive"
|
||||
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/lib"
|
||||
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/runtime")
|
||||
if (WIN32)
|
||||
set_target_properties(hsakmt PROPERTIES
|
||||
CXX_STANDARD 20
|
||||
CXX_STANDARD_REQUIRED ON)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNK_VIRTIO)
|
||||
add_rocm_subdir(libhsakmt/src/virtio "${THUNK_VIRTIO_DEFINITIONS}")
|
||||
@@ -128,6 +133,11 @@ if (BUILD_ROCR)
|
||||
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/archive"
|
||||
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/lib"
|
||||
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/runtime")
|
||||
if (WIN32)
|
||||
set_target_properties(hsa-runtime64 PROPERTIES
|
||||
CXX_STANDARD 20
|
||||
CXX_STANDARD_REQUIRED ON)
|
||||
endif()
|
||||
|
||||
if (BUILD_SHARED_LIBS)
|
||||
add_dependencies(hsa-runtime64 hsakmt)
|
||||
@@ -187,6 +197,7 @@ if (DEFINED ENV{ROCM_LIBPATCH_VERSION})
|
||||
message("Using CPACK_PACKAGE_VERSION ${CPACK_PACKAGE_VERSION}")
|
||||
endif()
|
||||
|
||||
if (UNIX)
|
||||
# Debian package specific variables
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_NAME "hsa-rocr")
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_NAME "hsa-rocr-dev")
|
||||
@@ -223,6 +234,10 @@ set(CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS "libdrm-amdgpu-dev | libdrm-dev, rocm-core
|
||||
set(CPACK_DEBIAN_ASAN_PACKAGE_RECOMMENDS "libdrm-amdgpu-dev")
|
||||
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_RECOMMENDS "libdrm-amdgpu-amdgpu1")
|
||||
else()
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS "hsakmt-roct")
|
||||
set(CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS "hsakmt-roct")
|
||||
endif()
|
||||
if (ROCM_DEP_ROCMCORE)
|
||||
string(APPEND CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS ", rocm-core")
|
||||
string(APPEND CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS ", rocm-core-asan")
|
||||
@@ -244,6 +259,7 @@ set(CPACK_RPM_DEV_PACKAGE_OBSOLETES "hsakmt-roct,hsakmt-roct-devel,hsakmt-roct-d
|
||||
|
||||
set(CPACK_RPM_DEV_PACKAGE_NAME "hsa-rocr-devel")
|
||||
set(CPACK_RPM_ASAN_PACKAGE_NAME "hsa-rocr-asan")
|
||||
if (UNIX)
|
||||
if (DEFINED ENV{CPACK_RPM_PACKAGE_RELEASE})
|
||||
set(CPACK_RPM_PACKAGE_RELEASE $ENV{CPACK_RPM_PACKAGE_RELEASE})
|
||||
else()
|
||||
@@ -254,6 +270,7 @@ string(APPEND CPACK_RPM_PACKAGE_RELEASE "%{?dist}")
|
||||
set(CPACK_RPM_FILE_NAME "RPM-DEFAULT")
|
||||
message("CPACK_RPM_PACKAGE_RELEASE: ${CPACK_RPM_PACKAGE_RELEASE}")
|
||||
set(CPACK_RPM_PACKAGE_LICENSE "NCSA")
|
||||
endif()
|
||||
|
||||
## Process the Rpm install/remove scripts to update the CPACK variables
|
||||
configure_file("${CMAKE_CURRENT_SOURCE_DIR}/RPM/Binary/post.in" RPM/Binary/post @ONLY)
|
||||
@@ -289,7 +306,7 @@ endif()
|
||||
if (ROCM_DEP_ROCMCORE)
|
||||
string(APPEND CPACK_RPM_BINARY_PACKAGE_REQUIRES " rocm-core")
|
||||
string(APPEND CPACK_RPM_ASAN_PACKAGE_REQUIRES " rocm-core-asan")
|
||||
else()
|
||||
elseif (UNIX)
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_PACKAGE_REQUIRES ${CPACK_RPM_PACKAGE_REQUIRES})
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_DEBIAN_PACKAGE_DEPENDS ${CPACK_DEBIAN_PACKAGE_DEPENDS})
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_DEV_PACKAGE_REQUIRES ${CPACK_RPM_DEV_PACKAGE_REQUIRES})
|
||||
|
||||
@@ -1,3 +1,11 @@
|
||||
################################################################################
|
||||
##
|
||||
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
|
||||
##
|
||||
## SPDX-License-Identifier: MIT
|
||||
##
|
||||
################################################################################
|
||||
|
||||
cmake_minimum_required ( VERSION 3.5.0 )
|
||||
|
||||
# Set ext runtime module name and project name.
|
||||
@@ -23,8 +31,10 @@ include ( utils )
|
||||
|
||||
## Compiler preproc definitions.
|
||||
#add_definitions ( -D__linux__ )
|
||||
add_definitions ( -DUNIX_OS )
|
||||
add_definitions ( -DLINUX )
|
||||
if(UNIX)
|
||||
add_definitions ( -DUNIX_OS )
|
||||
add_definitions ( -DLINUX )
|
||||
endif()
|
||||
add_definitions ( -D__AMD64__ )
|
||||
add_definitions ( -D__x86_64__ )
|
||||
add_definitions ( -DAMD_INTERNAL_BUILD )
|
||||
|
||||
@@ -48,7 +48,11 @@ cmake_minimum_required ( VERSION 3.7 )
|
||||
unset ( hsa-runtime64_LIB_DEPENDS CACHE )
|
||||
|
||||
set(CMAKE_VERBOSE_MAKEFILE ON)
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
if (UNIX)
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
else()
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
endif()
|
||||
|
||||
## Set core runtime module name and project name.
|
||||
set ( CORE_RUNTIME_NAME "hsa-runtime64" )
|
||||
@@ -89,35 +93,46 @@ if(NOT LibElf_FOUND)
|
||||
find_package(LibElf REQUIRED)
|
||||
endif()
|
||||
|
||||
pkg_check_modules(drm REQUIRED IMPORTED_TARGET libdrm)
|
||||
|
||||
## Create the rocr target.
|
||||
add_library( ${CORE_RUNTIME_TARGET} "" )
|
||||
|
||||
if (UNIX)
|
||||
pkg_check_modules(drm REQUIRED IMPORTED_TARGET libdrm)
|
||||
else()
|
||||
target_include_directories(${CORE_RUNTIME_TARGET} PRIVATE ${LIBELF_INCLUDE_DIR})
|
||||
if (${BUILD_SHARED_LIBS})
|
||||
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE oclelf)
|
||||
endif()
|
||||
endif()
|
||||
## Enforce uniform output file naming.
|
||||
set_property(TARGET ${CORE_RUNTIME_TARGET} PROPERTY OUTPUT_NAME ${CORE_RUNTIME_NAME} )
|
||||
|
||||
## Compiler preproc definitions.
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" __linux__ HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
|
||||
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}" )
|
||||
|
||||
## Check for memfd_create syscall
|
||||
include(CheckSymbolExists)
|
||||
CHECK_SYMBOL_EXISTS ( "__NR_memfd_create" "sys/syscall.h" HAVE_MEMFD_CREATE )
|
||||
if ( HAVE_MEMFD_CREATE )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_MEMFD_CREATE )
|
||||
endif()
|
||||
|
||||
## Check for _GNU_SOURCE pthread extensions
|
||||
set(CMAKE_REQUIRED_DEFINITIONS -D_GNU_SOURCE)
|
||||
CHECK_SYMBOL_EXISTS ( "pthread_attr_setaffinity_np" "pthread.h" HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
CHECK_SYMBOL_EXISTS ( "pthread_rwlockattr_setkind_np" "pthread.h" HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
unset(CMAKE_REQUIRED_DEFINITIONS)
|
||||
if ( HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
endif()
|
||||
if ( HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
if (UNIX)
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" __linux__ HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
|
||||
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}" )
|
||||
|
||||
## Check for memfd_create syscall
|
||||
include(CheckSymbolExists)
|
||||
CHECK_SYMBOL_EXISTS ( "__NR_memfd_create" "sys/syscall.h" HAVE_MEMFD_CREATE )
|
||||
if ( HAVE_MEMFD_CREATE )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_MEMFD_CREATE )
|
||||
endif()
|
||||
|
||||
## Check for _GNU_SOURCE pthread extensions
|
||||
set(CMAKE_REQUIRED_DEFINITIONS -D_GNU_SOURCE)
|
||||
CHECK_SYMBOL_EXISTS ( "pthread_attr_setaffinity_np" "pthread.h" HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
CHECK_SYMBOL_EXISTS ( "pthread_rwlockattr_setkind_np" "pthread.h" HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
unset(CMAKE_REQUIRED_DEFINITIONS)
|
||||
if ( HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
endif()
|
||||
if ( HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
endif()
|
||||
else()
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" AMD_LIBELF=1 HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
|
||||
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}")
|
||||
endif()
|
||||
|
||||
## Set include directories for ROCr runtime
|
||||
@@ -133,16 +148,18 @@ target_include_directories( ${CORE_RUNTIME_TARGET}
|
||||
|
||||
|
||||
## ------------------------- Linux Compiler and Linker options -------------------------
|
||||
set ( HSA_CXX_FLAGS ${HSA_COMMON_CXX_FLAGS} -fexceptions -fno-rtti -fvisibility=hidden -Wno-error=missing-braces -Wno-error=sign-compare -Wno-sign-compare -Wno-write-strings -Wno-conversion-null -fno-math-errno -fno-threadsafe-statics -fmerge-all-constants -fms-extensions -Wno-error=comment -Wno-comment -Wno-error=pointer-arith -Wno-pointer-arith -Wno-error=unused-variable -Wno-error=unused-function )
|
||||
if (UNIX)
|
||||
set ( HSA_CXX_FLAGS ${HSA_COMMON_CXX_FLAGS} -fexceptions -fno-rtti -fvisibility=hidden -Wno-error=missing-braces -Wno-error=sign-compare -Wno-sign-compare -Wno-write-strings -Wno-conversion-null -fno-math-errno -fno-threadsafe-statics -fmerge-all-constants -fms-extensions -Wno-error=comment -Wno-comment -Wno-error=pointer-arith -Wno-pointer-arith -Wno-error=unused-variable -Wno-error=unused-function )
|
||||
|
||||
## Extra x86 specific settings
|
||||
if ( CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64" )
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -mmwaitx )
|
||||
## Extra x86 specific settings
|
||||
if ( CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64" )
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -mmwaitx )
|
||||
endif()
|
||||
|
||||
## Extra image settings - audit!
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-deprecated-declarations )
|
||||
endif()
|
||||
|
||||
## Extra image settings - audit!
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-deprecated-declarations )
|
||||
|
||||
if ( CMAKE_COMPILER_IS_GNUCXX )
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-error=maybe-uninitialized -Wno-error=unused-but-set-variable)
|
||||
endif ()
|
||||
@@ -153,9 +170,14 @@ if ( CMAKE_CXX_COMPILER_ID MATCHES "Clang")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set ( DRVDEF "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.so.def" )
|
||||
set ( LNKSCR "hsacore.so.link" )
|
||||
set ( HSA_SHARED_LINK_FLAGS "-Wl,-Bdynamic -Wl,-z,noexecstack -Wl,${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR} -Wl,--version-script=${DRVDEF} -Wl,--enable-new-dtags" )
|
||||
if (UNIX)
|
||||
set ( LNKSCR "hsacore.so.link" )
|
||||
set ( DRVDEF "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.so.def" )
|
||||
set(HSA_SHARED_LINK_FLAGS "-Wl,-Bdynamic -Wl,-z,noexecstack -Wl,${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR} -Wl,--version-script=${DRVDEF} -Wl,--enable-new-dtags")
|
||||
else()
|
||||
target_sources(${CORE_RUNTIME_TARGET} PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.dll.def")
|
||||
set(HSA_SHARED_LINK_FLAGS "")
|
||||
endif()
|
||||
|
||||
target_compile_options(${CORE_RUNTIME_TARGET} PRIVATE ${HSA_CXX_FLAGS})
|
||||
#target_link_options not available prior to CMake 3.13
|
||||
@@ -165,13 +187,9 @@ set_property(TARGET ${CORE_RUNTIME_TARGET} PROPERTY LINK_FLAGS ${HSA_SHARED_LINK
|
||||
## Source files.
|
||||
set ( SRCS core/driver/driver.cpp
|
||||
core/driver/kfd/amd_kfd_driver.cpp
|
||||
core/driver/xdna/amd_xdna_driver.cpp
|
||||
core/util/lnx/os_linux.cpp
|
||||
core/util/small_heap.cpp
|
||||
core/util/timer.cpp
|
||||
core/util/flag.cpp
|
||||
core/runtime/amd_aie_agent.cpp
|
||||
core/runtime/amd_aie_aql_queue.cpp
|
||||
core/runtime/amd_blit_kernel.cpp
|
||||
core/runtime/amd_blit_sdma.cpp
|
||||
core/runtime/amd_cpu_agent.cpp
|
||||
@@ -208,12 +226,22 @@ set ( SRCS core/driver/driver.cpp
|
||||
libamdhsacode/amd_hsa_code.cpp
|
||||
libamdhsacode/amd_core_dump.cpp )
|
||||
|
||||
if(UNIX)
|
||||
set(SRC_OS core/util/lnx/os_linux.cpp)
|
||||
set(SRC_XDNA core/driver/xdna/amd_xdna_driver.cpp
|
||||
core/runtime/amd_aie_agent.cpp
|
||||
core/runtime/amd_aie_aql_queue.cpp)
|
||||
else()
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE NOMINMAX)
|
||||
set(SRC_OS core/util/win/os_win.cpp)
|
||||
endif()
|
||||
|
||||
if ( BUILD_THUNK_VIRTIO )
|
||||
list(APPEND SRCS core/driver/virtio/amd_kfd_virtio_driver.cpp)
|
||||
target_compile_definitions(hsa-runtime64 PRIVATE HSAKMT_VIRTIO_ENABLED=1)
|
||||
endif()
|
||||
|
||||
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${SRCS} )
|
||||
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${SRCS} ${SRC_OS} ${SRC_XDNA} )
|
||||
|
||||
## Depend on trap handler target.
|
||||
add_subdirectory( ${CMAKE_CURRENT_SOURCE_DIR}/core/runtime/trap_handler )
|
||||
@@ -233,10 +261,15 @@ if (${PC_SAMPLING_SUPPORT})
|
||||
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${PCS_SRCS} )
|
||||
endif()
|
||||
|
||||
if ( NOT DEFINED IMAGE_SUPPORT AND CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64|loongarch64" )
|
||||
set ( IMAGE_SUPPORT ON )
|
||||
endif()
|
||||
if (UNIX)
|
||||
if ( NOT DEFINED IMAGE_SUPPORT AND CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64|loongarch64" )
|
||||
set ( IMAGE_SUPPORT ON )
|
||||
endif()
|
||||
set ( IMAGE_SUPPORT ${IMAGE_SUPPORT} CACHE BOOL "Build with image support (default: ON for x86, OFF elsewise)." )
|
||||
else()
|
||||
# Force IMAGE_SUPPORT to be OFF
|
||||
set(IMAGE_SUPPORT OFF CACHE BOOL "Build with image support (forced to OFF)" FORCE)
|
||||
endif()
|
||||
|
||||
## Optional image module defintions.
|
||||
if(${IMAGE_SUPPORT})
|
||||
@@ -305,30 +338,45 @@ if(${IMAGE_SUPPORT})
|
||||
|
||||
endif()
|
||||
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE elf::elf dl pthread rt )
|
||||
if (UNIX)
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE elf::elf dl pthread rt )
|
||||
else()
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE Ws2_32 )
|
||||
target_link_directories(${CORE_RUNTIME_TARGET} PRIVATE ${DXCORE_LIB_PATH})
|
||||
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE dxcore)
|
||||
endif()
|
||||
|
||||
# For static package rocprofiler-register dependency is not required
|
||||
# Link to hsakmt target for shared library builds
|
||||
# Link to hsakmt-staticdrm target for static library builds
|
||||
if( BUILD_SHARED_LIBS )
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt PkgConfig::drm)
|
||||
if( BUILD_THUNK_VIRTIO )
|
||||
message(STATUS "Building with virtio support")
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt_virtio)
|
||||
endif()
|
||||
find_package(rocprofiler-register)
|
||||
if(rocprofiler-register_FOUND)
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HSA_ROCPROFILER_REGISTER=1
|
||||
HSA_VERSION_MAJOR=${VERSION_MAJOR}
|
||||
HSA_VERSION_MINOR=${VERSION_MINOR}
|
||||
HSA_VERSION_PATCH=${VERSION_PATCH})
|
||||
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE rocprofiler-register::rocprofiler-register)
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER ON CACHE INTERNAL "")
|
||||
if (UNIX)
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt PkgConfig::drm)
|
||||
if( BUILD_THUNK_VIRTIO )
|
||||
message(STATUS "Building with virtio support")
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt_virtio)
|
||||
endif()
|
||||
find_package(rocprofiler-register)
|
||||
if(rocprofiler-register_FOUND)
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HSA_ROCPROFILER_REGISTER=1
|
||||
HSA_VERSION_MAJOR=${VERSION_MAJOR}
|
||||
HSA_VERSION_MINOR=${VERSION_MINOR}
|
||||
HSA_VERSION_PATCH=${VERSION_PATCH})
|
||||
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE rocprofiler-register::rocprofiler-register)
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER ON CACHE INTERNAL "")
|
||||
else()
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER OFF CACHE INTERNAL "")
|
||||
endif() # end rocprofiler-register_FOUND
|
||||
else()
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER OFF CACHE INTERNAL "")
|
||||
endif() # end rocprofiler-register_FOUND
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt)
|
||||
endif()
|
||||
else()
|
||||
include_directories(${drm_INCLUDE_DIRS})
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
|
||||
if (UNIX)
|
||||
include_directories(${drm_INCLUDE_DIRS})
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
|
||||
else()
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
|
||||
endif()
|
||||
endif()#end BUILD_SHARED_LIBS
|
||||
|
||||
## Set the VERSION and SOVERSION values
|
||||
@@ -350,8 +398,11 @@ if( NOT ${BUILD_SHARED_LIBS} )
|
||||
|
||||
## Add external link requirements.
|
||||
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE hsakmt-staticdrm::hsakmt-staticdrm )
|
||||
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE elf::elf dl pthread rt )
|
||||
|
||||
if (UNIX)
|
||||
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE elf::elf dl pthread rt )
|
||||
else()
|
||||
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE rt )
|
||||
endif()
|
||||
install ( TARGETS ${CORE_RUNTIME_NAME} EXPORT ${CORE_RUNTIME_NAME}Targets )
|
||||
endif()
|
||||
|
||||
@@ -404,10 +455,12 @@ install(FILES ${CMAKE_CURRENT_BINARY_DIR}/${CORE_RUNTIME_NAME}-config.cmake ${CM
|
||||
|
||||
# Install build files needed only when using a static build.
|
||||
if( NOT ${BUILD_SHARED_LIBS} )
|
||||
# libelf find package module
|
||||
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/FindLibElf.cmake ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/COPYING-CMAKE-SCRIPTS
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
|
||||
COMPONENT dev)
|
||||
if (UNIX)
|
||||
# libelf find package module
|
||||
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/FindLibElf.cmake ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/COPYING-CMAKE-SCRIPTS
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
|
||||
COMPONENT dev)
|
||||
endif()
|
||||
# Linker script (defines function aliases)
|
||||
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR}
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
|
||||
|
||||
@@ -17,57 +17,78 @@ if (LIBELF_FOUND)
|
||||
return()
|
||||
endif (LIBELF_FOUND)
|
||||
|
||||
find_path (LIBELF_INCLUDE_DIRS
|
||||
NAMES
|
||||
libelf.h
|
||||
PATHS
|
||||
/usr/include
|
||||
/usr/include/libelf
|
||||
/usr/local/include
|
||||
/usr/local/include/libelf
|
||||
/opt/local/include
|
||||
/opt/local/include/libelf
|
||||
ENV CPATH)
|
||||
|
||||
find_library (LIBELF_LIBRARIES
|
||||
NAMES
|
||||
elf
|
||||
PATHS
|
||||
/usr/lib
|
||||
/usr/lib64
|
||||
/usr/local/lib
|
||||
/usr/local/lib64
|
||||
/opt/local/lib
|
||||
/opt/local/lib64
|
||||
ENV LIBRARY_PATH
|
||||
ENV LD_LIBRARY_PATH)
|
||||
|
||||
include (FindPackageHandleStandardArgs)
|
||||
|
||||
|
||||
# handle the QUIETLY and REQUIRED arguments and set LIBELF_FOUND to TRUE if all listed variables are TRUE
|
||||
FIND_PACKAGE_HANDLE_STANDARD_ARGS(LibElf DEFAULT_MSG
|
||||
LIBELF_LIBRARIES
|
||||
LIBELF_INCLUDE_DIRS)
|
||||
|
||||
SET(CMAKE_REQUIRED_LIBRARIES elf)
|
||||
if (CMAKE_CXX_COMPILER_LOADED)
|
||||
INCLUDE(CheckCXXSourceCompiles)
|
||||
CHECK_CXX_SOURCE_COMPILES("#include <libelf.h>
|
||||
int main() {
|
||||
Elf *e = (Elf*)0;
|
||||
size_t sz;
|
||||
elf_getshdrstrndx(e, &sz);
|
||||
return 0;
|
||||
}" ELF_GETSHDRSTRNDX)
|
||||
if (UNIX)
|
||||
find_path (LIBELF_INCLUDE_DIRS
|
||||
NAMES
|
||||
libelf.h
|
||||
PATHS
|
||||
/usr/include
|
||||
/usr/include/libelf
|
||||
/usr/local/include
|
||||
/usr/local/include/libelf
|
||||
/opt/local/include
|
||||
/opt/local/include/libelf
|
||||
ENV CPATH)
|
||||
|
||||
find_library (LIBELF_LIBRARIES
|
||||
NAMES
|
||||
elf
|
||||
PATHS
|
||||
/usr/lib
|
||||
/usr/lib64
|
||||
/usr/local/lib
|
||||
/usr/local/lib64
|
||||
/opt/local/lib
|
||||
/opt/local/lib64
|
||||
ENV LIBRARY_PATH
|
||||
ENV LD_LIBRARY_PATH)
|
||||
|
||||
include (FindPackageHandleStandardArgs)
|
||||
|
||||
|
||||
# handle the QUIETLY and REQUIRED arguments and set LIBELF_FOUND to TRUE if all listed variables are TRUE
|
||||
FIND_PACKAGE_HANDLE_STANDARD_ARGS(LibElf DEFAULT_MSG
|
||||
LIBELF_LIBRARIES
|
||||
LIBELF_INCLUDE_DIRS)
|
||||
|
||||
SET(CMAKE_REQUIRED_LIBRARIES elf)
|
||||
if (CMAKE_CXX_COMPILER_LOADED)
|
||||
INCLUDE(CheckCXXSourceCompiles)
|
||||
CHECK_CXX_SOURCE_COMPILES("#include <libelf.h>
|
||||
int main() {
|
||||
Elf *e = (Elf*)0;
|
||||
size_t sz;
|
||||
elf_getshdrstrndx(e, &sz);
|
||||
return 0;
|
||||
}" ELF_GETSHDRSTRNDX)
|
||||
else()
|
||||
set ( ELF_GETSHDRSTRNDX "TRUE" )
|
||||
endif(CMAKE_CXX_COMPILER_LOADED)
|
||||
|
||||
mark_as_advanced(LIBELF_INCLUDE_DIRS LIBELF_LIBRARIES ELF_GETSHDRSTRNDX)
|
||||
|
||||
if(LIBELF_FOUND)
|
||||
add_library(elf::elf UNKNOWN IMPORTED)
|
||||
set_property(TARGET elf::elf PROPERTY IMPORTED_LOCATION ${LIBELF_LIBRARIES})
|
||||
set_property(TARGET elf::elf PROPERTY INTERFACE_INCLUDE_DIRECTORIES ${LIBELF_INCLUDE_DIRS})
|
||||
endif()
|
||||
else()
|
||||
set ( ELF_GETSHDRSTRNDX "TRUE" )
|
||||
endif(CMAKE_CXX_COMPILER_LOADED)
|
||||
find_path(ROCR_LIBELF_INCLUDE_DIR libelf.h
|
||||
HINTS
|
||||
${AMD_LIBELF_PATH}
|
||||
PATHS
|
||||
${CMAKE_SOURCE_DIR}/hsail-compiler/lib/loaders/elf/utils/libelf
|
||||
${CMAKE_SOURCE_DIR}/../hsail-compiler/lib/loaders/elf/utils/libelf
|
||||
${CMAKE_SOURCE_DIR}/../../hsail-compiler/lib/loaders/elf/utils/libelf
|
||||
NO_DEFAULT_PATH)
|
||||
|
||||
mark_as_advanced(LIBELF_INCLUDE_DIRS LIBELF_LIBRARIES ELF_GETSHDRSTRNDX)
|
||||
|
||||
if(LIBELF_FOUND)
|
||||
add_library(elf::elf UNKNOWN IMPORTED)
|
||||
set_property(TARGET elf::elf PROPERTY IMPORTED_LOCATION ${LIBELF_LIBRARIES})
|
||||
set_property(TARGET elf::elf PROPERTY INTERFACE_INCLUDE_DIRECTORIES ${LIBELF_INCLUDE_DIRS})
|
||||
message("=> LibElf paths:" ${CMAKE_CURRENT_BINARY_DIR} ${ROCR_LIBELF_INCLUDE_DIR})
|
||||
if (${BUILD_SHARED_LIBS})
|
||||
mark_as_advanced(ROCR_LIBELF_INCLUDE_DIR)
|
||||
add_subdirectory("${ROCR_LIBELF_INCLUDE_DIR}" ${CMAKE_CURRENT_BINARY_DIR}/libelf)
|
||||
endif()
|
||||
set(USE_AMD_LIBELF "yes" CACHE FORCE "")
|
||||
set(AMD_ELFTOOLCHAIN_DIR ${ROCR_LIBELF_INCLUDE_DIR}/../..;${ROCR_LIBELF_INCLUDE_DIR}/../common/win32;${ROCR_LIBELF_INCLUDE_DIR}/../common)
|
||||
set(ROCR_LIBELF_INCLUDE_DIR ${ROCR_LIBELF_INCLUDE_DIR};${AMD_ELFTOOLCHAIN_DIR})
|
||||
set(LIBELF_INCLUDE_DIR ${ROCR_LIBELF_INCLUDE_DIR})
|
||||
endif()
|
||||
|
||||
@@ -45,9 +45,11 @@
|
||||
#include <memory>
|
||||
#include <string>
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <amdgpu_drm.h>
|
||||
#include <link.h>
|
||||
#include <sys/ioctl.h>
|
||||
#endif
|
||||
|
||||
#include "hsakmt/hsakmt.h"
|
||||
|
||||
@@ -55,11 +57,16 @@
|
||||
#include "core/inc/amd_memory_region.h"
|
||||
#include "core/inc/runtime.h"
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include "loader/executable.hpp"
|
||||
#endif
|
||||
|
||||
extern r_debug _amdgpu_r_debug;
|
||||
|
||||
namespace rocr {
|
||||
namespace AMD {
|
||||
|
||||
#if defined(__linux__)
|
||||
static_assert(
|
||||
(sizeof(core::ShareableHandle::handle) >= sizeof(amdgpu_bo_handle)) &&
|
||||
(alignof(core::ShareableHandle::handle) >= alignof(amdgpu_bo_handle)),
|
||||
@@ -82,6 +89,7 @@ __forceinline uint64_t drm_perm(hsa_access_permission_t perm) {
|
||||
}
|
||||
|
||||
} // namespace
|
||||
#endif
|
||||
|
||||
KfdDriver::KfdDriver(std::string devnode_name)
|
||||
: core::Driver(core::DriverType::KFD, std::move(devnode_name)) {}
|
||||
@@ -425,6 +433,7 @@ hsa_status_t KfdDriver::ExportDMABuf(void *mem, size_t size, int *dmabuf_fd,
|
||||
|
||||
hsa_status_t KfdDriver::ImportDMABuf(int dmabuf_fd, core::Agent &agent,
|
||||
core::ShareableHandle &handle) {
|
||||
#if defined(__linux__)
|
||||
auto &gpu_agent = static_cast<GpuAgent &>(agent);
|
||||
amdgpu_bo_import_result res;
|
||||
auto ret = DRM_CALL(amdgpu_bo_import(
|
||||
@@ -433,12 +442,16 @@ hsa_status_t KfdDriver::ImportDMABuf(int dmabuf_fd, core::Agent &agent,
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
handle.handle = reinterpret_cast<uint64_t>(res.buf_handle);
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
hsa_status_t KfdDriver::Map(core::ShareableHandle handle, void *mem,
|
||||
size_t offset, size_t size,
|
||||
hsa_access_permission_t perms) {
|
||||
#if defined(__linux__)
|
||||
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
|
||||
if (!ldrm_bo)
|
||||
return HSA_STATUS_ERROR;
|
||||
@@ -446,12 +459,15 @@ hsa_status_t KfdDriver::Map(core::ShareableHandle handle, void *mem,
|
||||
if (DRM_CALL(amdgpu_bo_va_op(ldrm_bo, offset, size, reinterpret_cast<uint64_t>(mem),
|
||||
drm_perm(perms), AMDGPU_VA_OP_MAP)) != 0)
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
hsa_status_t KfdDriver::Unmap(core::ShareableHandle handle, void *mem,
|
||||
size_t offset, size_t size) {
|
||||
#if defined(__linux__)
|
||||
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
|
||||
if (!ldrm_bo)
|
||||
return HSA_STATUS_ERROR;
|
||||
@@ -459,11 +475,14 @@ hsa_status_t KfdDriver::Unmap(core::ShareableHandle handle, void *mem,
|
||||
if (DRM_CALL(amdgpu_bo_va_op(ldrm_bo, offset, size, reinterpret_cast<uint64_t>(mem), 0,
|
||||
AMDGPU_VA_OP_UNMAP)) != 0)
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
hsa_status_t KfdDriver::ReleaseShareableHandle(core::ShareableHandle &handle) {
|
||||
#if defined(__linux__)
|
||||
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
|
||||
if (!ldrm_bo)
|
||||
return HSA_STATUS_ERROR;
|
||||
@@ -473,6 +492,9 @@ hsa_status_t KfdDriver::ReleaseShareableHandle(core::ShareableHandle &handle) {
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
handle = {};
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -650,6 +672,7 @@ hsa_status_t KfdDriver::DeregisterMemory(void* ptr) const {
|
||||
hsa_status_t KfdDriver::MakeMemoryResident(const void* mem, size_t size, uint64_t* alternate_va,
|
||||
const HsaMemMapFlags* mem_flags, uint32_t num_nodes,
|
||||
const uint32_t* nodes) const {
|
||||
#if defined(__linux__)
|
||||
if (mem_flags == nullptr && nodes == nullptr) {
|
||||
if (HSAKMT_CALL(hsaKmtMapMemoryToGPU(const_cast<void*>(mem), size, alternate_va)) !=
|
||||
HSAKMT_STATUS_SUCCESS) {
|
||||
@@ -663,7 +686,19 @@ hsa_status_t KfdDriver::MakeMemoryResident(const void* mem, size_t size, uint64_
|
||||
debug_print("Invalid memory flags ptr:%p nodes ptr:%p\n", mem_flags, nodes);
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
}
|
||||
#else
|
||||
assert(num_nodes > 0);
|
||||
assert(nodes != NULL);
|
||||
|
||||
*alternate_va = 0;
|
||||
const HSAKMT_STATUS status =
|
||||
HSAKMT_CALL(hsaKmtMapMemoryToGPUNodes(const_cast<void*>(mem), size, alternate_va, *mem_flags,
|
||||
num_nodes, const_cast<uint32_t*>(nodes)));
|
||||
|
||||
if (status != HSAKMT_STATUS_SUCCESS) {
|
||||
return HSA_STATUS_ERROR;
|
||||
}
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2024-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -43,11 +43,10 @@
|
||||
#ifndef HSA_RUNTME_CORE_INC_AMD_AVAILABLE_DRIVERS_H_
|
||||
#define HSA_RUNTME_CORE_INC_AMD_AVAILABLE_DRIVERS_H_
|
||||
|
||||
#ifdef __linux__
|
||||
|
||||
#include "core/inc/amd_kfd_driver.h"
|
||||
#include "core/inc/amd_xdna_driver.h"
|
||||
|
||||
#ifdef __linux__
|
||||
#include "core/inc/amd_xdna_driver.h"
|
||||
#endif
|
||||
|
||||
#endif // header guard
|
||||
|
||||
@@ -116,6 +116,7 @@ class BlitKernel : public core::Blit {
|
||||
virtual bool GangLeader() const override { return false; }
|
||||
|
||||
const uint16_t kInvalidPacketHeader = HSA_PACKET_TYPE_INVALID;
|
||||
|
||||
private:
|
||||
union KernelArgs {
|
||||
struct __ALIGNED__(16) {
|
||||
|
||||
@@ -482,6 +482,7 @@ public:
|
||||
|
||||
/// @brief Finds the handle of executable to which @p device_address
|
||||
/// belongs. Return NULL handle if device address is invalid.
|
||||
#undef FindExecutable
|
||||
virtual hsa_executable_t FindExecutable(uint64_t device_address) = 0;
|
||||
|
||||
/// @brief Returns host address given @p device_address. If @p device_address
|
||||
|
||||
@@ -51,11 +51,12 @@
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
#include <thread>
|
||||
#include <sys/un.h>
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <sys/un.h>
|
||||
#include <xf86drm.h>
|
||||
#include <amdgpu.h>
|
||||
#else
|
||||
#include <hsakmt/drm/amdgpu.h>
|
||||
#endif
|
||||
|
||||
#include "core/inc/hsa_ext_interface.h"
|
||||
@@ -232,6 +233,7 @@ class Runtime {
|
||||
/// @param [in] size Copy size in bytes.
|
||||
///
|
||||
/// @retval ::HSA_STATUS_SUCCESS if memory copy is successful and completed.
|
||||
#undef CopyMemory
|
||||
hsa_status_t CopyMemory(void* dst, const void* src, size_t size);
|
||||
|
||||
/// @brief Non-blocking memory copy from src to dst.
|
||||
@@ -302,6 +304,7 @@ class Runtime {
|
||||
/// @param [in] count Number of uint32_t element to be set.
|
||||
///
|
||||
/// @retval ::HSA_STATUS_SUCCESS if memory fill is successful and completed.
|
||||
#undef FillMemory
|
||||
hsa_status_t FillMemory(void* ptr, uint32_t value, size_t count);
|
||||
|
||||
/// @brief Set agents as the whitelist to access ptr.
|
||||
@@ -517,7 +520,8 @@ class Runtime {
|
||||
|
||||
static bool IsGPUDriver(DriverType driver_type) {
|
||||
return driver_type == core::DriverType::KFD
|
||||
#ifdef HSAKMT_VIRTIO_ENABLED
|
||||
|
||||
#if defined(HSAKMT_VIRTIO_ENABLED) && defined(__linux__)
|
||||
|| driver_type == core::DriverType::KFD_VIRTIO
|
||||
#endif
|
||||
;
|
||||
|
||||
@@ -44,7 +44,11 @@
|
||||
#define HSA_RUNTIME_CORE_INC_THUNK_LOADER_H
|
||||
|
||||
#include <string>
|
||||
#if defined(__linux__)
|
||||
#include <amdgpu.h>
|
||||
#else
|
||||
#include "hsakmt/drm/amdgpu.h"
|
||||
#endif
|
||||
#include "hsakmt/hsakmttypes.h"
|
||||
|
||||
class DtifPlatform;
|
||||
|
||||
@@ -50,8 +50,11 @@
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <algorithm>
|
||||
#ifdef _WIN32
|
||||
#define WIN32_NO_STATUS
|
||||
#include <Windows.h>
|
||||
#undef WIN32_NO_STATUS
|
||||
#endif
|
||||
|
||||
#include <stdio.h>
|
||||
@@ -967,7 +970,7 @@ void AqlQueue::HandleInsufficientScratch(hsa_signal_value_t& error_code,
|
||||
maxGroupsPerEngine < 16 &&
|
||||
lanes_per_group * maxGroupsPerEngine < 256) {
|
||||
uint64_t groups_per_interleave = (256 + lanes_per_group - 1) / lanes_per_group;
|
||||
maxGroupsPerEngine = Min(groups_per_interleave, 16ul);
|
||||
maxGroupsPerEngine = Min(groups_per_interleave, uint64_t(16ul));
|
||||
}
|
||||
|
||||
// Populate all engines at max group occupancy, then clip down to device limits.
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -903,9 +903,13 @@ void BlitKernel::PopulateQueue(uint64_t index, uint64_t code_handle, void* args,
|
||||
// Ensure the packet body is written as header may get reordered when writing over PCIE
|
||||
_mm_sfence();
|
||||
}
|
||||
#if defined(__linux__)
|
||||
__atomic_store_n(&(queue_buffer[index & queue_bitmask_].full_header),
|
||||
kDispatchPacketHeader | packet.setup << 16, __ATOMIC_RELEASE);
|
||||
|
||||
#else
|
||||
std::atomic_ref<uint32_t> atomic_header(queue_buffer[index & queue_bitmask_].full_header);
|
||||
atomic_header.store(kDispatchPacketHeader | packet.setup << 16, std::memory_order_release);
|
||||
#endif
|
||||
LogPrint(HSA_AMD_LOG_FLAG_AQL,
|
||||
"HWq=%p, id=%lu, Dispatch Header = "
|
||||
"0x%x (type=%d, barrier=%d, acquire=%d, release=%d), "
|
||||
|
||||
@@ -47,6 +47,7 @@
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <core/util/utils.h>
|
||||
|
||||
#include "core/inc/amd_gpu_agent.h"
|
||||
#include "core/inc/amd_memory_region.h"
|
||||
@@ -855,7 +856,7 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
|
||||
// width | 16 ensures that we don't return a higher element than is supported and avoids
|
||||
// issues with 0.
|
||||
auto maxAlignedElement = [](size_t width) {
|
||||
return __builtin_ctz(width | 16);
|
||||
return rocr::os::Ctz(width | 16);
|
||||
};
|
||||
|
||||
// GFX12 or later use a different packet format that is incompatible (fields changed in size and location).
|
||||
@@ -872,7 +873,7 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
|
||||
// Find maximum element that describes the pitch and slice.
|
||||
// Pitch and slice must both be represented in units of elements. No element larger than this
|
||||
// may be used in any tile as the pitches would not be exactly represented.
|
||||
int max_ele = Min(maxAlignedElement(src->pitch), maxAlignedElement(dst->pitch));
|
||||
auto max_ele = Min(maxAlignedElement(src->pitch), maxAlignedElement(dst->pitch));
|
||||
if (range->z != 1) // Only need to consider slice if HW will copy along Z.
|
||||
max_ele = Min(max_ele, maxAlignedElement(src->slice), maxAlignedElement(dst->slice));
|
||||
|
||||
@@ -895,8 +896,8 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
|
||||
src and dst base has already been checked for DWORD alignment so we only need to consider the
|
||||
offset here.
|
||||
*/
|
||||
int min_ele = Min(max_ele, maxAlignedElement(range->x), maxAlignedElement(src_offset->x % 4),
|
||||
maxAlignedElement(dst_offset->x % 4));
|
||||
auto min_ele = Min(max_ele, maxAlignedElement(range->x), maxAlignedElement(src_offset->x % 4),
|
||||
maxAlignedElement(dst_offset->x % 4));
|
||||
|
||||
// Check that pitch and slice can be represented in the tile with the smallest element
|
||||
if ((src->pitch >> min_ele) > max_pitch || (dst->pitch >> min_ele) > max_pitch)
|
||||
@@ -916,8 +917,8 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
|
||||
|
||||
// Get largest element which describes the start of this tile after its base address has
|
||||
// been aligned. Base addresses must be DWORD (4 byte) aligned.
|
||||
int aligned_ele = Min(maxAlignedElement((src_offset->x + x) % 4),
|
||||
maxAlignedElement((dst_offset->x + x) % 4), max_ele);
|
||||
auto aligned_ele = Min(maxAlignedElement((src_offset->x + x) % 4),
|
||||
maxAlignedElement((dst_offset->x + x) % 4), max_ele);
|
||||
|
||||
// Get largest permissible element which exactly covers width
|
||||
int element = Min(maxAlignedElement(width), aligned_ele);
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2023, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -1026,8 +1026,8 @@ hsa_status_t GpuAgent::DmaCopy(void* dst, core::Agent& dst_agent,
|
||||
std::vector<core::Signal*>& dep_signals,
|
||||
core::Signal& out_signal) {
|
||||
// Recommended SDMA engine copies only have gang factor 1
|
||||
uint32_t rec_sdma_eng = ffs(rec_sdma_eng_id_peers_info_[dst_agent.public_handle().handle]);
|
||||
|
||||
uint32_t rec_sdma_eng =
|
||||
rocr::os::Ffs(rec_sdma_eng_id_peers_info_[dst_agent.public_handle().handle]);
|
||||
if (rec_sdma_eng)
|
||||
return DmaCopyOnEngine(dst, dst_agent, src, src_agent, size,
|
||||
dep_signals, out_signal, rec_sdma_eng, false);
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -44,12 +44,16 @@
|
||||
#include "core/inc/runtime.h"
|
||||
|
||||
#include <assert.h>
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#include <linux/limits.h>
|
||||
#include <sys/mman.h>
|
||||
#include <stdlib.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
#include <stdlib.h>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
@@ -92,7 +96,7 @@ std::string EncodePathname(const char *file_path) {
|
||||
}
|
||||
|
||||
std::string GetUriFromMemoryAddress(const void *memory, size_t size) {
|
||||
pid_t pid = getpid();
|
||||
int pid = getpid();
|
||||
std::ostringstream uri_stream;
|
||||
uri_stream << "memory://" << pid
|
||||
<< "#offset=0x" << std::hex << (uintptr_t)memory << std::dec
|
||||
@@ -313,23 +317,7 @@ hsa_status_t CodeObjectReaderImpl::SetFile(
|
||||
code_object_size = _code_object_size;
|
||||
is_mmap = true;
|
||||
#else
|
||||
if (__lseek__(_code_object_file_descriptor, 0, SEEK_SET) == (off_t)-1) {
|
||||
return HSA_STATUS_ERROR_INVALID_FILE;
|
||||
}
|
||||
|
||||
std::unique_ptr<unsigned char> memory(new unsigned char[_code_object_size]);
|
||||
if (!memory) {
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
}
|
||||
|
||||
if (__read__(_code_object_file_descriptor, mmap_memory,
|
||||
_code_object_size) != _code_object_size) {
|
||||
return HSA_STATUS_ERROR_INVALID_FILE;
|
||||
}
|
||||
mmap_memory = memory.release();
|
||||
mmap_size = _code_object_size;
|
||||
code_object_memory = memory;
|
||||
code_object_size = _code_object_size;
|
||||
//@todo May need an implementation in Windows
|
||||
#endif // !defined(_WIN32) && !defined(_WIN64)
|
||||
|
||||
uri = GetUriFromFile(_code_object_file_descriptor, _code_object_offset,
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -40,6 +40,11 @@
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
#include "core/inc/amd_memory_region.h"
|
||||
|
||||
#include <algorithm>
|
||||
@@ -48,15 +53,15 @@
|
||||
#include "core/inc/amd_cpu_agent.h"
|
||||
#include "core/inc/amd_gpu_agent.h"
|
||||
#include "core/util/utils.h"
|
||||
#include "core/util/os.h"
|
||||
#include "core/inc/exceptions.h"
|
||||
#include <unistd.h>
|
||||
|
||||
namespace rocr {
|
||||
namespace AMD {
|
||||
|
||||
// Tracks aggregate size of system memory available on platform
|
||||
size_t MemoryRegion::max_sysmem_alloc_size_ = 0;
|
||||
const size_t MemoryRegion::kPageSize_ = sysconf(_SC_PAGESIZE);
|
||||
const size_t MemoryRegion::kPageSize_ = os::PageSize();
|
||||
|
||||
MemoryRegion::MemoryRegion(bool fine_grain, bool kernarg, bool full_profile,
|
||||
bool extended_scope_fine_grain, bool user_visible, core::Agent* owner,
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -58,8 +58,6 @@
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
#include <link.h>
|
||||
|
||||
#include "core/inc/amd_aie_agent.h"
|
||||
#include "core/inc/amd_available_drivers.h"
|
||||
#include "core/inc/amd_cpu_agent.h"
|
||||
@@ -72,7 +70,12 @@
|
||||
#include "core/inc/amd_virtio_driver.h"
|
||||
#endif
|
||||
|
||||
extern r_debug _amdgpu_r_debug;
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#else
|
||||
#include "loader/executable.hpp"
|
||||
#endif
|
||||
extern r_debug _amdgpu_r_debug_r;
|
||||
|
||||
namespace rocr {
|
||||
namespace AMD {
|
||||
@@ -81,17 +84,17 @@ namespace {
|
||||
|
||||
const std::array<std::function<hsa_status_t(std::unique_ptr<core::Driver>&)>,
|
||||
#if _WIN32
|
||||
0
|
||||
1
|
||||
#elif __linux__
|
||||
static_cast<size_t>(core::DriverType::NUM_DRIVER_TYPES)
|
||||
#endif
|
||||
>
|
||||
discover_driver_funcs = {
|
||||
KfdDriver::DiscoverDriver
|
||||
#ifdef __linux__
|
||||
KfdDriver::DiscoverDriver,
|
||||
XdnaDriver::DiscoverDriver,
|
||||
, XdnaDriver::DiscoverDriver
|
||||
#ifdef HSAKMT_VIRTIO_ENABLED
|
||||
KfdVirtioDriver::DiscoverDriver,
|
||||
, KfdVirtioDriver::DiscoverDriver
|
||||
#endif
|
||||
#endif
|
||||
};
|
||||
@@ -181,8 +184,10 @@ GpuAgent* DiscoverGpu(HSAuint32 node_id, HsaNodeProperties& node_prop, bool xnac
|
||||
}
|
||||
|
||||
void DiscoverAie(uint32_t node_id, HsaNodeProperties& node_prop) {
|
||||
#if defined(__linux__)
|
||||
AieAgent* aie = new AieAgent(node_id, node_prop);
|
||||
core::Runtime::runtime_singleton_->RegisterAgent(aie, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
void RegisterLinkInfo(const std::unique_ptr<core::Driver>& driver, uint32_t node_id,
|
||||
|
||||
+10
-2
@@ -3,7 +3,7 @@
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2014-2023, Advanced Micro Devices, Inc. All rights reserved.
|
||||
## Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
@@ -148,11 +148,19 @@ function(generate_bytecodeStrm HeaderFILE)
|
||||
COPYONLY)
|
||||
|
||||
# Add a custom command to generate the header file
|
||||
if (UNIX)
|
||||
add_custom_command(OUTPUT ${HeaderFILE}.h
|
||||
COMMAND ${CMAKE_CURRENT_BINARY_DIR}/create_blit_shader_header.sh ${ARG_LIST} ${HSACO_TARG_LIST}
|
||||
COMMENT "Collating blit shaders..."
|
||||
DEPENDS ${HSACO_TARG_LIST} ${CMAKE_CURRENT_BINARY_DIR}/create_blit_shader_header.sh)
|
||||
|
||||
else()
|
||||
find_package(Python3 COMPONENTS Interpreter REQUIRED)
|
||||
add_custom_command(
|
||||
OUTPUT ${HeaderFILE}.h
|
||||
COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/create_blit_shader_header.py ${ARG_LIST} ${HSACO_TARG_LIST}
|
||||
COMMENT "Collating blit shaders..."
|
||||
DEPENDS ${HSACO_TARG_LIST} create_blit_shader_header.py)
|
||||
endif()
|
||||
# Add a custom target that depends on the header file
|
||||
add_custom_target(${HeaderFILE} DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/${HeaderFILE}.h)
|
||||
|
||||
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
################################################################################
|
||||
##
|
||||
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
|
||||
##
|
||||
## SPDX-License-Identifier: MIT
|
||||
##
|
||||
################################################################################
|
||||
import sys
|
||||
|
||||
def GetSize(fileobject):
|
||||
fileobject.seek(0,2) # move the cursor to the end of the file
|
||||
size = fileobject.tell()
|
||||
return size
|
||||
|
||||
def DumpFile(header, input_name):
|
||||
try:
|
||||
with open(input_name, "rb") as binary_file:
|
||||
# Read the entire content of the file as bytes
|
||||
binary_data = binary_file.read()
|
||||
file_size = GetSize(binary_file)
|
||||
#print(f"Binary size: {file_size}")
|
||||
# Reset file pointer
|
||||
binary_file.seek(0)
|
||||
parts = input_name.split('.')
|
||||
file_name = parts[0]
|
||||
content = f"unsigned char {file_name}""[] = {\n "
|
||||
|
||||
header.write(content)
|
||||
line = 0
|
||||
count = 0
|
||||
for byte_value in binary_data:
|
||||
count += 1
|
||||
padded_hex = '{:02x}'.format(byte_value)
|
||||
if (count != file_size):
|
||||
header.write(f"0x{padded_hex},")
|
||||
else:
|
||||
header.write(f"0x{padded_hex}")
|
||||
line += 1
|
||||
if (line == 12):
|
||||
header.write(f"\n ")
|
||||
line = 0
|
||||
else:
|
||||
header.write(f" ")
|
||||
|
||||
header.write("\n};\nunsigned int "f"{file_name}_len = {file_size};\n")
|
||||
|
||||
except FileNotFoundError:
|
||||
print(f"Error: The file {input_name} was not found.")
|
||||
except Exception as e:
|
||||
print(f"An error occurred: {e}")
|
||||
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
header_name = sys.argv[1];
|
||||
with open(header_name, 'w') as header:
|
||||
header.write("//==============================================================================\n")
|
||||
header.write("// This file is automatically generated during build process, don't modify it\n")
|
||||
header.write("//==============================================================================\n\n")
|
||||
header.write("namespace rocr {\n")
|
||||
header.write("namespace AMD {\n\n")
|
||||
|
||||
for i, arg in enumerate(sys.argv):
|
||||
if (i > 1):
|
||||
#print(f"File {i}: {arg}\n")
|
||||
DumpFile(header, arg)
|
||||
header.write("} // namespace AMD\n")
|
||||
header.write("} // namespace rocr\n\n")
|
||||
|
||||
else:
|
||||
print("Empty arguments!")
|
||||
|
||||
@@ -835,12 +835,14 @@ hsa_status_t hsa_amd_agent_iterate_memory_pools(
|
||||
reinterpret_cast<hsa_status_t (*)(hsa_region_t memory_pool,
|
||||
void *data)>(callback),
|
||||
data);
|
||||
#if defined(__linux__)
|
||||
case core::Agent::kAmdAieDevice:
|
||||
return reinterpret_cast<const AMD::AieAgent *>(agent)->VisitRegion(
|
||||
false,
|
||||
reinterpret_cast<hsa_status_t (*)(hsa_region_t memory_pool,
|
||||
void *data)>(callback),
|
||||
data);
|
||||
#endif
|
||||
case core::Agent::kAmdGpuDevice:
|
||||
return reinterpret_cast<const AMD::GpuAgentInt *>(agent)->VisitRegion(
|
||||
false,
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -42,8 +42,9 @@
|
||||
|
||||
#include "core/inc/hsa_ven_amd_loader_impl.h"
|
||||
|
||||
#include "core/inc/amd_hsa_loader.hpp"
|
||||
#include "core/inc/runtime.h"
|
||||
#include "core/inc/amd_gpu_agent.h"
|
||||
#include "core/inc/amd_hsa_loader.hpp"
|
||||
|
||||
namespace rocr {
|
||||
|
||||
|
||||
@@ -48,13 +48,19 @@
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <list>
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#include <dlfcn.h>
|
||||
#include <amdgpu_drm.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/socket.h>
|
||||
#include <sys/un.h>
|
||||
#else
|
||||
#define debug_warning(__VA_ARGS__)
|
||||
#endif
|
||||
#include <iostream>
|
||||
#include <thread>
|
||||
#include <chrono>
|
||||
|
||||
#include "core/inc/runtime.h"
|
||||
#include "core/inc/hsa_table_interface.h"
|
||||
@@ -97,8 +103,12 @@
|
||||
ROCPROFILER_REGISTER_DEFINE_IMPORT(hsa, ROCP_REG_VERSION)
|
||||
#endif
|
||||
|
||||
#if defined(__linux__)
|
||||
const char rocrbuildid[] __attribute__((used)) = "ROCR BUILD ID: " STRING(ROCR_BUILD_ID);
|
||||
|
||||
#else
|
||||
#include "loader/executable.hpp"
|
||||
const char rocrbuildid[] = "ROCR BUILD ID: " STRING(ROCR_BUILD_ID);
|
||||
#endif
|
||||
extern r_debug _amdgpu_r_debug;
|
||||
|
||||
namespace rocr {
|
||||
@@ -591,7 +601,7 @@ hsa_status_t Runtime::CopyMemoryOnEngine(void* dst, core::Agent* dst_agent, cons
|
||||
core::Agent* copy_agent = (src_gpu) ? src_agent : dst_agent;
|
||||
|
||||
// engine_id is single bitset unique.
|
||||
int engine_offset = ffs(engine_id);
|
||||
int engine_offset = rocr::os::Ffs(engine_id);
|
||||
if (!engine_id || !!((engine_id >> engine_offset))) {
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
}
|
||||
@@ -1178,6 +1188,7 @@ hsa_status_t Runtime::SetPtrInfoData(const void* ptr, void* userptr) {
|
||||
|
||||
// Send the dmabuf_fd to from process via Unix socket
|
||||
static int SendDmaBufFd(int socket, int dmabuf_fd) {
|
||||
#if defined(__linux__)
|
||||
char iov_buf[1];
|
||||
struct msghdr msg = {0};
|
||||
char buf[CMSG_SPACE(sizeof(dmabuf_fd))];
|
||||
@@ -1205,10 +1216,15 @@ static int SendDmaBufFd(int socket, int dmabuf_fd) {
|
||||
ssize_t sent = sendmsg(socket, &msg, 0);
|
||||
|
||||
return (sent < 0) ? -1 : 0;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
// Receive the dmabuf_fd to from process via Unix socket
|
||||
static int ReceiveDmaBufFd(int socket) {
|
||||
#if defined(__linux__)
|
||||
struct msghdr msg = {0};
|
||||
|
||||
// The struct iovec is needed, even if it points to minimal data
|
||||
@@ -1233,6 +1249,10 @@ static int ReceiveDmaBufFd(int socket) {
|
||||
memcpy(&fd, CMSG_DATA(cmsg), sizeof(fd));
|
||||
|
||||
return fd;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
#define IPC_SOCK_SERVER_DMABUF_FD_HANDLE_LENGTH 64
|
||||
@@ -1363,6 +1383,7 @@ hsa_status_t Runtime::IPCCreate(void* ptr, size_t len, hsa_amd_ipc_memory_t* han
|
||||
close(dmabuf_fd);
|
||||
|
||||
ScopedAcquire<KernelMutex> lock(&ipc_sock_server_lock_);
|
||||
#if defined(__linux__)
|
||||
if (!ipc_sock_server_conns_.size()) { // create new runtime socket server
|
||||
struct sockaddr_un address;
|
||||
ipc_sock_server_fd_ = socket(AF_UNIX, SOCK_STREAM, 0);
|
||||
@@ -1393,7 +1414,9 @@ hsa_status_t Runtime::IPCCreate(void* ptr, size_t len, hsa_amd_ipc_memory_t* han
|
||||
// as the attach life cycle is unknown.
|
||||
os::CreateThread(AsyncIPCSockServerConnLoop, NULL);
|
||||
}
|
||||
|
||||
#else
|
||||
assert(!"Unimplemented! Do we really need this?");
|
||||
#endif
|
||||
ipc_sock_server_conns_[reinterpret_cast<uint64_t>(ptr)] = len;
|
||||
|
||||
// TODO: fragment block discard for better memory performance causes memory violations
|
||||
@@ -1406,7 +1429,6 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
|
||||
amdgpu_bo_import_result *res,
|
||||
unsigned int numNodes, HSAuint32 *nodes,
|
||||
void **importAddress, HSAuint64 *importSize) {
|
||||
struct sockaddr_un address;
|
||||
int dmabuf_fd = -1, socket_fd = socket(AF_UNIX, SOCK_STREAM, 0);
|
||||
assert(socket_fd > -1 && "DMA buffer could not be imported for IPC!");
|
||||
if (socket_fd == -1) return -1;
|
||||
@@ -1420,13 +1442,15 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
|
||||
if (status) return -1;
|
||||
|
||||
char buf[IPC_SOCK_SERVER_DMABUF_FD_HANDLE_LENGTH];
|
||||
memset(&address, 0, sizeof(struct sockaddr_un));
|
||||
memset(buf, 0, sizeof(buf));
|
||||
int timeoutLimitMs = 10000, timeoutMs = 0, timeoutIntervalMs = 1;
|
||||
#if defined(__linux__)
|
||||
struct sockaddr_un address;
|
||||
memset(&address, 0, sizeof(struct sockaddr_un));
|
||||
address.sun_family = AF_UNIX;
|
||||
snprintf(address.sun_path, IPC_SOCK_SERVER_NAME_LENGTH, "xhsa%i", conn_handle);
|
||||
address.sun_path[0] = 0; // first NULL char creates unlisted abstract socket
|
||||
|
||||
int timeoutLimitMs = 10000, timeoutMs = 0, timeoutIntervalMs = 1;
|
||||
while (timeoutMs < timeoutLimitMs) {
|
||||
if (connect(socket_fd, (struct sockaddr *) &address, sizeof(struct sockaddr_un))) {
|
||||
timeoutMs += timeoutIntervalMs;
|
||||
@@ -1435,7 +1459,9 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#else
|
||||
assert(!"Unimplmented!");
|
||||
#endif
|
||||
MAKE_SCOPE_GUARD([&]() { close(socket_fd); });
|
||||
|
||||
if (timeoutMs >= timeoutLimitMs) return -1;
|
||||
@@ -1545,6 +1571,7 @@ hsa_status_t Runtime::IPCAttach(const hsa_amd_ipc_memory_t* handle, size_t len,
|
||||
dmaBufFDHandle = (dmaBufFDHandleHi << 32) | dmaBufFDHandleLo;
|
||||
}
|
||||
|
||||
#if defined(__linux__)
|
||||
if (num_agents == 0) {
|
||||
amdgpu_bo_import_result res;
|
||||
bool isDmabufSysMem = ipc_dmabuf_supported_ && importHandle.handle[3];
|
||||
@@ -1575,6 +1602,9 @@ hsa_status_t Runtime::IPCAttach(const hsa_amd_ipc_memory_t* handle, size_t len,
|
||||
*mapped_ptr = importAddress;
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
|
||||
HSAuint32* nodes = nullptr;
|
||||
if (num_agents > tinyArraySize)
|
||||
@@ -1602,6 +1632,7 @@ hsa_status_t Runtime::IPCDetach(void* ptr) {
|
||||
const auto& it = allocation_map_.find(ptr);
|
||||
if (it != allocation_map_.end()) {
|
||||
if (it->second.region != nullptr) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
#if defined(__linux__)
|
||||
if (it->second.ldrm_bo) {
|
||||
if (DRM_CALL(amdgpu_bo_va_op(it->second.ldrm_bo, 0, it->second.size,
|
||||
reinterpret_cast<uint64_t>(ptr), 0, AMDGPU_VA_OP_UNMAP)))
|
||||
@@ -1610,6 +1641,9 @@ hsa_status_t Runtime::IPCDetach(void* ptr) {
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
ldrmImportCleaned = true;
|
||||
}
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
allocation_map_.erase(it);
|
||||
lock.Release(); // Can't hold memory lock when using pointer info.
|
||||
|
||||
@@ -2325,6 +2359,7 @@ int fn_amdgpu_device_get_fd_nosupport(HsaAMDGPUDeviceHandle device_handle) {
|
||||
|
||||
int Runtime::GetAmdgpuDeviceArgs(Agent *agent, ShareableHandle handle,
|
||||
int *drm_fd, uint64_t *cpu_addr) {
|
||||
#if defined(__linux__)
|
||||
int renderFd = fn_amdgpu_device_get_fd(static_cast<AMD::GpuAgent*>(agent)->libDrmDev());
|
||||
if (renderFd < 0) return HSA_STATUS_ERROR;
|
||||
|
||||
@@ -2343,6 +2378,9 @@ int Runtime::GetAmdgpuDeviceArgs(Agent *agent, ShareableHandle handle,
|
||||
|
||||
*drm_fd = renderFd;
|
||||
*cpu_addr = args.out.addr_ptr;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -2353,6 +2391,7 @@ void Runtime::CheckVirtualMemApiSupport() {
|
||||
if (kfd_version.KernelInterfaceMajorVersion > 1 ||
|
||||
(kfd_version.KernelInterfaceMajorVersion == 1 &&
|
||||
kfd_version.KernelInterfaceMinorVersion >= 15)) {
|
||||
#if defined(__linux__)
|
||||
char* error;
|
||||
|
||||
fn_amdgpu_device_get_fd =
|
||||
@@ -2365,6 +2404,9 @@ void Runtime::CheckVirtualMemApiSupport() {
|
||||
} else {
|
||||
virtual_mem_api_supported_ = true;
|
||||
}
|
||||
#else
|
||||
virtual_mem_api_supported_ = false;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2379,7 +2421,7 @@ void Runtime::InitIPCDmaBufSupport() {
|
||||
|
||||
GetSystemInfo(HSA_AMD_SYSTEM_INFO_DMABUF_SUPPORTED, &dmabuf_supported);
|
||||
if (!dmabuf_supported) return;
|
||||
|
||||
#if defined(__linux__)
|
||||
char* error;
|
||||
fn_amdgpu_device_get_fd =
|
||||
(int (*)(HsaAMDGPUDeviceHandle device_handle))dlsym(
|
||||
@@ -2391,6 +2433,9 @@ void Runtime::InitIPCDmaBufSupport() {
|
||||
} else {
|
||||
ipc_dmabuf_supported_ = !flag().enable_ipc_mode_legacy();
|
||||
}
|
||||
#else
|
||||
ipc_dmabuf_supported_ = false;
|
||||
#endif
|
||||
}
|
||||
|
||||
void Runtime::LoadTools() {
|
||||
@@ -3237,15 +3282,14 @@ hsa_status_t Runtime::VMemoryAddressReserve(void** va, size_t size, uint64_t add
|
||||
void* addr = (void*)address;
|
||||
HsaMemFlags memFlags = {};
|
||||
|
||||
if (!alignment)
|
||||
alignment = sysconf(_SC_PAGE_SIZE);
|
||||
if (!alignment) alignment = rocr::os::PageSize();
|
||||
|
||||
ScopedAcquire<KernelSharedMutex> lock(&memory_lock_);
|
||||
|
||||
if (flags & HSA_AMD_VMEM_ADDRESS_NO_REGISTER) {
|
||||
size_t requested = size + alignment - sysconf(_SC_PAGE_SIZE);
|
||||
auto mem = mmap(addr, requested, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE | MAP_NORESERVE, -1, 0);
|
||||
if (mem == MAP_FAILED)
|
||||
size_t requested = size + alignment - rocr::os::PageSize();
|
||||
auto mem = rocr::os::ReserveMemory(addr, requested, alignment, rocr::os::MEM_PROT_RW);
|
||||
if (mem == nullptr)
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
|
||||
auto aligned = AlignUp(mem, alignment);
|
||||
@@ -3253,8 +3297,10 @@ hsa_status_t Runtime::VMemoryAddressReserve(void** va, size_t size, uint64_t add
|
||||
// Hint to enable THP for large host allocations which can help in performance gain
|
||||
constexpr size_t kLargePageSize = 2*1024*1024;
|
||||
if (size >= kLargePageSize) {
|
||||
#if defined(__linux__)
|
||||
if (madvise(aligned, size, MADV_HUGEPAGE))
|
||||
debug_warning(false && "madvise with MADV_HUGEPAGE failed");
|
||||
#endif
|
||||
}
|
||||
|
||||
reserved_address_map_[aligned] = AddressHandle(mem, size, false);
|
||||
@@ -3292,10 +3338,11 @@ hsa_status_t Runtime::VMemoryAddressFree(void* va, size_t size) {
|
||||
if (it->second.use_count > 0) return HSA_STATUS_ERROR_RESOURCE_FREE;
|
||||
|
||||
if (it->second.registered) {
|
||||
if (HSAKMT_CALL(hsaKmtFreeMemory(it->second.os_addr, size)) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
|
||||
} else {
|
||||
if (munmap(it->second.os_addr, size)) return HSA_STATUS_ERROR;
|
||||
if (HSAKMT_CALL(hsaKmtFreeMemory(it->second.os_addr, size)) != HSAKMT_STATUS_SUCCESS)
|
||||
return HSA_STATUS_ERROR;
|
||||
}
|
||||
else if (!rocr::os::ReleaseMemory(it->second.os_addr, size))
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
reserved_address_map_.erase(it);
|
||||
return HSA_STATUS_SUCCESS;
|
||||
@@ -3488,10 +3535,10 @@ hsa_status_t Runtime::VMemoryHandleUnmap(void* va, size_t size) {
|
||||
}
|
||||
|
||||
Runtime::MappedHandleAllowedAgent::MappedHandleAllowedAgent(
|
||||
MappedHandle *mappedHandle, Agent *targetAgent, void *va, size_t size,
|
||||
MappedHandle* _mappedHandle, Agent *targetAgent, void *va, size_t size,
|
||||
hsa_access_permission_t perms)
|
||||
: va(va), size(size), targetAgent(targetAgent), permissions(perms),
|
||||
mappedHandle(mappedHandle) {
|
||||
mappedHandle(_mappedHandle) {
|
||||
|
||||
// CPU agents have access as the memory is already mapped to the host.
|
||||
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) return;
|
||||
@@ -3527,6 +3574,7 @@ Runtime::MappedHandleAllowedAgent::~MappedHandleAllowedAgent() {
|
||||
|
||||
hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permission_t perms) {
|
||||
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) {
|
||||
#if defined(__linux__)
|
||||
void* mapped_ptr =
|
||||
mmap(va, size, PermissionsToMmapFlags(perms), MAP_SHARED | MAP_FIXED, mappedHandle->drm_fd,
|
||||
reinterpret_cast<uint64_t>(mappedHandle->drm_cpu_addr));
|
||||
@@ -3537,6 +3585,9 @@ hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permissi
|
||||
shareable_handle, va, mappedHandle->offset, size, perms);
|
||||
if (status != HSA_STATUS_SUCCESS)
|
||||
return status;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
}
|
||||
permissions = perms;
|
||||
return HSA_STATUS_SUCCESS;
|
||||
@@ -3544,8 +3595,12 @@ hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permissi
|
||||
|
||||
hsa_status_t Runtime::MappedHandleAllowedAgent::RemoveAccess() {
|
||||
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) {
|
||||
#if defined(__linux__)
|
||||
if (munmap(va, size) != 0)
|
||||
return HSA_STATUS_ERROR;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
} else {
|
||||
return targetAgent->driver().Unmap(
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -51,6 +51,9 @@
|
||||
|
||||
#include "core/util/timer.h"
|
||||
#include "core/inc/runtime.h"
|
||||
#if defined(_WIN32)
|
||||
#include "malloc.h"
|
||||
#endif
|
||||
|
||||
namespace rocr {
|
||||
namespace core {
|
||||
@@ -234,8 +237,11 @@ uint32_t Signal::WaitMultiple(uint32_t signal_count, const hsa_signal_t* hsa_sig
|
||||
MAKE_SCOPE_GUARD([&]() {
|
||||
if (signal_count > small_size) delete[] evts;
|
||||
});
|
||||
|
||||
#if defined(__linux__)
|
||||
uint64_t event_age[unique_evts];
|
||||
#else
|
||||
auto event_age = reinterpret_cast<uint64_t*>(_alloca(unique_evts * sizeof(unique_evts)));
|
||||
#endif
|
||||
memset(event_age, 0, unique_evts * sizeof(uint64_t));
|
||||
if (core::Runtime::runtime_singleton_->KfdVersion().supports_event_age)
|
||||
for (uint32_t i = 0; i < unique_evts; i++)
|
||||
@@ -367,8 +373,11 @@ uint32_t Signal::WaitAnyExceptions(uint32_t signal_count, const hsa_signal_t* hs
|
||||
std::sort(evts, evts + signal_count);
|
||||
HsaEvent** end = std::unique(evts, evts + signal_count);
|
||||
unique_evts = uint32_t(end - evts);
|
||||
|
||||
#if defined(__linux__)
|
||||
uint64_t event_age[unique_evts];
|
||||
#else
|
||||
auto event_age = reinterpret_cast<uint64_t*>(_alloca(unique_evts * sizeof(unique_evts)));
|
||||
#endif
|
||||
memset(event_age, 0, unique_evts * sizeof(uint64_t));
|
||||
if (core::Runtime::runtime_singleton_->KfdVersion().supports_event_age)
|
||||
for (uint32_t i = 0; i < unique_evts; i++)
|
||||
|
||||
@@ -44,8 +44,17 @@
|
||||
|
||||
#include <stdint.h>
|
||||
#include <algorithm>
|
||||
#if defined(__linux__)
|
||||
#include <sys/eventfd.h>
|
||||
#include <poll.h>
|
||||
#else
|
||||
struct pollfd {
|
||||
int fd;
|
||||
short int events;
|
||||
short int revents;
|
||||
};
|
||||
#define POLLIN 0x001 // from poll.h...
|
||||
#endif
|
||||
|
||||
#include "core/util/utils.h"
|
||||
#include "core/inc/runtime.h"
|
||||
@@ -171,7 +180,12 @@ void SvmProfileControl::PollSmi() {
|
||||
};
|
||||
|
||||
while (!exit) {
|
||||
#if defined(__linux__)
|
||||
int ready = poll(&files[0], files.size(), -1);
|
||||
#else
|
||||
int ready = 0;
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
if (ready < 1) {
|
||||
assert(false && "poll failed!");
|
||||
return;
|
||||
@@ -345,9 +359,10 @@ void SvmProfileControl::PollSmi() {
|
||||
}
|
||||
|
||||
SvmProfileControl::SvmProfileControl() : event(-1), exit(false) {
|
||||
#if defined(__linux__)
|
||||
event = eventfd(0, EFD_CLOEXEC);
|
||||
if (event == -1) return;
|
||||
|
||||
#endif
|
||||
poll_smi_thread_ = os::CreateThread(PollSmiRun, (void*)this);
|
||||
if (poll_smi_thread_ == NULL) {
|
||||
assert(false && "Poll SMI thread creation error.");
|
||||
@@ -356,10 +371,12 @@ SvmProfileControl::SvmProfileControl() : event(-1), exit(false) {
|
||||
}
|
||||
|
||||
SvmProfileControl::~SvmProfileControl() {
|
||||
#if defined(__linux__)
|
||||
if (event != -1) {
|
||||
eventfd_write(event, 1);
|
||||
close(event);
|
||||
}
|
||||
#endif
|
||||
if (poll_smi_thread_ != NULL) {
|
||||
exit = true;
|
||||
os::WaitForThread(poll_smi_thread_);
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -43,9 +43,12 @@
|
||||
#include "core/inc/thunk_loader.h"
|
||||
#include "core/inc/runtime.h"
|
||||
|
||||
#include <dlfcn.h>
|
||||
#include <core/util/os.h>
|
||||
#include <iostream>
|
||||
#if defined(__linux__)
|
||||
#include <dlfcn.h>
|
||||
#include <fcntl.h>
|
||||
#endif
|
||||
|
||||
namespace rocr {
|
||||
namespace core {
|
||||
@@ -57,6 +60,7 @@ namespace core {
|
||||
return "libdtif.so";
|
||||
}
|
||||
|
||||
#if defined(__linux__)
|
||||
if (core::Runtime::runtime_singleton_->flag().enable_dxg_detection()) {
|
||||
int fd = open("/dev/dxg", O_RDWR);
|
||||
if (fd >= 0) {
|
||||
@@ -65,6 +69,9 @@ namespace core {
|
||||
return "librocdxg.so";
|
||||
}
|
||||
}
|
||||
#else
|
||||
is_dxg_ = true;
|
||||
#endif
|
||||
|
||||
return "";
|
||||
}
|
||||
@@ -74,10 +81,10 @@ namespace core {
|
||||
library_name(whoami()),
|
||||
is_loaded_(false) {
|
||||
if (!library_name.empty()) {
|
||||
dlerror(); // Clear any existing error messages
|
||||
thunk_handle = dlopen(library_name.c_str(), RTLD_LAZY);
|
||||
rocr::os::DlError(); // Clear any existing error messages
|
||||
thunk_handle = rocr::os::LoadLib(library_name.c_str());
|
||||
if (thunk_handle == NULL) {
|
||||
fprintf(stderr, "Cannot load %s, failed:%s\n", library_name.c_str(), dlerror());
|
||||
fprintf(stderr, "Cannot load %s, failed:%s\n", library_name.c_str(), rocr::os::DlError());
|
||||
} else {
|
||||
debug_print("Load %s successully!\n", library_name.c_str());
|
||||
}
|
||||
@@ -88,8 +95,8 @@ namespace core {
|
||||
ThunkLoader::~ThunkLoader() {
|
||||
if (IsSharedLibraryLoaded()
|
||||
&& (thunk_handle != NULL)) {
|
||||
if (dlclose(thunk_handle) != 0) {
|
||||
fprintf(stderr, "Cannot unload %s, failed:%s\n", library_name.c_str(), dlerror());
|
||||
if (!rocr::os::CloseLib(thunk_handle)) {
|
||||
fprintf(stderr, "Cannot unload %s, failed:%s\n", library_name.c_str(), rocr::os::DlError());
|
||||
} else {
|
||||
debug_print("Unload %s successully!\n", library_name.c_str());
|
||||
}
|
||||
@@ -98,6 +105,7 @@ namespace core {
|
||||
|
||||
void ThunkLoader::LoadThunkApiTable() {
|
||||
if (IsSharedLibraryLoaded()) {
|
||||
#if defined(__linux__)
|
||||
dlerror(); // Clear any existing error messages
|
||||
|
||||
HSAKMT_PFN(hsaKmtOpenKFD) = (HSAKMT_DEF(hsaKmtOpenKFD)*)dlsym(thunk_handle, "hsaKmtOpenKFD");
|
||||
@@ -402,12 +410,12 @@ namespace core {
|
||||
|
||||
DRM_PFN(drmCommandWriteRead) = (DRM_DEF(drmCommandWriteRead)*)dlsym(thunk_handle, "drmCommandWriteRead");
|
||||
if (DRM_PFN(drmCommandWriteRead) == NULL) goto ERROR;
|
||||
|
||||
debug_print("Load all DTIF APIs OK!\n");
|
||||
return;
|
||||
|
||||
ERROR:
|
||||
fprintf(stderr, "dlsym failed: %s\n", dlerror());
|
||||
#endif
|
||||
} else {
|
||||
HSAKMT_PFN(hsaKmtOpenKFD) = (HSAKMT_DEF(hsaKmtOpenKFD)*)(&hsaKmtOpenKFD);
|
||||
HSAKMT_PFN(hsaKmtCloseKFD) = (HSAKMT_DEF(hsaKmtCloseKFD)*)(&hsaKmtCloseKFD);
|
||||
@@ -499,6 +507,9 @@ ERROR:
|
||||
HSAKMT_PFN(hsaKmtPcSamplingStart) = (HSAKMT_DEF(hsaKmtPcSamplingStart)*)(&hsaKmtPcSamplingStart);
|
||||
HSAKMT_PFN(hsaKmtPcSamplingStop) = (HSAKMT_DEF(hsaKmtPcSamplingStop)*)(&hsaKmtPcSamplingStop);
|
||||
HSAKMT_PFN(hsaKmtPcSamplingSupport) = (HSAKMT_DEF(hsaKmtPcSamplingSupport)*)(&hsaKmtPcSamplingSupport);
|
||||
#if defined(_WIN32)
|
||||
HSAKMT_PFN(hsaKmtQueueRingDoorbell) = (HSAKMT_DEF(hsaKmtQueueRingDoorbell)*)(&hsaKmtQueueRingDoorbell);
|
||||
#endif
|
||||
HSAKMT_PFN(hsaKmtModelEnabled) = (HSAKMT_DEF(hsaKmtModelEnabled)*)(&hsaKmtModelEnabled);
|
||||
|
||||
DRM_PFN(amdgpu_device_initialize) = (DRM_DEF(amdgpu_device_initialize)*)(&amdgpu_device_initialize);
|
||||
@@ -509,7 +520,9 @@ ERROR:
|
||||
DRM_PFN(amdgpu_bo_export) = (DRM_DEF(amdgpu_bo_export)*)(&amdgpu_bo_export);
|
||||
DRM_PFN(amdgpu_bo_import) = (DRM_DEF(amdgpu_bo_import)*)(&amdgpu_bo_import);
|
||||
DRM_PFN(amdgpu_bo_va_op) = (DRM_DEF(amdgpu_bo_va_op)*)(&amdgpu_bo_va_op);
|
||||
#if defined(__linux__)
|
||||
DRM_PFN(drmCommandWriteRead) = (DRM_DEF(drmCommandWriteRead)*)(&drmCommandWriteRead);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -517,7 +530,8 @@ ERROR:
|
||||
if (!IsDTIF())
|
||||
return true;
|
||||
|
||||
DtifCreateFunc* pfnDtifCreate = (DtifCreateFunc*)dlsym(thunk_handle, "DtifCreate");
|
||||
DtifCreateFunc* pfnDtifCreate =
|
||||
(DtifCreateFunc*)rocr::os::GetExportAddress(thunk_handle, "DtifCreate");
|
||||
if (pfnDtifCreate != NULL) {
|
||||
if (pfnDtifCreate("HSA") != NULL) {
|
||||
debug_print("DtifCreate OK!\n");
|
||||
@@ -537,7 +551,8 @@ ERROR:
|
||||
if (thunk_handle == NULL)
|
||||
return false;
|
||||
|
||||
DtifDestroyFunc* pfnDtifDestroy = (DtifDestroyFunc*)dlsym(thunk_handle, "DtifDestroy");
|
||||
DtifDestroyFunc* pfnDtifDestroy =
|
||||
(DtifDestroyFunc*)rocr::os::GetExportAddress(thunk_handle, "DtifDestroy");
|
||||
if (pfnDtifDestroy != NULL) {
|
||||
pfnDtifDestroy();
|
||||
debug_print("DtifDestroy OK!\n");
|
||||
|
||||
+12
-1
@@ -3,7 +3,7 @@
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2022, Advanced Micro Devices, Inc. All rights reserved.
|
||||
## Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
@@ -139,10 +139,21 @@ function(generate_bytecodeStrm HeaderFILE)
|
||||
|
||||
## Add a custom command that generates amd_trap_handler_v2.h
|
||||
## This depends on all the generated code object files and the C++ generator script.
|
||||
|
||||
if (UNIX)
|
||||
add_custom_command(OUTPUT ${HeaderFILE}.h
|
||||
COMMAND ${CMAKE_CURRENT_SOURCE_DIR}/create_trap_handler_header.sh ${ARG_LIST}
|
||||
COMMENT "Collating trap handlers..."
|
||||
DEPENDS ${HSACO_TARG_LIST} create_trap_handler_header.sh )
|
||||
else()
|
||||
find_package(Python3 COMPONENTS Interpreter REQUIRED)
|
||||
add_custom_command(
|
||||
OUTPUT ${HeaderFILE}.h
|
||||
COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/create_trap_handler_header.py ${ARG_LIST}
|
||||
COMMENT "Collating blit shaders..."
|
||||
DEPENDS ${HSACO_TARG_LIST} create_trap_handler_header.py)
|
||||
endif()
|
||||
|
||||
|
||||
## Export a target that builds (and depends on) amd_trap_handler_v2.h
|
||||
add_custom_target( ${HeaderFILE} DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/${HeaderFILE}.h )
|
||||
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
################################################################################
|
||||
##
|
||||
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
|
||||
##
|
||||
## SPDX-License-Identifier: MIT
|
||||
##
|
||||
################################################################################
|
||||
import sys
|
||||
|
||||
def GetSize(fileobject):
|
||||
fileobject.seek(0,2) # move the cursor to the end of the file
|
||||
size = fileobject.tell()
|
||||
return size
|
||||
|
||||
def DumpFile(header, input_name):
|
||||
try:
|
||||
with open(input_name, "rb") as binary_file:
|
||||
# Read the entire content of the file as bytes
|
||||
binary_data = binary_file.read()
|
||||
file_size = GetSize(binary_file)
|
||||
#print(f"Binary size: {file_size}")
|
||||
# Reset file pointer
|
||||
binary_file.seek(0)
|
||||
parts = input_name.split('.')
|
||||
file_name = parts[0]
|
||||
content = f"unsigned char {file_name}""[] = {\n "
|
||||
|
||||
header.write(content)
|
||||
line = 0
|
||||
count = 0
|
||||
for byte_value in binary_data:
|
||||
count += 1
|
||||
padded_hex = '{:02x}'.format(byte_value)
|
||||
if (count != file_size):
|
||||
header.write(f"0x{padded_hex},")
|
||||
else:
|
||||
header.write(f"0x{padded_hex}")
|
||||
line += 1
|
||||
if (line == 12):
|
||||
header.write(f"\n ")
|
||||
line = 0
|
||||
else:
|
||||
header.write(f" ")
|
||||
|
||||
header.write("\n};\nunsigned int "f"{file_name}_len = {file_size};\n")
|
||||
|
||||
except FileNotFoundError:
|
||||
print(f"Error: The file {input_name} was not found.")
|
||||
except Exception as e:
|
||||
print(f"An error occurred: {e}")
|
||||
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
header_name = sys.argv[1];
|
||||
with open(header_name, 'w') as header:
|
||||
header.write("//==============================================================================\n")
|
||||
header.write("// This file is automatically generated during build process, don't modify it\n")
|
||||
header.write("//==============================================================================\n\n")
|
||||
header.write("namespace rocr {\n")
|
||||
header.write("namespace AMD {\n\n")
|
||||
|
||||
for i, arg in enumerate(sys.argv):
|
||||
if (i > 1):
|
||||
#print(f"File {i}: {arg}\n")
|
||||
DumpFile(header, arg)
|
||||
header.write("} // namespace AMD\n")
|
||||
header.write("} // namespace rocr\n\n")
|
||||
|
||||
else:
|
||||
print("Empty arguments!")
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -48,6 +48,133 @@
|
||||
#ifndef HSA_RUNTIME_CORE_UTIL_ATOMIC_HELPERS_H_
|
||||
#define HSA_RUNTIME_CORE_UTIL_ATOMIC_HELPERS_H_
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define WIN32_NO_STATUS
|
||||
#include <Windows.h>
|
||||
#undef WIN32_NO_STATUS
|
||||
|
||||
template <class T>
|
||||
void __atomic_load(const T* object, typename std::remove_volatile<T>::type* ret, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
*ret = InterlockedOr64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
0);
|
||||
} else {
|
||||
*ret = InterlockedOr(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
0);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
void __atomic_store(const T* object, typename std::remove_volatile<T>::type* val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
InterlockedExchange64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
*val);
|
||||
} else {
|
||||
InterlockedExchange(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
*val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_or(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedOr64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
} else {
|
||||
return InterlockedOr(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_and(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedAnd64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
} else {
|
||||
return InterlockedAnd(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_xor(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedXor64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
} else {
|
||||
return InterlockedXor(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_add(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedExchangeAdd64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
} else {
|
||||
return InterlockedExchangeAdd(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_sub(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedExchangeAdd64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val * (-1));
|
||||
} else {
|
||||
return InterlockedExchangeAdd(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val * (-1));
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
void __atomic_compare_exchange(
|
||||
T* object, typename std::remove_volatile<T>::type* expected,
|
||||
typename std::remove_volatile<T>::type* val, int arg0, int arg1, int arg2) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
InterlockedCompareExchange64(reinterpret_cast<volatile LONG64*>(object),
|
||||
*val, *expected);
|
||||
} else {
|
||||
InterlockedCompareExchange(reinterpret_cast<volatile LONG*>(object),
|
||||
*val, *expected);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
void __atomic_exchange(T* object, typename std::remove_volatile<T>::type* val,
|
||||
typename std::remove_volatile<T>::type* ret, int arg0) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
*ret = InterlockedExchange64(reinterpret_cast<volatile LONG64*>(object), *val);
|
||||
} else {
|
||||
*ret = InterlockedExchange(reinterpret_cast<volatile LONG*>(object), *val);
|
||||
}
|
||||
}
|
||||
|
||||
#define __ATOMIC_RELAXED 0
|
||||
#endif
|
||||
|
||||
#include <atomic>
|
||||
|
||||
//ALWAYS_CONSERVATIVE will very likely overfence your code.
|
||||
@@ -145,14 +272,18 @@ static __forceinline void Fence(std::memory_order order=std::memory_order_seq_cs
|
||||
|
||||
template <class T>
|
||||
static __forceinline void BasicCheck(const T* ptr) {
|
||||
#if defined(__linux__)
|
||||
constexpr bool value = __atomic_always_lock_free(sizeof(T), 0);
|
||||
static_assert(value, "Atomic type may not be compatible with peripheral atomics.");
|
||||
#endif
|
||||
};
|
||||
|
||||
template <class T>
|
||||
static __forceinline void BasicCheck(const volatile T* ptr) {
|
||||
#if defined(__linux__)
|
||||
constexpr bool value = __atomic_always_lock_free(sizeof(T), 0);
|
||||
static_assert(value, "Atomic type may not be compatible with peripheral atomics.");
|
||||
#endif
|
||||
};
|
||||
|
||||
/// @brief: Load value of type T atomically with specified memory order.
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -61,6 +61,7 @@
|
||||
#include <utility>
|
||||
#include <semaphore.h>
|
||||
#include "core/inc/runtime.h"
|
||||
#include <sys/mman.h>
|
||||
#if defined(__i386__) || defined(__x86_64__)
|
||||
#include <cpuid.h>
|
||||
#endif
|
||||
@@ -294,7 +295,7 @@ void* GetExportAddress(LibHandle lib, std::string export_name) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void CloseLib(LibHandle lib) { dlclose(*(void**)&lib); }
|
||||
bool CloseLib(LibHandle lib) { return (dlclose(*(void**)&lib) == 0) ? true : false; }
|
||||
|
||||
/*
|
||||
* @brief Look for a symbol called "HSA_AMD_TOOL_PRIORITY" across all loaded
|
||||
@@ -579,7 +580,7 @@ int WaitForOsEvent(EventHandle event, unsigned int milli_seconds) {
|
||||
}
|
||||
|
||||
int ret_code = 0;
|
||||
|
||||
|
||||
if (!eventDescrp->state) {
|
||||
if (milli_seconds == 0) {
|
||||
ret_code = 1;
|
||||
@@ -816,6 +817,124 @@ bool ParseCpuID(cpuid_t* cpuinfo) {
|
||||
#endif
|
||||
}
|
||||
|
||||
uint64_t TimeNanos() {
|
||||
struct timespec tp;
|
||||
::clock_gettime(CLOCK_MONOTONIC, &tp);
|
||||
return (uint64_t)tp.tv_sec * (1000ULL * 1000ULL * 1000ULL) + (uint64_t)tp.tv_nsec;
|
||||
}
|
||||
|
||||
static inline int MemProtToOsProt(MemProt prot) {
|
||||
switch (prot) {
|
||||
case MEM_PROT_NONE:
|
||||
return PROT_NONE;
|
||||
case MEM_PROT_READ:
|
||||
return PROT_READ;
|
||||
case MEM_PROT_RW:
|
||||
return PROT_READ | PROT_WRITE;
|
||||
case MEM_PROT_RWX:
|
||||
return PROT_READ | PROT_WRITE | PROT_EXEC;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
size_t PageSize() {
|
||||
static size_t g_page_size_ = 0; //!< The default os page size
|
||||
if (g_page_size_ == 0) {
|
||||
g_page_size_ = (size_t)::sysconf(_SC_PAGESIZE);
|
||||
}
|
||||
return g_page_size_;
|
||||
}
|
||||
|
||||
void* ReserveMemory(void* start, size_t size, size_t alignment, MemProt prot) {
|
||||
size = AlignUp(size, PageSize());
|
||||
// check for invalid input size
|
||||
if (size == 0) {
|
||||
return NULL;
|
||||
}
|
||||
alignment = std::max(PageSize(), AlignUp(alignment, PageSize()));
|
||||
assert(IsPowerOfTwo(alignment) && "not a power of 2");
|
||||
|
||||
size_t requested = size + alignment - PageSize();
|
||||
address mem = (address)::mmap(start, requested, MemProtToOsProt(prot),
|
||||
MAP_PRIVATE | MAP_NORESERVE | MAP_ANONYMOUS, 0, 0);
|
||||
|
||||
// check for out of memory
|
||||
if (mem == MAP_FAILED) return NULL;
|
||||
|
||||
address aligned = AlignUp(mem, alignment);
|
||||
|
||||
// return the unused leading pages to the free state
|
||||
if (&aligned[0] != &mem[0]) {
|
||||
assert(&aligned[0] > &mem[0] && "check this code");
|
||||
if (::munmap(&mem[0], &aligned[0] - &mem[0]) != 0) {
|
||||
assert(!"::munmap failed");
|
||||
}
|
||||
}
|
||||
// return the unused trailing pages to the free state
|
||||
if (&aligned[size] != &mem[requested]) {
|
||||
assert(&aligned[size] < &mem[requested] && "check this code");
|
||||
if (::munmap(&aligned[size], &mem[requested] - &aligned[size]) != 0) {
|
||||
assert(!"::munmap failed");
|
||||
}
|
||||
}
|
||||
|
||||
// Hint to enable THP for large host allocations which can help in performance gain
|
||||
constexpr size_t kLargePageSize = 2 * 1024 * 1024;
|
||||
if (size >= kLargePageSize) {
|
||||
int status = madvise(aligned, size, MADV_HUGEPAGE);
|
||||
if (status) {
|
||||
LogPrint(HSA_AMD_LOG_FLAG_INFO,
|
||||
"madvise with advice MADV_HUGEPAGE"
|
||||
" starting at address %p and page size 0x%zx, returned %d, errno: %s",
|
||||
aligned, size, status, strerror(errno));
|
||||
}
|
||||
}
|
||||
|
||||
return aligned;
|
||||
}
|
||||
|
||||
bool ReleaseMemory(void* addr, size_t size) {
|
||||
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
|
||||
size = AlignUp(size, PageSize());
|
||||
|
||||
return 0 == ::munmap(addr, size);
|
||||
}
|
||||
|
||||
bool CommitMemory(void* addr, size_t size, MemProt prot) {
|
||||
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
|
||||
size = AlignUp(size, PageSize());
|
||||
|
||||
return ::mmap(addr, size, MemProtToOsProt(prot), MAP_PRIVATE | MAP_FIXED | MAP_ANONYMOUS, -1,
|
||||
0) != MAP_FAILED;
|
||||
}
|
||||
|
||||
bool UncommitMemory(void* addr, size_t size) {
|
||||
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
|
||||
size = AlignUp(size, PageSize());
|
||||
|
||||
return ::mmap(addr, size, PROT_NONE, MAP_PRIVATE | MAP_FIXED | MAP_NORESERVE | MAP_ANONYMOUS, -1,
|
||||
0) != MAP_FAILED;
|
||||
}
|
||||
|
||||
uint64_t HostTotalPhysicalMemory() {
|
||||
static uint64_t totalPhys = 0;
|
||||
|
||||
if (totalPhys != 0) {
|
||||
return totalPhys;
|
||||
}
|
||||
|
||||
totalPhys = sysconf(_SC_PAGESIZE) * sysconf(_SC_PHYS_PAGES);
|
||||
return totalPhys;
|
||||
}
|
||||
|
||||
int Ffs(int i) { return ffs(i); }
|
||||
|
||||
int Ctz(uint64_t i) { return __builtin_ctz(i); }
|
||||
|
||||
char* DlError() { return dlerror(); }
|
||||
|
||||
} // namespace os
|
||||
} // namespace rocr
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -91,7 +91,7 @@ void* GetExportAddress(LibHandle lib, std::string export_name);
|
||||
|
||||
/// @brief: Unloads the dynamic library.
|
||||
/// @param: lib(Input), library handle which will be unloaded.
|
||||
void CloseLib(LibHandle lib);
|
||||
bool CloseLib(LibHandle lib);
|
||||
|
||||
/// @brief: Lists loaded tool libraries that contain
|
||||
/// symbol HSA_AMD_TOOL_PRIORITY
|
||||
@@ -106,6 +106,7 @@ std::string GetLibraryName(LibHandle lib);
|
||||
/// @brief: Creates a Semaphore, will return NULL if failed.
|
||||
/// @param: void.
|
||||
/// @return: Semaphore.
|
||||
#undef CreateSemaphore
|
||||
Semaphore CreateSemaphore();
|
||||
|
||||
/// @brief: Waits for the semaphore. This is a blocking wait.
|
||||
@@ -127,6 +128,7 @@ void DestroySemaphore(Semaphore sem);
|
||||
/// @brief: Creates a mutex, will return NULL if failed.
|
||||
/// @param: void.
|
||||
/// @return: Mutex.
|
||||
#undef CreateMutex
|
||||
Mutex CreateMutex();
|
||||
|
||||
/// @brief: Tries to acquire the mutex once, if successed, return true.
|
||||
@@ -319,15 +321,48 @@ uint64_t ReadSystemClock();
|
||||
/// @brief read the system clock frequency
|
||||
uint64_t SystemClockFrequency();
|
||||
|
||||
typedef struct cpuid_s {
|
||||
struct cpuid_t {
|
||||
char ManufacturerID[13]; // 12 char, NULL terminated
|
||||
bool mwaitx;
|
||||
} cpuid_t;
|
||||
};
|
||||
|
||||
/// @brief parse CPUID
|
||||
/// @param: cpuinfo struct to be filled
|
||||
bool ParseCpuID(cpuid_t* cpuinfo);
|
||||
|
||||
//! Return the default os page size.
|
||||
size_t PageSize();
|
||||
|
||||
/// @brief CPU time in nanoseconds
|
||||
/// @param: None
|
||||
uint64_t TimeNanos();
|
||||
|
||||
using address = char*;
|
||||
enum MemProt { MEM_PROT_NONE = 0, MEM_PROT_READ, MEM_PROT_RW, MEM_PROT_RWX };
|
||||
|
||||
/// @brief Reserves a chunk of memory (priv | anon | noreserve)
|
||||
/// @param:
|
||||
void* ReserveMemory(void* start, size_t size, size_t alignment = 0,
|
||||
MemProt prot = MEM_PROT_NONE);
|
||||
|
||||
/// Release a chunk of memory reserved with reserveMemory.
|
||||
bool ReleaseMemory(void* addr, size_t size);
|
||||
/// Commit a chunk of memory previously reserved with reserveMemory.
|
||||
bool CommitMemory(void* addr, size_t size, MemProt prot = MEM_PROT_NONE);
|
||||
/// Uncommit a chunk of memory previously committed with commitMemory.
|
||||
bool UncommitMemory(void* addr, size_t size);
|
||||
|
||||
uint64_t HostTotalPhysicalMemory();
|
||||
|
||||
/// Find First Set for any OS
|
||||
int Ffs(int i);
|
||||
|
||||
/// Find the count of leading zeros
|
||||
int Ctz(uint64_t i);
|
||||
|
||||
/// Shared library or DLL load error
|
||||
char* DlError();
|
||||
|
||||
} // namespace os
|
||||
} // namespace rocr
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -196,6 +196,35 @@ template <typename Allocator> class SimpleHeap {
|
||||
return reinterpret_cast<void*>(base);
|
||||
}
|
||||
|
||||
/* Return block-base the ptr belongs to if the ptr is a valid ptr which is allocated
|
||||
* from this simpleheap and the block-base is allocated from block_allocator_*/
|
||||
void* block_base(void* ptr) {
|
||||
if (ptr == nullptr)
|
||||
return nullptr;
|
||||
|
||||
uintptr_t base = reinterpret_cast<uintptr_t>(ptr);
|
||||
|
||||
// Find fragment and validate.
|
||||
auto frag_map_it = block_list_.upper_bound(base);
|
||||
if (frag_map_it == block_list_.begin())
|
||||
return nullptr;
|
||||
frag_map_it--;
|
||||
auto& frag_map = frag_map_it->second;
|
||||
auto fragment = frag_map.find(base);
|
||||
if (fragment == frag_map.end() || isFree(fragment->second))
|
||||
return nullptr;
|
||||
|
||||
return reinterpret_cast<void*>(frag_map_it->first);
|
||||
}
|
||||
|
||||
void reset() {
|
||||
free_list_.clear();
|
||||
block_list_.clear();
|
||||
block_cache_.clear();
|
||||
in_use_size_ = 0;
|
||||
cache_size_ = 0;
|
||||
}
|
||||
|
||||
bool free(void* ptr) {
|
||||
if (ptr == nullptr) return true;
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -49,13 +49,16 @@
|
||||
#include "stddef.h"
|
||||
#include "stdlib.h"
|
||||
#include "stdarg.h"
|
||||
#if defined(__linux__)
|
||||
#include "unistd.h"
|
||||
#endif
|
||||
#include <assert.h>
|
||||
#include <iostream>
|
||||
#include <string>
|
||||
#include <algorithm>
|
||||
#include <sstream>
|
||||
#include <thread>
|
||||
#include <locale>
|
||||
|
||||
namespace rocr {
|
||||
extern FILE* log_file;
|
||||
@@ -64,6 +67,14 @@ extern uint8_t log_flags[8];
|
||||
typedef unsigned int uint;
|
||||
typedef uint64_t uint64;
|
||||
|
||||
// 2MB huge page size
|
||||
#define GPU_HUGE_PAGE_SIZE (2 << 20)
|
||||
|
||||
// 4KB page size
|
||||
#define DEFAULT_GPU_PAGE_SIZE (1 << 12)
|
||||
|
||||
void log_printf(const char* file, int line, const char* format, ...);
|
||||
|
||||
#if defined(__GNUC__)
|
||||
#if defined(__i386__) || defined(__x86_64__)
|
||||
#include <x86intrin.h>
|
||||
@@ -75,8 +86,6 @@ typedef uint64_t uint64;
|
||||
#define __stdcall // __attribute__((__stdcall__))
|
||||
#define __ALIGNED__(x) __attribute__((aligned(x)))
|
||||
|
||||
void log_printf(const char* file, int line, const char* format, ...);
|
||||
|
||||
static __forceinline void* _aligned_malloc(size_t size, size_t alignment) {
|
||||
#ifdef _ISOC11_SOURCE
|
||||
return aligned_alloc(alignment, size);
|
||||
@@ -114,6 +123,7 @@ static __forceinline unsigned long long int strtoull(const char* str,
|
||||
do { \
|
||||
} while (false)
|
||||
#else
|
||||
#if defined(__linux__)
|
||||
#define debug_warning_n(exp, limit) \
|
||||
do { \
|
||||
static std::atomic<int> count(0); \
|
||||
@@ -123,6 +133,18 @@ static __forceinline unsigned long long int strtoull(const char* str,
|
||||
count++; \
|
||||
} \
|
||||
} while (false)
|
||||
#else
|
||||
#define debug_warning_n(exp, limit) \
|
||||
do { \
|
||||
static std::atomic<int> count(0); \
|
||||
if (!(exp) && (limit == 0 || count < limit)) { \
|
||||
fprintf(stderr, "Warning: " STRING(exp) " in %s, " __FILE__ ":" STRING(__LINE__) "\n" \
|
||||
); \
|
||||
count++; \
|
||||
} \
|
||||
} while (false)
|
||||
|
||||
#endif
|
||||
#endif
|
||||
#define debug_warning(exp) debug_warning_n((exp), 0)
|
||||
|
||||
@@ -369,10 +391,15 @@ inline void FlushCpuCache(const void* base, size_t offset, size_t len) {
|
||||
static long cacheline_size = 0;
|
||||
|
||||
if (!cacheline_size) {
|
||||
#ifdef _SC_LEVEL1_DCACHE_LINESIZE
|
||||
long sz = sysconf(_SC_LEVEL1_DCACHE_LINESIZE);
|
||||
long sz = 64;
|
||||
#if defined(__linux__)
|
||||
#ifdef _SC_LEVEL1_DCACHE_LINESIZE
|
||||
sz = sysconf(_SC_LEVEL1_DCACHE_LINESIZE);
|
||||
#else
|
||||
sz = 0;
|
||||
#endif
|
||||
#else
|
||||
long sz = 0;
|
||||
//@todo abstract GetLogicalProcessorInformation call
|
||||
#endif
|
||||
if (sz <= 0) return;
|
||||
cacheline_size = sz;
|
||||
@@ -421,6 +448,17 @@ inline uint32_t PtrHigh64Shift40(const void* p) {
|
||||
return (uint32_t)((ptr & 0xFFFFFF0000000000ULL) >> 40);
|
||||
}
|
||||
|
||||
static inline uint8_t Ptr48High8(const void* p) {
|
||||
uintptr_t ptr = reinterpret_cast<uintptr_t>(p);
|
||||
return (uint8_t)((ptr & 0xFF0000000000ULL) >> 40);
|
||||
}
|
||||
|
||||
static inline uint32_t Ptr48Low32(const void* p) {
|
||||
uintptr_t ptr = reinterpret_cast<uintptr_t>(p);
|
||||
assert((ptr & 0xFFFFFFFFFF00ULL) == ptr);
|
||||
return (uint32_t)((ptr & 0xFFFFFFFFFFULL) >> 8);
|
||||
}
|
||||
|
||||
inline uint32_t PtrLow32(const void* p) {
|
||||
return static_cast<uint32_t>(reinterpret_cast<uintptr_t>(p));
|
||||
}
|
||||
@@ -433,6 +471,10 @@ inline uint32_t PtrHigh32(const void* p) {
|
||||
return ptr;
|
||||
}
|
||||
|
||||
inline uint32_t HighPart(uint64_t value) { return (value & 0xFFFFFFFF00000000) >> 32; }
|
||||
|
||||
inline uint32_t LowPart(uint64_t value) { return (value & 0x00000000FFFFFFFF); }
|
||||
|
||||
/// @brief: Concatenates two numbers of type InType to a number of type OutType
|
||||
/// @param: hi(Input), To be placed in the upper bits of the output
|
||||
/// @param: lo(Input), To be placed in the lower bits of the output
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -41,7 +41,6 @@
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#ifdef _WIN32 // Are we compiling for windows?
|
||||
#define NOMINMAX
|
||||
|
||||
#include "core/util/os.h"
|
||||
|
||||
@@ -49,10 +48,13 @@
|
||||
#include <process.h>
|
||||
#include <string>
|
||||
#include <windows.h>
|
||||
#include <ntstatus.h>
|
||||
#include <psapi.h>
|
||||
|
||||
#include <emmintrin.h>
|
||||
#include <pmmintrin.h>
|
||||
#include <xmmintrin.h>
|
||||
#include <shared_mutex>
|
||||
|
||||
#undef Yield
|
||||
#undef CreateMutex
|
||||
@@ -82,37 +84,37 @@ void* GetExportAddress(LibHandle lib, std::string export_name) {
|
||||
return GetProcAddress(*(HMODULE*)&lib, export_name.c_str());
|
||||
}
|
||||
|
||||
void CloseLib(LibHandle lib) { FreeLibrary(*(::HMODULE*)&lib); }
|
||||
bool CloseLib(LibHandle lib) { return FreeLibrary(*(::HMODULE*)&lib); }
|
||||
|
||||
std::vector<LibHandle> GetLoadedLibs() {
|
||||
// Use EnumProcessModulesEx
|
||||
static_assert(false, "Not implemented.");
|
||||
assert(!"Not implemented.");
|
||||
return std::vector<LibHandle>{};
|
||||
}
|
||||
|
||||
std::string GetLibraryName(LibHandle lib) {
|
||||
static_assert(false, "Not implemented.");
|
||||
assert(!"Not implemented.");
|
||||
return std::string{};
|
||||
}
|
||||
|
||||
Semaphore CreateSemaphore() {
|
||||
sem = static_cast<void*>(CreateSemaphore(NULL, 0, LONG_MAX, NULL));
|
||||
assert(sem != NULL && "CreateSemaphore failed");
|
||||
|
||||
auto sem = static_cast<void*>(CreateSemaphoreA(nullptr, 0, LONG_MAX, nullptr));
|
||||
assert(sem != nullptr && "CreateSemaphore failed");
|
||||
return *(Semaphore*)&sem;
|
||||
}
|
||||
|
||||
bool WaitSemaphore(Semaphore sem) {
|
||||
return WaitForSingleObject(*(::HANDLE*)&lock, INFINITE) == WAIT_OBJECT_0;
|
||||
return WaitForSingleObject(sem, INFINITE) == WAIT_OBJECT_0;
|
||||
}
|
||||
|
||||
void PostSemaphore(Semaphore sem) {
|
||||
ReleaseSemaphore(static_cast<HANDLE>(*sem), 1, NULL);
|
||||
ReleaseSemaphore(sem, 1, nullptr);
|
||||
}
|
||||
|
||||
void DestroySemaphore(Semaphore sem) {
|
||||
if (!CloseHandle(static_cast<HANDLE>(*sem))) {
|
||||
if (!CloseHandle(sem)) {
|
||||
assert("CloseHandle() failed");
|
||||
}
|
||||
*sem = NULL;
|
||||
}
|
||||
|
||||
Mutex CreateMutex() { return CreateEvent(NULL, false, true, NULL); }
|
||||
@@ -259,48 +261,37 @@ uint64_t AccurateClockFrequency() {
|
||||
}
|
||||
|
||||
SharedMutex CreateSharedMutex() {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return nullptr;
|
||||
return reinterpret_cast<SharedMutex>(new std::shared_mutex());
|
||||
}
|
||||
|
||||
bool TryAcquireSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
return reinterpret_cast<std::shared_mutex*>(lock)->try_lock();
|
||||
}
|
||||
|
||||
bool AcquireSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
reinterpret_cast<std::shared_mutex*>(lock)->lock();
|
||||
return true;
|
||||
}
|
||||
|
||||
void ReleaseSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
reinterpret_cast<std::shared_mutex*>(lock)->unlock();
|
||||
}
|
||||
|
||||
bool TrySharedAcquireSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
return reinterpret_cast<std::shared_mutex*>(lock)->try_lock_shared();
|
||||
}
|
||||
|
||||
bool SharedAcquireSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
reinterpret_cast<std::shared_mutex*>(lock)->lock_shared();
|
||||
return true;
|
||||
}
|
||||
|
||||
void SharedReleaseSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
reinterpret_cast<std::shared_mutex*>(lock)->unlock_shared();
|
||||
}
|
||||
|
||||
void DestroySharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
delete reinterpret_cast<std::shared_mutex*>(lock);
|
||||
}
|
||||
|
||||
uint64_t ReadSystemClock() {
|
||||
@@ -310,17 +301,183 @@ uint64_t ReadSystemClock() {
|
||||
}
|
||||
|
||||
uint64_t SystemClockFrequency() {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return 0;
|
||||
LARGE_INTEGER frequency;
|
||||
QueryPerformanceFrequency(&frequency);
|
||||
return frequency.QuadPart;
|
||||
}
|
||||
|
||||
bool ParseCpuID(cpuid_t* cpuinfo) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
int regs[4] = {};
|
||||
int info{};
|
||||
|
||||
__cpuid(regs, info);
|
||||
memset(cpuinfo->ManufacturerID, 0, sizeof(cpuinfo->ManufacturerID));
|
||||
*reinterpret_cast<int*>(cpuinfo->ManufacturerID) = regs[1];
|
||||
*reinterpret_cast<int*>(cpuinfo->ManufacturerID + 4) = regs[3];
|
||||
*reinterpret_cast<int*>(cpuinfo->ManufacturerID + 8) = regs[2];
|
||||
// @todo fill the rest of CPU info
|
||||
return true;
|
||||
}
|
||||
|
||||
bool IsEnvVarSet(std::string env_var_name) {
|
||||
char* buff = NULL;
|
||||
buff = getenv(env_var_name.c_str());
|
||||
return (buff != NULL);
|
||||
}
|
||||
|
||||
std::vector<LibHandle> GetLoadedToolsLib() {
|
||||
std::vector<LibHandle> ret;
|
||||
std::vector<std::string> names;
|
||||
HMODULE hMods[1024];
|
||||
HANDLE hProcess = GetCurrentProcess();
|
||||
DWORD cbNeeded;
|
||||
unsigned int i;
|
||||
|
||||
if (EnumProcessModules(hProcess, hMods, sizeof(hMods), &cbNeeded)) {
|
||||
for (i = 0; i < (cbNeeded / sizeof(HMODULE)); i++) {
|
||||
TCHAR szModName[MAX_PATH];
|
||||
|
||||
// Get the full path to the module's file.
|
||||
|
||||
if (GetModuleFileNameEx(hProcess, hMods[i], szModName, sizeof(szModName) / sizeof(TCHAR))) {
|
||||
// Print the module name and handle value.
|
||||
names.push_back(szModName);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!names.empty()) {
|
||||
for (auto& name : names) ret.push_back(LoadLib(name));
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
int GetProcessId() { return ::_getpid(); }
|
||||
|
||||
uint64_t TimeNanos() {
|
||||
static double PerformanceFrequency = 0.f;
|
||||
if (PerformanceFrequency == 0) {
|
||||
LARGE_INTEGER frequency;
|
||||
QueryPerformanceFrequency(&frequency);
|
||||
PerformanceFrequency = (double)frequency.QuadPart;
|
||||
}
|
||||
LARGE_INTEGER current;
|
||||
QueryPerformanceCounter(¤t);
|
||||
return (uint64_t)((double)current.QuadPart / PerformanceFrequency * 1e9);
|
||||
}
|
||||
|
||||
static inline int memProtToOsProt(MemProt prot) {
|
||||
switch (prot) {
|
||||
case MEM_PROT_NONE:
|
||||
return PAGE_NOACCESS;
|
||||
case MEM_PROT_READ:
|
||||
return PAGE_READONLY;
|
||||
case MEM_PROT_RW:
|
||||
return PAGE_READWRITE;
|
||||
case MEM_PROT_RWX:
|
||||
return PAGE_EXECUTE_READWRITE;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
static size_t g_page_size_ = 0; //!< The default os page size
|
||||
static int processorCount_; //!< The number of active processors
|
||||
static size_t allocationGranularity_;
|
||||
|
||||
//! Return the default os page size.
|
||||
size_t PageSize() {
|
||||
if (g_page_size_ == 0) {
|
||||
SYSTEM_INFO si{};
|
||||
::GetSystemInfo(&si);
|
||||
g_page_size_ = si.dwPageSize;
|
||||
}
|
||||
return g_page_size_;
|
||||
}
|
||||
|
||||
void* ReserveMemory(void* start, size_t size, size_t alignment, MemProt prot) {
|
||||
size = AlignUp(size, PageSize());
|
||||
if (allocationGranularity_ == 0) {
|
||||
SYSTEM_INFO si;
|
||||
::GetSystemInfo(&si);
|
||||
g_page_size_ = si.dwPageSize;
|
||||
allocationGranularity_ = (size_t)si.dwAllocationGranularity;
|
||||
}
|
||||
alignment = std::max(allocationGranularity_, AlignUp(alignment, allocationGranularity_));
|
||||
assert(IsPowerOfTwo(alignment) && "not a power of 2");
|
||||
|
||||
size_t requested = size + alignment - allocationGranularity_;
|
||||
address mem, aligned;
|
||||
do {
|
||||
mem = reinterpret_cast<address>(VirtualAlloc(start, requested, MEM_RESERVE, memProtToOsProt(prot)));
|
||||
|
||||
// check for out of memory.
|
||||
if (mem == NULL) return NULL;
|
||||
|
||||
aligned = AlignUp(mem, alignment);
|
||||
|
||||
// check for already aligned memory.
|
||||
if (aligned == mem && size == requested) {
|
||||
return mem;
|
||||
}
|
||||
|
||||
// try to reserve the aligned address.
|
||||
if (VirtualFree(mem, 0, MEM_RELEASE) == 0) {
|
||||
assert(!"VirtualFree failed");
|
||||
}
|
||||
|
||||
mem = (address)VirtualAlloc(aligned, size, MEM_RESERVE, memProtToOsProt(prot));
|
||||
assert((mem == NULL || mem == aligned) && "VirtualAlloc failed");
|
||||
|
||||
} while (mem != aligned);
|
||||
|
||||
return mem;
|
||||
}
|
||||
bool ReleaseMemory(void* addr, size_t size) { return VirtualFree(addr, 0, MEM_RELEASE) != 0; }
|
||||
|
||||
bool CommitMemory(void* addr, size_t size, MemProt prot) {
|
||||
return VirtualAlloc(addr, size, MEM_COMMIT, memProtToOsProt(prot)) != NULL;
|
||||
}
|
||||
|
||||
bool UncommitMemory(void* addr, size_t size) { return VirtualFree(addr, size, MEM_DECOMMIT) != 0; }
|
||||
|
||||
uint64_t HostTotalPhysicalMemory() {
|
||||
static uint64_t totalPhys = 0;
|
||||
|
||||
if (totalPhys != 0) {
|
||||
return totalPhys;
|
||||
}
|
||||
|
||||
MEMORYSTATUSEX mstatus;
|
||||
mstatus.dwLength = sizeof(mstatus);
|
||||
|
||||
::GlobalMemoryStatusEx(&mstatus);
|
||||
|
||||
totalPhys = mstatus.ullTotalPhys;
|
||||
return totalPhys;
|
||||
}
|
||||
|
||||
int Ffs(int i) {
|
||||
int res = 0;
|
||||
unsigned long index;
|
||||
if (_BitScanForward(&index, i) != 0) {
|
||||
res = index + 1;
|
||||
}
|
||||
return res;
|
||||
}
|
||||
|
||||
int Ctz(uint64_t i) {
|
||||
unsigned long index;
|
||||
if (_BitScanReverse64(&index, i)) {
|
||||
return sizeof(i) * 8 - 1 - index;
|
||||
} else {
|
||||
return sizeof(i) * 8;
|
||||
}
|
||||
}
|
||||
|
||||
char* DlError() { return nullptr; }
|
||||
} // namespace os
|
||||
} // namespace rocr
|
||||
|
||||
|
||||
@@ -51,8 +51,9 @@ if( NOT _is_hsa_runtime_dynamic )
|
||||
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} "${CMAKE_CURRENT_LIST_DIR}")
|
||||
|
||||
find_dependency(hsakmt 1.0)
|
||||
find_dependency(LibElf)
|
||||
|
||||
if (UNIX)
|
||||
find_dependency(LibElf)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
include( "${CMAKE_CURRENT_LIST_DIR}/@CORE_RUNTIME_NAME@Targets.cmake" )
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
EXPORTS
|
||||
hsa_init
|
||||
hsa_shut_down
|
||||
hsa_system_get_info
|
||||
hsa_extension_get_name
|
||||
hsa_system_extension_supported
|
||||
hsa_system_major_extension_supported
|
||||
hsa_system_get_extension_table
|
||||
hsa_system_get_major_extension_table
|
||||
hsa_iterate_agents
|
||||
hsa_agent_get_info
|
||||
hsa_agent_get_exception_policies
|
||||
hsa_cache_get_info
|
||||
hsa_agent_iterate_caches
|
||||
hsa_agent_extension_supported
|
||||
hsa_agent_major_extension_supported
|
||||
hsa_queue_create
|
||||
hsa_soft_queue_create
|
||||
hsa_queue_destroy
|
||||
hsa_queue_inactivate
|
||||
hsa_queue_load_read_index_scacquire
|
||||
hsa_queue_load_read_index_relaxed
|
||||
hsa_queue_load_write_index_scacquire
|
||||
hsa_queue_load_write_index_relaxed
|
||||
hsa_queue_store_write_index_relaxed
|
||||
hsa_queue_store_write_index_screlease
|
||||
hsa_queue_cas_write_index_scacq_screl
|
||||
hsa_queue_cas_write_index_scacquire
|
||||
hsa_queue_cas_write_index_relaxed
|
||||
hsa_queue_cas_write_index_screlease
|
||||
hsa_queue_add_write_index_scacq_screl
|
||||
hsa_queue_add_write_index_scacquire
|
||||
hsa_queue_add_write_index_relaxed
|
||||
hsa_queue_add_write_index_screlease
|
||||
hsa_queue_store_read_index_relaxed
|
||||
hsa_queue_store_read_index_screlease
|
||||
hsa_agent_iterate_regions
|
||||
hsa_region_get_info
|
||||
hsa_memory_register
|
||||
hsa_memory_deregister
|
||||
hsa_memory_allocate
|
||||
hsa_memory_free
|
||||
hsa_memory_copy
|
||||
hsa_memory_assign_agent
|
||||
hsa_signal_create
|
||||
hsa_signal_destroy
|
||||
hsa_signal_load_relaxed
|
||||
hsa_signal_load_scacquire
|
||||
hsa_signal_store_relaxed
|
||||
hsa_signal_store_screlease
|
||||
hsa_signal_silent_store_relaxed
|
||||
hsa_signal_silent_store_screlease
|
||||
hsa_signal_wait_relaxed
|
||||
hsa_signal_wait_scacquire
|
||||
hsa_signal_group_create
|
||||
hsa_signal_group_destroy
|
||||
hsa_signal_group_wait_any_scacquire
|
||||
hsa_signal_group_wait_any_relaxed
|
||||
hsa_signal_and_relaxed
|
||||
hsa_signal_and_scacquire
|
||||
hsa_signal_and_screlease
|
||||
hsa_signal_and_scacq_screl
|
||||
hsa_signal_or_relaxed
|
||||
hsa_signal_or_scacquire
|
||||
hsa_signal_or_screlease
|
||||
hsa_signal_or_scacq_screl
|
||||
hsa_signal_xor_relaxed
|
||||
hsa_signal_xor_scacquire
|
||||
hsa_signal_xor_screlease
|
||||
hsa_signal_xor_scacq_screl
|
||||
hsa_signal_exchange_relaxed
|
||||
hsa_signal_exchange_scacquire
|
||||
hsa_signal_exchange_screlease
|
||||
hsa_signal_exchange_scacq_screl
|
||||
hsa_signal_add_relaxed
|
||||
hsa_signal_add_scacquire
|
||||
hsa_signal_add_screlease
|
||||
hsa_signal_add_scacq_screl
|
||||
hsa_signal_subtract_relaxed
|
||||
hsa_signal_subtract_scacquire
|
||||
hsa_signal_subtract_screlease
|
||||
hsa_signal_subtract_scacq_screl
|
||||
hsa_signal_cas_relaxed
|
||||
hsa_signal_cas_scacquire
|
||||
hsa_signal_cas_screlease
|
||||
hsa_signal_cas_scacq_screl
|
||||
hsa_isa_from_name
|
||||
hsa_agent_iterate_isas
|
||||
hsa_isa_get_info
|
||||
hsa_isa_get_info_alt
|
||||
hsa_isa_get_exception_policies
|
||||
hsa_isa_get_round_method
|
||||
hsa_wavefront_get_info
|
||||
hsa_isa_iterate_wavefronts
|
||||
hsa_isa_compatible
|
||||
hsa_code_object_serialize
|
||||
hsa_code_object_deserialize
|
||||
hsa_code_object_destroy
|
||||
hsa_code_object_get_info
|
||||
hsa_code_object_get_symbol
|
||||
hsa_code_object_get_symbol_from_name
|
||||
hsa_code_symbol_get_info
|
||||
hsa_code_object_iterate_symbols
|
||||
hsa_code_object_reader_create_from_file
|
||||
hsa_code_object_reader_create_from_memory
|
||||
hsa_code_object_reader_destroy
|
||||
hsa_executable_create
|
||||
hsa_executable_create_alt
|
||||
hsa_executable_destroy
|
||||
hsa_executable_load_code_object
|
||||
hsa_executable_load_program_code_object
|
||||
hsa_executable_load_agent_code_object
|
||||
hsa_executable_freeze
|
||||
hsa_executable_get_info
|
||||
hsa_executable_global_variable_define
|
||||
hsa_executable_agent_global_variable_define
|
||||
hsa_executable_readonly_variable_define
|
||||
hsa_executable_validate
|
||||
hsa_executable_validate_alt
|
||||
hsa_executable_get_symbol
|
||||
hsa_executable_get_symbol_by_name
|
||||
hsa_executable_symbol_get_info
|
||||
hsa_executable_iterate_symbols
|
||||
hsa_executable_iterate_agent_symbols
|
||||
hsa_executable_iterate_program_symbols
|
||||
hsa_status_string
|
||||
hsa_ext_program_create
|
||||
hsa_ext_program_destroy
|
||||
hsa_ext_program_add_module
|
||||
hsa_ext_program_iterate_modules
|
||||
hsa_ext_program_get_info
|
||||
hsa_ext_program_finalize
|
||||
hsa_amd_coherency_get_type
|
||||
hsa_amd_coherency_set_type
|
||||
hsa_amd_profiling_set_profiler_enabled
|
||||
hsa_amd_profiling_get_dispatch_time
|
||||
hsa_amd_profiling_async_copy_enable
|
||||
hsa_amd_profiling_get_async_copy_time
|
||||
hsa_amd_profiling_convert_tick_to_system_domain
|
||||
hsa_amd_signal_create
|
||||
hsa_amd_signal_wait_any
|
||||
hsa_amd_signal_async_handler
|
||||
hsa_amd_async_function
|
||||
hsa_amd_image_get_info_max_dim
|
||||
hsa_amd_queue_cu_set_mask
|
||||
hsa_amd_queue_cu_get_mask
|
||||
hsa_amd_memory_fill
|
||||
hsa_amd_memory_async_copy
|
||||
hsa_amd_memory_async_copy_on_engine
|
||||
hsa_amd_memory_copy_engine_status
|
||||
hsa_amd_memory_get_preferred_copy_engine
|
||||
hsa_amd_memory_async_copy_rect
|
||||
hsa_amd_memory_lock
|
||||
hsa_amd_memory_lock_to_pool
|
||||
hsa_amd_memory_unlock
|
||||
hsa_amd_agent_iterate_memory_pools
|
||||
hsa_amd_agent_memory_pool_get_info
|
||||
hsa_amd_agents_allow_access
|
||||
hsa_amd_memory_pool_get_info
|
||||
hsa_amd_memory_pool_allocate
|
||||
hsa_amd_memory_pool_free
|
||||
hsa_amd_memory_pool_can_migrate
|
||||
hsa_amd_memory_migrate
|
||||
hsa_amd_interop_map_buffer
|
||||
hsa_amd_interop_unmap_buffer
|
||||
hsa_amd_image_create
|
||||
hsa_ext_image_get_capability
|
||||
hsa_ext_image_data_get_info
|
||||
hsa_ext_image_create
|
||||
hsa_ext_image_import
|
||||
hsa_ext_image_export
|
||||
hsa_ext_image_copy
|
||||
hsa_ext_image_clear
|
||||
hsa_ext_image_destroy
|
||||
hsa_ext_sampler_create
|
||||
hsa_ext_sampler_create_v2
|
||||
hsa_ext_sampler_destroy
|
||||
hsa_ext_image_get_capability_with_layout
|
||||
hsa_ext_image_data_get_info_with_layout
|
||||
hsa_ext_image_create_with_layout
|
||||
hsa_amd_pointer_info
|
||||
hsa_amd_pointer_info_set_userdata
|
||||
hsa_amd_ipc_memory_create
|
||||
hsa_amd_ipc_memory_attach
|
||||
hsa_amd_ipc_memory_detach
|
||||
hsa_amd_ipc_signal_create
|
||||
hsa_amd_ipc_signal_attach
|
||||
hsa_amd_register_system_event_handler
|
||||
hsa_amd_queue_set_priority
|
||||
hsa_amd_register_deallocation_callback
|
||||
hsa_amd_deregister_deallocation_callback
|
||||
hsa_amd_signal_value_pointer
|
||||
_amdgpu_r_debug
|
||||
hsa_amd_svm_attributes_set
|
||||
hsa_amd_svm_attributes_get
|
||||
hsa_amd_svm_prefetch_async
|
||||
hsa_amd_spm_acquire
|
||||
hsa_amd_spm_release
|
||||
hsa_amd_spm_set_dest_buffer
|
||||
hsa_amd_portable_export_dmabuf
|
||||
hsa_amd_portable_close_dmabuf
|
||||
hsa_amd_vmem_address_reserve
|
||||
hsa_amd_vmem_address_reserve_align
|
||||
hsa_amd_vmem_address_free
|
||||
hsa_amd_vmem_handle_create
|
||||
hsa_amd_vmem_handle_release
|
||||
hsa_amd_vmem_map
|
||||
hsa_amd_vmem_unmap
|
||||
hsa_amd_vmem_set_access
|
||||
hsa_amd_vmem_get_access
|
||||
hsa_amd_vmem_export_shareable_handle
|
||||
hsa_amd_vmem_import_shareable_handle
|
||||
hsa_amd_vmem_retain_alloc_handle
|
||||
hsa_amd_vmem_get_alloc_properties_from_handle
|
||||
hsa_amd_agent_set_async_scratch_limit
|
||||
hsa_ven_amd_pcs_iterate_configuration
|
||||
hsa_ven_amd_pcs_create
|
||||
hsa_ven_amd_pcs_create_from_id
|
||||
hsa_ven_amd_pcs_destroy
|
||||
hsa_ven_amd_pcs_start
|
||||
hsa_ven_amd_pcs_stop
|
||||
hsa_ven_amd_pcs_flush
|
||||
hsa_amd_queue_get_info
|
||||
hsa_amd_enable_logging
|
||||
hsa_amd_signal_wait_all
|
||||
hsa_amd_portable_export_dmabuf_v2
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2023, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2023-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -40,10 +40,16 @@
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <unistd.h>
|
||||
#include <elf.h>
|
||||
#include <fcntl.h>
|
||||
#include <sys/resource.h>
|
||||
#include <elf.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#include <stdio.h>
|
||||
#include <win32/elf.h>
|
||||
#endif
|
||||
#include <fcntl.h>
|
||||
#include <cstring>
|
||||
#include <vector>
|
||||
#include <sstream>
|
||||
@@ -270,11 +276,14 @@ struct LoadSegmentBuilder : public SegmentBuilder {
|
||||
if (fd_ == -1) return HSA_STATUS_ERROR;
|
||||
|
||||
size_t done = 0;
|
||||
ssize_t read;
|
||||
size_t read;
|
||||
do {
|
||||
#if defined(__linux__)
|
||||
read = pread(fd_, static_cast<char *>(buf) + done, buf_size - done,
|
||||
offset + done);
|
||||
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
if (read == -1 && errno != EINTR) {
|
||||
perror("Failed to read GPU memory");
|
||||
return HSA_STATUS_ERROR;
|
||||
@@ -305,6 +314,7 @@ hsa_status_t build_core_dump(const std::string& filename, const SegmentsInfo& se
|
||||
debug_print("Core file size over limit\n");
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
#if defined(__linux__)
|
||||
int fd = open(filename.c_str(), O_WRONLY | O_CREAT | O_EXCL, S_IRUSR | S_IWUSR);
|
||||
if (fd == -1) {
|
||||
perror("Failed to create GPU coredump");
|
||||
@@ -423,6 +433,9 @@ hsa_status_t build_core_dump(const std::string& filename, const SegmentsInfo& se
|
||||
}
|
||||
printf("GPU core dump created: %s\n", filename.c_str());
|
||||
close(fd);
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
} // namespace impl
|
||||
@@ -431,7 +444,7 @@ hsa_status_t dump_gpu_core() {
|
||||
impl::NoteSegmentBuilder nbuilder;
|
||||
impl::LoadSegmentBuilder lbuilder;
|
||||
impl::SegmentsInfo segments;
|
||||
|
||||
#if defined(__linux__)
|
||||
struct rlimit rlimit;
|
||||
|
||||
if (getrlimit(RLIMIT_CORE, &rlimit)) {
|
||||
@@ -452,6 +465,10 @@ hsa_status_t dump_gpu_core() {
|
||||
std::stringstream st;
|
||||
st << PREFIX_FILE_NAME << "." << getpid();
|
||||
return build_core_dump(st.str(), segments, rlimit.rlim_cur);
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
return HSA_STATUS_SUCCESS;
|
||||
#endif
|
||||
}
|
||||
} // namespace coredump
|
||||
} // namespace amd
|
||||
|
||||
Plik diff jest za duży
Load Diff
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -40,12 +40,14 @@
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include "executable.hpp"
|
||||
|
||||
#include <libelf.h>
|
||||
#include <limits.h>
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstddef>
|
||||
@@ -61,6 +63,8 @@
|
||||
#include "amd_options.hpp"
|
||||
#include "core/util/utils.h"
|
||||
|
||||
#include "executable.hpp"
|
||||
|
||||
#include "AMDHSAKernelDescriptor.h"
|
||||
|
||||
using namespace rocr::amd::hsa;
|
||||
@@ -88,7 +92,12 @@ static __forceinline link_map*& r_debug_tail() {
|
||||
namespace rocr {
|
||||
// Having a side effect prevents call site optimization that allows removal of a noinline function call
|
||||
// with no side effect.
|
||||
__attribute__((noinline)) void _loader_debug_state() {
|
||||
#if defined(__linux__)
|
||||
__attribute__((noinline))
|
||||
#else
|
||||
__declspec(noinline)
|
||||
#endif
|
||||
void _loader_debug_state() {
|
||||
static volatile int function_needs_a_side_effect = 0;
|
||||
function_needs_a_side_effect ^= 1;
|
||||
}
|
||||
|
||||
@@ -48,7 +48,9 @@
|
||||
#include <cstdint>
|
||||
#include <iostream>
|
||||
#include <libelf.h>
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#endif
|
||||
#include <list>
|
||||
#include <string>
|
||||
#include <unordered_map>
|
||||
@@ -62,6 +64,74 @@
|
||||
#include "inc/amd_hsa_kernel_code.h"
|
||||
#include "amd_hsa_locks.hpp"
|
||||
|
||||
#if defined(_WIN32) || defined(_WIN64)
|
||||
// r_version history:
|
||||
// 1: Initial debug protocol
|
||||
// 2: New trap handler ABI. The reason for halting a wave is recorded in ttmp11[8:7].
|
||||
// 3: New trap handler ABI. A wave halted at S_ENDPGM rewinds its PC by 8 bytes, and sets
|
||||
// ttmp11[9]=1. 4: New trap handler ABI. Save the trap id in ttmp11[16:9] 5: New trap handler ABI.
|
||||
// Save the PC in ttmp11[22:7] ttmp6[31:0], and park the wave if stopped 6: New trap handler ABI.
|
||||
// ttmp6[25:0] contains dispatch index modulo queue size 7: New trap handler ABI. Send interrupts as
|
||||
// a bitmask, coalescing concurrent exceptions. 8: New trap handler ABI. for gfx942: Initialize
|
||||
// ttmp[4:5] if ttmp11[31] == 0. 9: New trap handler ABI. For gfx11: Save PC in ttmp11[22:7]
|
||||
// ttmp6[31:0], and park the wave if stopped. 10: New trap handler ABI. Set status.skip_export when
|
||||
// halting the wave.
|
||||
// For gfx942, set ttmp6[31] = 0 if ttmp11[31] == 0.
|
||||
#if _WIN64
|
||||
#define __WORDSIZE 64
|
||||
#else
|
||||
#define __WORDSIZE 32
|
||||
#endif
|
||||
|
||||
#define __ELF_NATIVE_CLASS __WORDSIZE
|
||||
|
||||
/* We use this macro to refer to ELF types independent of the native wordsize.
|
||||
`ElfW(TYPE)' is used in place of `Elf32_TYPE' or `Elf64_TYPE'. */
|
||||
#define _ElfW_1(e, w, t) e##w##t
|
||||
#define _ElfW(e, w, t) _ElfW_1(e, w, _##t)
|
||||
#define ElfW(type) _ElfW(Elf, __ELF_NATIVE_CLASS, type)
|
||||
|
||||
/* Structure describing a loaded shared object. The `l_next' and `l_prev'
|
||||
members form a chain of all the shared objects loaded at startup.
|
||||
|
||||
These data structures exist in space used by the run-time dynamic linker;
|
||||
modifying them may have disastrous results. */
|
||||
|
||||
struct link_map {
|
||||
/* These first few members are part of the protocol with the debugger.
|
||||
This is the same format used in SVR4. */
|
||||
|
||||
ElfW(Addr) l_addr; /* Difference between the address in the ELF
|
||||
file and the addresses in memory. */
|
||||
char* l_name; /* Absolute file name object was found in. */
|
||||
ElfW(Dyn) * l_ld; /* Dynamic section of the shared object. */
|
||||
struct link_map *l_next, *l_prev; /* Chain of loaded objects. */
|
||||
};
|
||||
|
||||
struct r_debug {
|
||||
/* Version number for this protocol. It should be greater than 0. */
|
||||
int r_version;
|
||||
|
||||
struct link_map* r_map; /* Head of the chain of loaded objects. */
|
||||
|
||||
/* This is the address of a function internal to the run-time linker,
|
||||
that will always be called when the linker begins to map in a
|
||||
library or unmap it, and again when the mapping change is complete.
|
||||
The debugger can set a breakpoint at this address if it wants to
|
||||
notice shared object mapping changes. */
|
||||
ElfW(Addr) r_brk;
|
||||
enum RT {
|
||||
/* This state value describes the mapping change taking place when
|
||||
the `r_brk' address is called. */
|
||||
RT_CONSISTENT, /* Mapping change is complete. */
|
||||
RT_ADD, /* Beginning to add a new object. */
|
||||
RT_DELETE /* Beginning to remove an object mapping. */
|
||||
} r_state;
|
||||
|
||||
ElfW(Addr) r_ldbase; /* Base address the linker is loaded at. */
|
||||
};
|
||||
#endif
|
||||
|
||||
namespace rocr {
|
||||
namespace amd {
|
||||
namespace hsa {
|
||||
@@ -604,7 +674,7 @@ public:
|
||||
hsa_status_t QuerySegmentDescriptors(
|
||||
hsa_ven_amd_loader_segment_descriptor_t *segment_descriptors,
|
||||
size_t *num_segment_descriptors) override;
|
||||
|
||||
#undef FindExecutable
|
||||
hsa_executable_t FindExecutable(uint64_t device_address) override;
|
||||
|
||||
uint64_t FindHostAddress(uint64_t device_address) override;
|
||||
|
||||
@@ -306,7 +306,7 @@ hsa_status_t PcsRuntime::PcSamplingCreateInternal(
|
||||
|
||||
hsa_status_t PcsRuntime::PcSamplingDestroy(hsa_ven_amd_pcs_t handle) {
|
||||
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
|
||||
if (pcSamplingSessionIt == pc_sampling_.end()) {
|
||||
debug_warning(false && "Cannot find PcSampling session");
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
@@ -320,7 +320,7 @@ hsa_status_t PcsRuntime::PcSamplingDestroy(hsa_ven_amd_pcs_t handle) {
|
||||
|
||||
hsa_status_t PcsRuntime::PcSamplingStart(hsa_ven_amd_pcs_t handle) {
|
||||
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
|
||||
if (pcSamplingSessionIt == pc_sampling_.end()) {
|
||||
debug_warning(false && "Cannot find PcSampling session");
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
@@ -332,7 +332,7 @@ hsa_status_t PcsRuntime::PcSamplingStart(hsa_ven_amd_pcs_t handle) {
|
||||
|
||||
hsa_status_t PcsRuntime::PcSamplingStop(hsa_ven_amd_pcs_t handle) {
|
||||
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
|
||||
if (pcSamplingSessionIt == pc_sampling_.end()) {
|
||||
debug_warning(false && "Cannot find PcSampling session");
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
@@ -344,7 +344,7 @@ hsa_status_t PcsRuntime::PcSamplingStop(hsa_ven_amd_pcs_t handle) {
|
||||
|
||||
hsa_status_t PcsRuntime::PcSamplingFlush(hsa_ven_amd_pcs_t handle) {
|
||||
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
|
||||
if (pcSamplingSessionIt == pc_sampling_.end()) {
|
||||
debug_warning(false && "Cannot find PcSampling session");
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
|
||||
Reference in New Issue
Block a user