Add windows build support into ROCr (#912)
Make sure ROCR can be compiled under windows. Extra setup for the windows build environment is required. The change should not have any functional changes under Linux.
Этот коммит содержится в:
коммит произвёл
GitHub
родитель
96a0d16eda
Коммит
913743d433
@@ -117,6 +117,11 @@ set_target_properties(hsakmt PROPERTIES
|
||||
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/archive"
|
||||
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/lib"
|
||||
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/runtime")
|
||||
if (WIN32)
|
||||
set_target_properties(hsakmt PROPERTIES
|
||||
CXX_STANDARD 20
|
||||
CXX_STANDARD_REQUIRED ON)
|
||||
endif()
|
||||
|
||||
if (BUILD_THUNK_VIRTIO)
|
||||
add_rocm_subdir(libhsakmt/src/virtio "${THUNK_VIRTIO_DEFINITIONS}")
|
||||
@@ -128,6 +133,11 @@ if (BUILD_ROCR)
|
||||
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/archive"
|
||||
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/lib"
|
||||
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/runtime")
|
||||
if (WIN32)
|
||||
set_target_properties(hsa-runtime64 PROPERTIES
|
||||
CXX_STANDARD 20
|
||||
CXX_STANDARD_REQUIRED ON)
|
||||
endif()
|
||||
|
||||
if (BUILD_SHARED_LIBS)
|
||||
add_dependencies(hsa-runtime64 hsakmt)
|
||||
@@ -187,6 +197,7 @@ if (DEFINED ENV{ROCM_LIBPATCH_VERSION})
|
||||
message("Using CPACK_PACKAGE_VERSION ${CPACK_PACKAGE_VERSION}")
|
||||
endif()
|
||||
|
||||
if (UNIX)
|
||||
# Debian package specific variables
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_NAME "hsa-rocr")
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_NAME "hsa-rocr-dev")
|
||||
@@ -223,6 +234,10 @@ set(CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS "libdrm-amdgpu-dev | libdrm-dev, rocm-core
|
||||
set(CPACK_DEBIAN_ASAN_PACKAGE_RECOMMENDS "libdrm-amdgpu-dev")
|
||||
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_RECOMMENDS "libdrm-amdgpu-amdgpu1")
|
||||
else()
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS "hsakmt-roct")
|
||||
set(CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS "hsakmt-roct")
|
||||
endif()
|
||||
if (ROCM_DEP_ROCMCORE)
|
||||
string(APPEND CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS ", rocm-core")
|
||||
string(APPEND CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS ", rocm-core-asan")
|
||||
@@ -244,6 +259,7 @@ set(CPACK_RPM_DEV_PACKAGE_OBSOLETES "hsakmt-roct,hsakmt-roct-devel,hsakmt-roct-d
|
||||
|
||||
set(CPACK_RPM_DEV_PACKAGE_NAME "hsa-rocr-devel")
|
||||
set(CPACK_RPM_ASAN_PACKAGE_NAME "hsa-rocr-asan")
|
||||
if (UNIX)
|
||||
if (DEFINED ENV{CPACK_RPM_PACKAGE_RELEASE})
|
||||
set(CPACK_RPM_PACKAGE_RELEASE $ENV{CPACK_RPM_PACKAGE_RELEASE})
|
||||
else()
|
||||
@@ -254,6 +270,7 @@ string(APPEND CPACK_RPM_PACKAGE_RELEASE "%{?dist}")
|
||||
set(CPACK_RPM_FILE_NAME "RPM-DEFAULT")
|
||||
message("CPACK_RPM_PACKAGE_RELEASE: ${CPACK_RPM_PACKAGE_RELEASE}")
|
||||
set(CPACK_RPM_PACKAGE_LICENSE "NCSA")
|
||||
endif()
|
||||
|
||||
## Process the Rpm install/remove scripts to update the CPACK variables
|
||||
configure_file("${CMAKE_CURRENT_SOURCE_DIR}/RPM/Binary/post.in" RPM/Binary/post @ONLY)
|
||||
@@ -289,7 +306,7 @@ endif()
|
||||
if (ROCM_DEP_ROCMCORE)
|
||||
string(APPEND CPACK_RPM_BINARY_PACKAGE_REQUIRES " rocm-core")
|
||||
string(APPEND CPACK_RPM_ASAN_PACKAGE_REQUIRES " rocm-core-asan")
|
||||
else()
|
||||
elseif (UNIX)
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_PACKAGE_REQUIRES ${CPACK_RPM_PACKAGE_REQUIRES})
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_DEBIAN_PACKAGE_DEPENDS ${CPACK_DEBIAN_PACKAGE_DEPENDS})
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_DEV_PACKAGE_REQUIRES ${CPACK_RPM_DEV_PACKAGE_REQUIRES})
|
||||
|
||||
@@ -1,3 +1,11 @@
|
||||
################################################################################
|
||||
##
|
||||
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
|
||||
##
|
||||
## SPDX-License-Identifier: MIT
|
||||
##
|
||||
################################################################################
|
||||
|
||||
cmake_minimum_required ( VERSION 3.5.0 )
|
||||
|
||||
# Set ext runtime module name and project name.
|
||||
@@ -23,8 +31,10 @@ include ( utils )
|
||||
|
||||
## Compiler preproc definitions.
|
||||
#add_definitions ( -D__linux__ )
|
||||
add_definitions ( -DUNIX_OS )
|
||||
add_definitions ( -DLINUX )
|
||||
if(UNIX)
|
||||
add_definitions ( -DUNIX_OS )
|
||||
add_definitions ( -DLINUX )
|
||||
endif()
|
||||
add_definitions ( -D__AMD64__ )
|
||||
add_definitions ( -D__x86_64__ )
|
||||
add_definitions ( -DAMD_INTERNAL_BUILD )
|
||||
|
||||
@@ -48,7 +48,11 @@ cmake_minimum_required ( VERSION 3.7 )
|
||||
unset ( hsa-runtime64_LIB_DEPENDS CACHE )
|
||||
|
||||
set(CMAKE_VERBOSE_MAKEFILE ON)
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
if (UNIX)
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
else()
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
endif()
|
||||
|
||||
## Set core runtime module name and project name.
|
||||
set ( CORE_RUNTIME_NAME "hsa-runtime64" )
|
||||
@@ -89,35 +93,46 @@ if(NOT LibElf_FOUND)
|
||||
find_package(LibElf REQUIRED)
|
||||
endif()
|
||||
|
||||
pkg_check_modules(drm REQUIRED IMPORTED_TARGET libdrm)
|
||||
|
||||
## Create the rocr target.
|
||||
add_library( ${CORE_RUNTIME_TARGET} "" )
|
||||
|
||||
if (UNIX)
|
||||
pkg_check_modules(drm REQUIRED IMPORTED_TARGET libdrm)
|
||||
else()
|
||||
target_include_directories(${CORE_RUNTIME_TARGET} PRIVATE ${LIBELF_INCLUDE_DIR})
|
||||
if (${BUILD_SHARED_LIBS})
|
||||
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE oclelf)
|
||||
endif()
|
||||
endif()
|
||||
## Enforce uniform output file naming.
|
||||
set_property(TARGET ${CORE_RUNTIME_TARGET} PROPERTY OUTPUT_NAME ${CORE_RUNTIME_NAME} )
|
||||
|
||||
## Compiler preproc definitions.
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" __linux__ HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
|
||||
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}" )
|
||||
|
||||
## Check for memfd_create syscall
|
||||
include(CheckSymbolExists)
|
||||
CHECK_SYMBOL_EXISTS ( "__NR_memfd_create" "sys/syscall.h" HAVE_MEMFD_CREATE )
|
||||
if ( HAVE_MEMFD_CREATE )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_MEMFD_CREATE )
|
||||
endif()
|
||||
|
||||
## Check for _GNU_SOURCE pthread extensions
|
||||
set(CMAKE_REQUIRED_DEFINITIONS -D_GNU_SOURCE)
|
||||
CHECK_SYMBOL_EXISTS ( "pthread_attr_setaffinity_np" "pthread.h" HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
CHECK_SYMBOL_EXISTS ( "pthread_rwlockattr_setkind_np" "pthread.h" HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
unset(CMAKE_REQUIRED_DEFINITIONS)
|
||||
if ( HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
endif()
|
||||
if ( HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
if (UNIX)
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" __linux__ HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
|
||||
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}" )
|
||||
|
||||
## Check for memfd_create syscall
|
||||
include(CheckSymbolExists)
|
||||
CHECK_SYMBOL_EXISTS ( "__NR_memfd_create" "sys/syscall.h" HAVE_MEMFD_CREATE )
|
||||
if ( HAVE_MEMFD_CREATE )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_MEMFD_CREATE )
|
||||
endif()
|
||||
|
||||
## Check for _GNU_SOURCE pthread extensions
|
||||
set(CMAKE_REQUIRED_DEFINITIONS -D_GNU_SOURCE)
|
||||
CHECK_SYMBOL_EXISTS ( "pthread_attr_setaffinity_np" "pthread.h" HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
CHECK_SYMBOL_EXISTS ( "pthread_rwlockattr_setkind_np" "pthread.h" HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
unset(CMAKE_REQUIRED_DEFINITIONS)
|
||||
if ( HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_ATTR_SETAFFINITY_NP )
|
||||
endif()
|
||||
if ( HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP )
|
||||
endif()
|
||||
else()
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" AMD_LIBELF=1 HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED=
|
||||
ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}")
|
||||
endif()
|
||||
|
||||
## Set include directories for ROCr runtime
|
||||
@@ -133,16 +148,18 @@ target_include_directories( ${CORE_RUNTIME_TARGET}
|
||||
|
||||
|
||||
## ------------------------- Linux Compiler and Linker options -------------------------
|
||||
set ( HSA_CXX_FLAGS ${HSA_COMMON_CXX_FLAGS} -fexceptions -fno-rtti -fvisibility=hidden -Wno-error=missing-braces -Wno-error=sign-compare -Wno-sign-compare -Wno-write-strings -Wno-conversion-null -fno-math-errno -fno-threadsafe-statics -fmerge-all-constants -fms-extensions -Wno-error=comment -Wno-comment -Wno-error=pointer-arith -Wno-pointer-arith -Wno-error=unused-variable -Wno-error=unused-function )
|
||||
if (UNIX)
|
||||
set ( HSA_CXX_FLAGS ${HSA_COMMON_CXX_FLAGS} -fexceptions -fno-rtti -fvisibility=hidden -Wno-error=missing-braces -Wno-error=sign-compare -Wno-sign-compare -Wno-write-strings -Wno-conversion-null -fno-math-errno -fno-threadsafe-statics -fmerge-all-constants -fms-extensions -Wno-error=comment -Wno-comment -Wno-error=pointer-arith -Wno-pointer-arith -Wno-error=unused-variable -Wno-error=unused-function )
|
||||
|
||||
## Extra x86 specific settings
|
||||
if ( CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64" )
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -mmwaitx )
|
||||
## Extra x86 specific settings
|
||||
if ( CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64" )
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -mmwaitx )
|
||||
endif()
|
||||
|
||||
## Extra image settings - audit!
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-deprecated-declarations )
|
||||
endif()
|
||||
|
||||
## Extra image settings - audit!
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-deprecated-declarations )
|
||||
|
||||
if ( CMAKE_COMPILER_IS_GNUCXX )
|
||||
set ( HSA_CXX_FLAGS ${HSA_CXX_FLAGS} -Wno-error=maybe-uninitialized -Wno-error=unused-but-set-variable)
|
||||
endif ()
|
||||
@@ -153,9 +170,14 @@ if ( CMAKE_CXX_COMPILER_ID MATCHES "Clang")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
set ( DRVDEF "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.so.def" )
|
||||
set ( LNKSCR "hsacore.so.link" )
|
||||
set ( HSA_SHARED_LINK_FLAGS "-Wl,-Bdynamic -Wl,-z,noexecstack -Wl,${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR} -Wl,--version-script=${DRVDEF} -Wl,--enable-new-dtags" )
|
||||
if (UNIX)
|
||||
set ( LNKSCR "hsacore.so.link" )
|
||||
set ( DRVDEF "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.so.def" )
|
||||
set(HSA_SHARED_LINK_FLAGS "-Wl,-Bdynamic -Wl,-z,noexecstack -Wl,${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR} -Wl,--version-script=${DRVDEF} -Wl,--enable-new-dtags")
|
||||
else()
|
||||
target_sources(${CORE_RUNTIME_TARGET} PRIVATE "${CMAKE_CURRENT_SOURCE_DIR}/hsacore.dll.def")
|
||||
set(HSA_SHARED_LINK_FLAGS "")
|
||||
endif()
|
||||
|
||||
target_compile_options(${CORE_RUNTIME_TARGET} PRIVATE ${HSA_CXX_FLAGS})
|
||||
#target_link_options not available prior to CMake 3.13
|
||||
@@ -165,13 +187,9 @@ set_property(TARGET ${CORE_RUNTIME_TARGET} PROPERTY LINK_FLAGS ${HSA_SHARED_LINK
|
||||
## Source files.
|
||||
set ( SRCS core/driver/driver.cpp
|
||||
core/driver/kfd/amd_kfd_driver.cpp
|
||||
core/driver/xdna/amd_xdna_driver.cpp
|
||||
core/util/lnx/os_linux.cpp
|
||||
core/util/small_heap.cpp
|
||||
core/util/timer.cpp
|
||||
core/util/flag.cpp
|
||||
core/runtime/amd_aie_agent.cpp
|
||||
core/runtime/amd_aie_aql_queue.cpp
|
||||
core/runtime/amd_blit_kernel.cpp
|
||||
core/runtime/amd_blit_sdma.cpp
|
||||
core/runtime/amd_cpu_agent.cpp
|
||||
@@ -208,12 +226,22 @@ set ( SRCS core/driver/driver.cpp
|
||||
libamdhsacode/amd_hsa_code.cpp
|
||||
libamdhsacode/amd_core_dump.cpp )
|
||||
|
||||
if(UNIX)
|
||||
set(SRC_OS core/util/lnx/os_linux.cpp)
|
||||
set(SRC_XDNA core/driver/xdna/amd_xdna_driver.cpp
|
||||
core/runtime/amd_aie_agent.cpp
|
||||
core/runtime/amd_aie_aql_queue.cpp)
|
||||
else()
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE NOMINMAX)
|
||||
set(SRC_OS core/util/win/os_win.cpp)
|
||||
endif()
|
||||
|
||||
if ( BUILD_THUNK_VIRTIO )
|
||||
list(APPEND SRCS core/driver/virtio/amd_kfd_virtio_driver.cpp)
|
||||
target_compile_definitions(hsa-runtime64 PRIVATE HSAKMT_VIRTIO_ENABLED=1)
|
||||
endif()
|
||||
|
||||
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${SRCS} )
|
||||
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${SRCS} ${SRC_OS} ${SRC_XDNA} )
|
||||
|
||||
## Depend on trap handler target.
|
||||
add_subdirectory( ${CMAKE_CURRENT_SOURCE_DIR}/core/runtime/trap_handler )
|
||||
@@ -233,10 +261,15 @@ if (${PC_SAMPLING_SUPPORT})
|
||||
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${PCS_SRCS} )
|
||||
endif()
|
||||
|
||||
if ( NOT DEFINED IMAGE_SUPPORT AND CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64|loongarch64" )
|
||||
set ( IMAGE_SUPPORT ON )
|
||||
endif()
|
||||
if (UNIX)
|
||||
if ( NOT DEFINED IMAGE_SUPPORT AND CMAKE_SYSTEM_PROCESSOR MATCHES "i?86|x86_64|amd64|AMD64|loongarch64" )
|
||||
set ( IMAGE_SUPPORT ON )
|
||||
endif()
|
||||
set ( IMAGE_SUPPORT ${IMAGE_SUPPORT} CACHE BOOL "Build with image support (default: ON for x86, OFF elsewise)." )
|
||||
else()
|
||||
# Force IMAGE_SUPPORT to be OFF
|
||||
set(IMAGE_SUPPORT OFF CACHE BOOL "Build with image support (forced to OFF)" FORCE)
|
||||
endif()
|
||||
|
||||
## Optional image module defintions.
|
||||
if(${IMAGE_SUPPORT})
|
||||
@@ -305,30 +338,45 @@ if(${IMAGE_SUPPORT})
|
||||
|
||||
endif()
|
||||
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE elf::elf dl pthread rt )
|
||||
if (UNIX)
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE elf::elf dl pthread rt )
|
||||
else()
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE Ws2_32 )
|
||||
target_link_directories(${CORE_RUNTIME_TARGET} PRIVATE ${DXCORE_LIB_PATH})
|
||||
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE dxcore)
|
||||
endif()
|
||||
|
||||
# For static package rocprofiler-register dependency is not required
|
||||
# Link to hsakmt target for shared library builds
|
||||
# Link to hsakmt-staticdrm target for static library builds
|
||||
if( BUILD_SHARED_LIBS )
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt PkgConfig::drm)
|
||||
if( BUILD_THUNK_VIRTIO )
|
||||
message(STATUS "Building with virtio support")
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt_virtio)
|
||||
endif()
|
||||
find_package(rocprofiler-register)
|
||||
if(rocprofiler-register_FOUND)
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HSA_ROCPROFILER_REGISTER=1
|
||||
HSA_VERSION_MAJOR=${VERSION_MAJOR}
|
||||
HSA_VERSION_MINOR=${VERSION_MINOR}
|
||||
HSA_VERSION_PATCH=${VERSION_PATCH})
|
||||
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE rocprofiler-register::rocprofiler-register)
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER ON CACHE INTERNAL "")
|
||||
if (UNIX)
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt PkgConfig::drm)
|
||||
if( BUILD_THUNK_VIRTIO )
|
||||
message(STATUS "Building with virtio support")
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt_virtio)
|
||||
endif()
|
||||
find_package(rocprofiler-register)
|
||||
if(rocprofiler-register_FOUND)
|
||||
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HSA_ROCPROFILER_REGISTER=1
|
||||
HSA_VERSION_MAJOR=${VERSION_MAJOR}
|
||||
HSA_VERSION_MINOR=${VERSION_MINOR}
|
||||
HSA_VERSION_PATCH=${VERSION_PATCH})
|
||||
target_link_libraries(${CORE_RUNTIME_TARGET} PRIVATE rocprofiler-register::rocprofiler-register)
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER ON CACHE INTERNAL "")
|
||||
else()
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER OFF CACHE INTERNAL "")
|
||||
endif() # end rocprofiler-register_FOUND
|
||||
else()
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER OFF CACHE INTERNAL "")
|
||||
endif() # end rocprofiler-register_FOUND
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt)
|
||||
endif()
|
||||
else()
|
||||
include_directories(${drm_INCLUDE_DIRS})
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
|
||||
if (UNIX)
|
||||
include_directories(${drm_INCLUDE_DIRS})
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
|
||||
else()
|
||||
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt-staticdrm::hsakmt-staticdrm)
|
||||
endif()
|
||||
endif()#end BUILD_SHARED_LIBS
|
||||
|
||||
## Set the VERSION and SOVERSION values
|
||||
@@ -350,8 +398,11 @@ if( NOT ${BUILD_SHARED_LIBS} )
|
||||
|
||||
## Add external link requirements.
|
||||
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE hsakmt-staticdrm::hsakmt-staticdrm )
|
||||
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE elf::elf dl pthread rt )
|
||||
|
||||
if (UNIX)
|
||||
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE elf::elf dl pthread rt )
|
||||
else()
|
||||
target_link_libraries ( ${CORE_RUNTIME_NAME} INTERFACE rt )
|
||||
endif()
|
||||
install ( TARGETS ${CORE_RUNTIME_NAME} EXPORT ${CORE_RUNTIME_NAME}Targets )
|
||||
endif()
|
||||
|
||||
@@ -404,10 +455,12 @@ install(FILES ${CMAKE_CURRENT_BINARY_DIR}/${CORE_RUNTIME_NAME}-config.cmake ${CM
|
||||
|
||||
# Install build files needed only when using a static build.
|
||||
if( NOT ${BUILD_SHARED_LIBS} )
|
||||
# libelf find package module
|
||||
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/FindLibElf.cmake ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/COPYING-CMAKE-SCRIPTS
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
|
||||
COMPONENT dev)
|
||||
if (UNIX)
|
||||
# libelf find package module
|
||||
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/FindLibElf.cmake ${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules/COPYING-CMAKE-SCRIPTS
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
|
||||
COMPONENT dev)
|
||||
endif()
|
||||
# Linker script (defines function aliases)
|
||||
install(FILES ${CMAKE_CURRENT_SOURCE_DIR}/${LNKSCR}
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${CORE_RUNTIME_NAME}
|
||||
|
||||
@@ -17,57 +17,78 @@ if (LIBELF_FOUND)
|
||||
return()
|
||||
endif (LIBELF_FOUND)
|
||||
|
||||
find_path (LIBELF_INCLUDE_DIRS
|
||||
NAMES
|
||||
libelf.h
|
||||
PATHS
|
||||
/usr/include
|
||||
/usr/include/libelf
|
||||
/usr/local/include
|
||||
/usr/local/include/libelf
|
||||
/opt/local/include
|
||||
/opt/local/include/libelf
|
||||
ENV CPATH)
|
||||
|
||||
find_library (LIBELF_LIBRARIES
|
||||
NAMES
|
||||
elf
|
||||
PATHS
|
||||
/usr/lib
|
||||
/usr/lib64
|
||||
/usr/local/lib
|
||||
/usr/local/lib64
|
||||
/opt/local/lib
|
||||
/opt/local/lib64
|
||||
ENV LIBRARY_PATH
|
||||
ENV LD_LIBRARY_PATH)
|
||||
|
||||
include (FindPackageHandleStandardArgs)
|
||||
|
||||
|
||||
# handle the QUIETLY and REQUIRED arguments and set LIBELF_FOUND to TRUE if all listed variables are TRUE
|
||||
FIND_PACKAGE_HANDLE_STANDARD_ARGS(LibElf DEFAULT_MSG
|
||||
LIBELF_LIBRARIES
|
||||
LIBELF_INCLUDE_DIRS)
|
||||
|
||||
SET(CMAKE_REQUIRED_LIBRARIES elf)
|
||||
if (CMAKE_CXX_COMPILER_LOADED)
|
||||
INCLUDE(CheckCXXSourceCompiles)
|
||||
CHECK_CXX_SOURCE_COMPILES("#include <libelf.h>
|
||||
int main() {
|
||||
Elf *e = (Elf*)0;
|
||||
size_t sz;
|
||||
elf_getshdrstrndx(e, &sz);
|
||||
return 0;
|
||||
}" ELF_GETSHDRSTRNDX)
|
||||
if (UNIX)
|
||||
find_path (LIBELF_INCLUDE_DIRS
|
||||
NAMES
|
||||
libelf.h
|
||||
PATHS
|
||||
/usr/include
|
||||
/usr/include/libelf
|
||||
/usr/local/include
|
||||
/usr/local/include/libelf
|
||||
/opt/local/include
|
||||
/opt/local/include/libelf
|
||||
ENV CPATH)
|
||||
|
||||
find_library (LIBELF_LIBRARIES
|
||||
NAMES
|
||||
elf
|
||||
PATHS
|
||||
/usr/lib
|
||||
/usr/lib64
|
||||
/usr/local/lib
|
||||
/usr/local/lib64
|
||||
/opt/local/lib
|
||||
/opt/local/lib64
|
||||
ENV LIBRARY_PATH
|
||||
ENV LD_LIBRARY_PATH)
|
||||
|
||||
include (FindPackageHandleStandardArgs)
|
||||
|
||||
|
||||
# handle the QUIETLY and REQUIRED arguments and set LIBELF_FOUND to TRUE if all listed variables are TRUE
|
||||
FIND_PACKAGE_HANDLE_STANDARD_ARGS(LibElf DEFAULT_MSG
|
||||
LIBELF_LIBRARIES
|
||||
LIBELF_INCLUDE_DIRS)
|
||||
|
||||
SET(CMAKE_REQUIRED_LIBRARIES elf)
|
||||
if (CMAKE_CXX_COMPILER_LOADED)
|
||||
INCLUDE(CheckCXXSourceCompiles)
|
||||
CHECK_CXX_SOURCE_COMPILES("#include <libelf.h>
|
||||
int main() {
|
||||
Elf *e = (Elf*)0;
|
||||
size_t sz;
|
||||
elf_getshdrstrndx(e, &sz);
|
||||
return 0;
|
||||
}" ELF_GETSHDRSTRNDX)
|
||||
else()
|
||||
set ( ELF_GETSHDRSTRNDX "TRUE" )
|
||||
endif(CMAKE_CXX_COMPILER_LOADED)
|
||||
|
||||
mark_as_advanced(LIBELF_INCLUDE_DIRS LIBELF_LIBRARIES ELF_GETSHDRSTRNDX)
|
||||
|
||||
if(LIBELF_FOUND)
|
||||
add_library(elf::elf UNKNOWN IMPORTED)
|
||||
set_property(TARGET elf::elf PROPERTY IMPORTED_LOCATION ${LIBELF_LIBRARIES})
|
||||
set_property(TARGET elf::elf PROPERTY INTERFACE_INCLUDE_DIRECTORIES ${LIBELF_INCLUDE_DIRS})
|
||||
endif()
|
||||
else()
|
||||
set ( ELF_GETSHDRSTRNDX "TRUE" )
|
||||
endif(CMAKE_CXX_COMPILER_LOADED)
|
||||
find_path(ROCR_LIBELF_INCLUDE_DIR libelf.h
|
||||
HINTS
|
||||
${AMD_LIBELF_PATH}
|
||||
PATHS
|
||||
${CMAKE_SOURCE_DIR}/hsail-compiler/lib/loaders/elf/utils/libelf
|
||||
${CMAKE_SOURCE_DIR}/../hsail-compiler/lib/loaders/elf/utils/libelf
|
||||
${CMAKE_SOURCE_DIR}/../../hsail-compiler/lib/loaders/elf/utils/libelf
|
||||
NO_DEFAULT_PATH)
|
||||
|
||||
mark_as_advanced(LIBELF_INCLUDE_DIRS LIBELF_LIBRARIES ELF_GETSHDRSTRNDX)
|
||||
|
||||
if(LIBELF_FOUND)
|
||||
add_library(elf::elf UNKNOWN IMPORTED)
|
||||
set_property(TARGET elf::elf PROPERTY IMPORTED_LOCATION ${LIBELF_LIBRARIES})
|
||||
set_property(TARGET elf::elf PROPERTY INTERFACE_INCLUDE_DIRECTORIES ${LIBELF_INCLUDE_DIRS})
|
||||
message("=> LibElf paths:" ${CMAKE_CURRENT_BINARY_DIR} ${ROCR_LIBELF_INCLUDE_DIR})
|
||||
if (${BUILD_SHARED_LIBS})
|
||||
mark_as_advanced(ROCR_LIBELF_INCLUDE_DIR)
|
||||
add_subdirectory("${ROCR_LIBELF_INCLUDE_DIR}" ${CMAKE_CURRENT_BINARY_DIR}/libelf)
|
||||
endif()
|
||||
set(USE_AMD_LIBELF "yes" CACHE FORCE "")
|
||||
set(AMD_ELFTOOLCHAIN_DIR ${ROCR_LIBELF_INCLUDE_DIR}/../..;${ROCR_LIBELF_INCLUDE_DIR}/../common/win32;${ROCR_LIBELF_INCLUDE_DIR}/../common)
|
||||
set(ROCR_LIBELF_INCLUDE_DIR ${ROCR_LIBELF_INCLUDE_DIR};${AMD_ELFTOOLCHAIN_DIR})
|
||||
set(LIBELF_INCLUDE_DIR ${ROCR_LIBELF_INCLUDE_DIR})
|
||||
endif()
|
||||
|
||||
+37
-2
@@ -45,9 +45,11 @@
|
||||
#include <memory>
|
||||
#include <string>
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <amdgpu_drm.h>
|
||||
#include <link.h>
|
||||
#include <sys/ioctl.h>
|
||||
#endif
|
||||
|
||||
#include "hsakmt/hsakmt.h"
|
||||
|
||||
@@ -55,11 +57,16 @@
|
||||
#include "core/inc/amd_memory_region.h"
|
||||
#include "core/inc/runtime.h"
|
||||
|
||||
#if defined(_WIN32)
|
||||
#include "loader/executable.hpp"
|
||||
#endif
|
||||
|
||||
extern r_debug _amdgpu_r_debug;
|
||||
|
||||
namespace rocr {
|
||||
namespace AMD {
|
||||
|
||||
#if defined(__linux__)
|
||||
static_assert(
|
||||
(sizeof(core::ShareableHandle::handle) >= sizeof(amdgpu_bo_handle)) &&
|
||||
(alignof(core::ShareableHandle::handle) >= alignof(amdgpu_bo_handle)),
|
||||
@@ -82,6 +89,7 @@ __forceinline uint64_t drm_perm(hsa_access_permission_t perm) {
|
||||
}
|
||||
|
||||
} // namespace
|
||||
#endif
|
||||
|
||||
KfdDriver::KfdDriver(std::string devnode_name)
|
||||
: core::Driver(core::DriverType::KFD, std::move(devnode_name)) {}
|
||||
@@ -425,6 +433,7 @@ hsa_status_t KfdDriver::ExportDMABuf(void *mem, size_t size, int *dmabuf_fd,
|
||||
|
||||
hsa_status_t KfdDriver::ImportDMABuf(int dmabuf_fd, core::Agent &agent,
|
||||
core::ShareableHandle &handle) {
|
||||
#if defined(__linux__)
|
||||
auto &gpu_agent = static_cast<GpuAgent &>(agent);
|
||||
amdgpu_bo_import_result res;
|
||||
auto ret = DRM_CALL(amdgpu_bo_import(
|
||||
@@ -433,12 +442,16 @@ hsa_status_t KfdDriver::ImportDMABuf(int dmabuf_fd, core::Agent &agent,
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
handle.handle = reinterpret_cast<uint64_t>(res.buf_handle);
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
hsa_status_t KfdDriver::Map(core::ShareableHandle handle, void *mem,
|
||||
size_t offset, size_t size,
|
||||
hsa_access_permission_t perms) {
|
||||
#if defined(__linux__)
|
||||
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
|
||||
if (!ldrm_bo)
|
||||
return HSA_STATUS_ERROR;
|
||||
@@ -446,12 +459,15 @@ hsa_status_t KfdDriver::Map(core::ShareableHandle handle, void *mem,
|
||||
if (DRM_CALL(amdgpu_bo_va_op(ldrm_bo, offset, size, reinterpret_cast<uint64_t>(mem),
|
||||
drm_perm(perms), AMDGPU_VA_OP_MAP)) != 0)
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
hsa_status_t KfdDriver::Unmap(core::ShareableHandle handle, void *mem,
|
||||
size_t offset, size_t size) {
|
||||
#if defined(__linux__)
|
||||
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
|
||||
if (!ldrm_bo)
|
||||
return HSA_STATUS_ERROR;
|
||||
@@ -459,11 +475,14 @@ hsa_status_t KfdDriver::Unmap(core::ShareableHandle handle, void *mem,
|
||||
if (DRM_CALL(amdgpu_bo_va_op(ldrm_bo, offset, size, reinterpret_cast<uint64_t>(mem), 0,
|
||||
AMDGPU_VA_OP_UNMAP)) != 0)
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
hsa_status_t KfdDriver::ReleaseShareableHandle(core::ShareableHandle &handle) {
|
||||
#if defined(__linux__)
|
||||
const auto ldrm_bo = reinterpret_cast<amdgpu_bo_handle>(handle.handle);
|
||||
if (!ldrm_bo)
|
||||
return HSA_STATUS_ERROR;
|
||||
@@ -473,6 +492,9 @@ hsa_status_t KfdDriver::ReleaseShareableHandle(core::ShareableHandle &handle) {
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
handle = {};
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -650,6 +672,7 @@ hsa_status_t KfdDriver::DeregisterMemory(void* ptr) const {
|
||||
hsa_status_t KfdDriver::MakeMemoryResident(const void* mem, size_t size, uint64_t* alternate_va,
|
||||
const HsaMemMapFlags* mem_flags, uint32_t num_nodes,
|
||||
const uint32_t* nodes) const {
|
||||
#if defined(__linux__)
|
||||
if (mem_flags == nullptr && nodes == nullptr) {
|
||||
if (HSAKMT_CALL(hsaKmtMapMemoryToGPU(const_cast<void*>(mem), size, alternate_va)) !=
|
||||
HSAKMT_STATUS_SUCCESS) {
|
||||
@@ -663,7 +686,19 @@ hsa_status_t KfdDriver::MakeMemoryResident(const void* mem, size_t size, uint64_
|
||||
debug_print("Invalid memory flags ptr:%p nodes ptr:%p\n", mem_flags, nodes);
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
}
|
||||
#else
|
||||
assert(num_nodes > 0);
|
||||
assert(nodes != NULL);
|
||||
|
||||
*alternate_va = 0;
|
||||
const HSAKMT_STATUS status =
|
||||
HSAKMT_CALL(hsaKmtMapMemoryToGPUNodes(const_cast<void*>(mem), size, alternate_va, *mem_flags,
|
||||
num_nodes, const_cast<uint32_t*>(nodes)));
|
||||
|
||||
if (status != HSAKMT_STATUS_SUCCESS) {
|
||||
return HSA_STATUS_ERROR;
|
||||
}
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2024-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -43,11 +43,10 @@
|
||||
#ifndef HSA_RUNTME_CORE_INC_AMD_AVAILABLE_DRIVERS_H_
|
||||
#define HSA_RUNTME_CORE_INC_AMD_AVAILABLE_DRIVERS_H_
|
||||
|
||||
#ifdef __linux__
|
||||
|
||||
#include "core/inc/amd_kfd_driver.h"
|
||||
#include "core/inc/amd_xdna_driver.h"
|
||||
|
||||
#ifdef __linux__
|
||||
#include "core/inc/amd_xdna_driver.h"
|
||||
#endif
|
||||
|
||||
#endif // header guard
|
||||
|
||||
@@ -116,6 +116,7 @@ class BlitKernel : public core::Blit {
|
||||
virtual bool GangLeader() const override { return false; }
|
||||
|
||||
const uint16_t kInvalidPacketHeader = HSA_PACKET_TYPE_INVALID;
|
||||
|
||||
private:
|
||||
union KernelArgs {
|
||||
struct __ALIGNED__(16) {
|
||||
|
||||
@@ -482,6 +482,7 @@ public:
|
||||
|
||||
/// @brief Finds the handle of executable to which @p device_address
|
||||
/// belongs. Return NULL handle if device address is invalid.
|
||||
#undef FindExecutable
|
||||
virtual hsa_executable_t FindExecutable(uint64_t device_address) = 0;
|
||||
|
||||
/// @brief Returns host address given @p device_address. If @p device_address
|
||||
|
||||
@@ -51,11 +51,12 @@
|
||||
#include <tuple>
|
||||
#include <utility>
|
||||
#include <thread>
|
||||
#include <sys/un.h>
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <sys/un.h>
|
||||
#include <xf86drm.h>
|
||||
#include <amdgpu.h>
|
||||
#else
|
||||
#include <hsakmt/drm/amdgpu.h>
|
||||
#endif
|
||||
|
||||
#include "core/inc/hsa_ext_interface.h"
|
||||
@@ -232,6 +233,7 @@ class Runtime {
|
||||
/// @param [in] size Copy size in bytes.
|
||||
///
|
||||
/// @retval ::HSA_STATUS_SUCCESS if memory copy is successful and completed.
|
||||
#undef CopyMemory
|
||||
hsa_status_t CopyMemory(void* dst, const void* src, size_t size);
|
||||
|
||||
/// @brief Non-blocking memory copy from src to dst.
|
||||
@@ -302,6 +304,7 @@ class Runtime {
|
||||
/// @param [in] count Number of uint32_t element to be set.
|
||||
///
|
||||
/// @retval ::HSA_STATUS_SUCCESS if memory fill is successful and completed.
|
||||
#undef FillMemory
|
||||
hsa_status_t FillMemory(void* ptr, uint32_t value, size_t count);
|
||||
|
||||
/// @brief Set agents as the whitelist to access ptr.
|
||||
@@ -517,7 +520,8 @@ class Runtime {
|
||||
|
||||
static bool IsGPUDriver(DriverType driver_type) {
|
||||
return driver_type == core::DriverType::KFD
|
||||
#ifdef HSAKMT_VIRTIO_ENABLED
|
||||
|
||||
#if defined(HSAKMT_VIRTIO_ENABLED) && defined(__linux__)
|
||||
|| driver_type == core::DriverType::KFD_VIRTIO
|
||||
#endif
|
||||
;
|
||||
|
||||
@@ -44,7 +44,11 @@
|
||||
#define HSA_RUNTIME_CORE_INC_THUNK_LOADER_H
|
||||
|
||||
#include <string>
|
||||
#if defined(__linux__)
|
||||
#include <amdgpu.h>
|
||||
#else
|
||||
#include "hsakmt/drm/amdgpu.h"
|
||||
#endif
|
||||
#include "hsakmt/hsakmttypes.h"
|
||||
|
||||
class DtifPlatform;
|
||||
|
||||
@@ -50,8 +50,11 @@
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
#include <algorithm>
|
||||
#ifdef _WIN32
|
||||
#define WIN32_NO_STATUS
|
||||
#include <Windows.h>
|
||||
#undef WIN32_NO_STATUS
|
||||
#endif
|
||||
|
||||
#include <stdio.h>
|
||||
@@ -967,7 +970,7 @@ void AqlQueue::HandleInsufficientScratch(hsa_signal_value_t& error_code,
|
||||
maxGroupsPerEngine < 16 &&
|
||||
lanes_per_group * maxGroupsPerEngine < 256) {
|
||||
uint64_t groups_per_interleave = (256 + lanes_per_group - 1) / lanes_per_group;
|
||||
maxGroupsPerEngine = Min(groups_per_interleave, 16ul);
|
||||
maxGroupsPerEngine = Min(groups_per_interleave, uint64_t(16ul));
|
||||
}
|
||||
|
||||
// Populate all engines at max group occupancy, then clip down to device limits.
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -903,9 +903,13 @@ void BlitKernel::PopulateQueue(uint64_t index, uint64_t code_handle, void* args,
|
||||
// Ensure the packet body is written as header may get reordered when writing over PCIE
|
||||
_mm_sfence();
|
||||
}
|
||||
#if defined(__linux__)
|
||||
__atomic_store_n(&(queue_buffer[index & queue_bitmask_].full_header),
|
||||
kDispatchPacketHeader | packet.setup << 16, __ATOMIC_RELEASE);
|
||||
|
||||
#else
|
||||
std::atomic_ref<uint32_t> atomic_header(queue_buffer[index & queue_bitmask_].full_header);
|
||||
atomic_header.store(kDispatchPacketHeader | packet.setup << 16, std::memory_order_release);
|
||||
#endif
|
||||
LogPrint(HSA_AMD_LOG_FLAG_AQL,
|
||||
"HWq=%p, id=%lu, Dispatch Header = "
|
||||
"0x%x (type=%d, barrier=%d, acquire=%d, release=%d), "
|
||||
|
||||
@@ -47,6 +47,7 @@
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <limits>
|
||||
#include <core/util/utils.h>
|
||||
|
||||
#include "core/inc/amd_gpu_agent.h"
|
||||
#include "core/inc/amd_memory_region.h"
|
||||
@@ -855,7 +856,7 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
|
||||
// width | 16 ensures that we don't return a higher element than is supported and avoids
|
||||
// issues with 0.
|
||||
auto maxAlignedElement = [](size_t width) {
|
||||
return __builtin_ctz(width | 16);
|
||||
return rocr::os::Ctz(width | 16);
|
||||
};
|
||||
|
||||
// GFX12 or later use a different packet format that is incompatible (fields changed in size and location).
|
||||
@@ -872,7 +873,7 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
|
||||
// Find maximum element that describes the pitch and slice.
|
||||
// Pitch and slice must both be represented in units of elements. No element larger than this
|
||||
// may be used in any tile as the pitches would not be exactly represented.
|
||||
int max_ele = Min(maxAlignedElement(src->pitch), maxAlignedElement(dst->pitch));
|
||||
auto max_ele = Min(maxAlignedElement(src->pitch), maxAlignedElement(dst->pitch));
|
||||
if (range->z != 1) // Only need to consider slice if HW will copy along Z.
|
||||
max_ele = Min(max_ele, maxAlignedElement(src->slice), maxAlignedElement(dst->slice));
|
||||
|
||||
@@ -895,8 +896,8 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
|
||||
src and dst base has already been checked for DWORD alignment so we only need to consider the
|
||||
offset here.
|
||||
*/
|
||||
int min_ele = Min(max_ele, maxAlignedElement(range->x), maxAlignedElement(src_offset->x % 4),
|
||||
maxAlignedElement(dst_offset->x % 4));
|
||||
auto min_ele = Min(max_ele, maxAlignedElement(range->x), maxAlignedElement(src_offset->x % 4),
|
||||
maxAlignedElement(dst_offset->x % 4));
|
||||
|
||||
// Check that pitch and slice can be represented in the tile with the smallest element
|
||||
if ((src->pitch >> min_ele) > max_pitch || (dst->pitch >> min_ele) > max_pitch)
|
||||
@@ -916,8 +917,8 @@ void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>&
|
||||
|
||||
// Get largest element which describes the start of this tile after its base address has
|
||||
// been aligned. Base addresses must be DWORD (4 byte) aligned.
|
||||
int aligned_ele = Min(maxAlignedElement((src_offset->x + x) % 4),
|
||||
maxAlignedElement((dst_offset->x + x) % 4), max_ele);
|
||||
auto aligned_ele = Min(maxAlignedElement((src_offset->x + x) % 4),
|
||||
maxAlignedElement((dst_offset->x + x) % 4), max_ele);
|
||||
|
||||
// Get largest permissible element which exactly covers width
|
||||
int element = Min(maxAlignedElement(width), aligned_ele);
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2023, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -1026,8 +1026,8 @@ hsa_status_t GpuAgent::DmaCopy(void* dst, core::Agent& dst_agent,
|
||||
std::vector<core::Signal*>& dep_signals,
|
||||
core::Signal& out_signal) {
|
||||
// Recommended SDMA engine copies only have gang factor 1
|
||||
uint32_t rec_sdma_eng = ffs(rec_sdma_eng_id_peers_info_[dst_agent.public_handle().handle]);
|
||||
|
||||
uint32_t rec_sdma_eng =
|
||||
rocr::os::Ffs(rec_sdma_eng_id_peers_info_[dst_agent.public_handle().handle]);
|
||||
if (rec_sdma_eng)
|
||||
return DmaCopyOnEngine(dst, dst_agent, src, src_agent, size,
|
||||
dep_signals, out_signal, rec_sdma_eng, false);
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -44,12 +44,16 @@
|
||||
#include "core/inc/runtime.h"
|
||||
|
||||
#include <assert.h>
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#include <linux/limits.h>
|
||||
#include <sys/mman.h>
|
||||
#include <stdlib.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
#include <stdlib.h>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
@@ -92,7 +96,7 @@ std::string EncodePathname(const char *file_path) {
|
||||
}
|
||||
|
||||
std::string GetUriFromMemoryAddress(const void *memory, size_t size) {
|
||||
pid_t pid = getpid();
|
||||
int pid = getpid();
|
||||
std::ostringstream uri_stream;
|
||||
uri_stream << "memory://" << pid
|
||||
<< "#offset=0x" << std::hex << (uintptr_t)memory << std::dec
|
||||
@@ -313,23 +317,7 @@ hsa_status_t CodeObjectReaderImpl::SetFile(
|
||||
code_object_size = _code_object_size;
|
||||
is_mmap = true;
|
||||
#else
|
||||
if (__lseek__(_code_object_file_descriptor, 0, SEEK_SET) == (off_t)-1) {
|
||||
return HSA_STATUS_ERROR_INVALID_FILE;
|
||||
}
|
||||
|
||||
std::unique_ptr<unsigned char> memory(new unsigned char[_code_object_size]);
|
||||
if (!memory) {
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
}
|
||||
|
||||
if (__read__(_code_object_file_descriptor, mmap_memory,
|
||||
_code_object_size) != _code_object_size) {
|
||||
return HSA_STATUS_ERROR_INVALID_FILE;
|
||||
}
|
||||
mmap_memory = memory.release();
|
||||
mmap_size = _code_object_size;
|
||||
code_object_memory = memory;
|
||||
code_object_size = _code_object_size;
|
||||
//@todo May need an implementation in Windows
|
||||
#endif // !defined(_WIN32) && !defined(_WIN64)
|
||||
|
||||
uri = GetUriFromFile(_code_object_file_descriptor, _code_object_offset,
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -40,6 +40,11 @@
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
#include "core/inc/amd_memory_region.h"
|
||||
|
||||
#include <algorithm>
|
||||
@@ -48,15 +53,15 @@
|
||||
#include "core/inc/amd_cpu_agent.h"
|
||||
#include "core/inc/amd_gpu_agent.h"
|
||||
#include "core/util/utils.h"
|
||||
#include "core/util/os.h"
|
||||
#include "core/inc/exceptions.h"
|
||||
#include <unistd.h>
|
||||
|
||||
namespace rocr {
|
||||
namespace AMD {
|
||||
|
||||
// Tracks aggregate size of system memory available on platform
|
||||
size_t MemoryRegion::max_sysmem_alloc_size_ = 0;
|
||||
const size_t MemoryRegion::kPageSize_ = sysconf(_SC_PAGESIZE);
|
||||
const size_t MemoryRegion::kPageSize_ = os::PageSize();
|
||||
|
||||
MemoryRegion::MemoryRegion(bool fine_grain, bool kernarg, bool full_profile,
|
||||
bool extended_scope_fine_grain, bool user_visible, core::Agent* owner,
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -58,8 +58,6 @@
|
||||
#include <unordered_map>
|
||||
#include <vector>
|
||||
|
||||
#include <link.h>
|
||||
|
||||
#include "core/inc/amd_aie_agent.h"
|
||||
#include "core/inc/amd_available_drivers.h"
|
||||
#include "core/inc/amd_cpu_agent.h"
|
||||
@@ -72,7 +70,12 @@
|
||||
#include "core/inc/amd_virtio_driver.h"
|
||||
#endif
|
||||
|
||||
extern r_debug _amdgpu_r_debug;
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#else
|
||||
#include "loader/executable.hpp"
|
||||
#endif
|
||||
extern r_debug _amdgpu_r_debug_r;
|
||||
|
||||
namespace rocr {
|
||||
namespace AMD {
|
||||
@@ -81,17 +84,17 @@ namespace {
|
||||
|
||||
const std::array<std::function<hsa_status_t(std::unique_ptr<core::Driver>&)>,
|
||||
#if _WIN32
|
||||
0
|
||||
1
|
||||
#elif __linux__
|
||||
static_cast<size_t>(core::DriverType::NUM_DRIVER_TYPES)
|
||||
#endif
|
||||
>
|
||||
discover_driver_funcs = {
|
||||
KfdDriver::DiscoverDriver
|
||||
#ifdef __linux__
|
||||
KfdDriver::DiscoverDriver,
|
||||
XdnaDriver::DiscoverDriver,
|
||||
, XdnaDriver::DiscoverDriver
|
||||
#ifdef HSAKMT_VIRTIO_ENABLED
|
||||
KfdVirtioDriver::DiscoverDriver,
|
||||
, KfdVirtioDriver::DiscoverDriver
|
||||
#endif
|
||||
#endif
|
||||
};
|
||||
@@ -181,8 +184,10 @@ GpuAgent* DiscoverGpu(HSAuint32 node_id, HsaNodeProperties& node_prop, bool xnac
|
||||
}
|
||||
|
||||
void DiscoverAie(uint32_t node_id, HsaNodeProperties& node_prop) {
|
||||
#if defined(__linux__)
|
||||
AieAgent* aie = new AieAgent(node_id, node_prop);
|
||||
core::Runtime::runtime_singleton_->RegisterAgent(aie, true);
|
||||
#endif
|
||||
}
|
||||
|
||||
void RegisterLinkInfo(const std::unique_ptr<core::Driver>& driver, uint32_t node_id,
|
||||
|
||||
+10
-2
@@ -3,7 +3,7 @@
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2014-2023, Advanced Micro Devices, Inc. All rights reserved.
|
||||
## Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
@@ -148,11 +148,19 @@ function(generate_bytecodeStrm HeaderFILE)
|
||||
COPYONLY)
|
||||
|
||||
# Add a custom command to generate the header file
|
||||
if (UNIX)
|
||||
add_custom_command(OUTPUT ${HeaderFILE}.h
|
||||
COMMAND ${CMAKE_CURRENT_BINARY_DIR}/create_blit_shader_header.sh ${ARG_LIST} ${HSACO_TARG_LIST}
|
||||
COMMENT "Collating blit shaders..."
|
||||
DEPENDS ${HSACO_TARG_LIST} ${CMAKE_CURRENT_BINARY_DIR}/create_blit_shader_header.sh)
|
||||
|
||||
else()
|
||||
find_package(Python3 COMPONENTS Interpreter REQUIRED)
|
||||
add_custom_command(
|
||||
OUTPUT ${HeaderFILE}.h
|
||||
COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/create_blit_shader_header.py ${ARG_LIST} ${HSACO_TARG_LIST}
|
||||
COMMENT "Collating blit shaders..."
|
||||
DEPENDS ${HSACO_TARG_LIST} create_blit_shader_header.py)
|
||||
endif()
|
||||
# Add a custom target that depends on the header file
|
||||
add_custom_target(${HeaderFILE} DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/${HeaderFILE}.h)
|
||||
|
||||
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
################################################################################
|
||||
##
|
||||
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
|
||||
##
|
||||
## SPDX-License-Identifier: MIT
|
||||
##
|
||||
################################################################################
|
||||
import sys
|
||||
|
||||
def GetSize(fileobject):
|
||||
fileobject.seek(0,2) # move the cursor to the end of the file
|
||||
size = fileobject.tell()
|
||||
return size
|
||||
|
||||
def DumpFile(header, input_name):
|
||||
try:
|
||||
with open(input_name, "rb") as binary_file:
|
||||
# Read the entire content of the file as bytes
|
||||
binary_data = binary_file.read()
|
||||
file_size = GetSize(binary_file)
|
||||
#print(f"Binary size: {file_size}")
|
||||
# Reset file pointer
|
||||
binary_file.seek(0)
|
||||
parts = input_name.split('.')
|
||||
file_name = parts[0]
|
||||
content = f"unsigned char {file_name}""[] = {\n "
|
||||
|
||||
header.write(content)
|
||||
line = 0
|
||||
count = 0
|
||||
for byte_value in binary_data:
|
||||
count += 1
|
||||
padded_hex = '{:02x}'.format(byte_value)
|
||||
if (count != file_size):
|
||||
header.write(f"0x{padded_hex},")
|
||||
else:
|
||||
header.write(f"0x{padded_hex}")
|
||||
line += 1
|
||||
if (line == 12):
|
||||
header.write(f"\n ")
|
||||
line = 0
|
||||
else:
|
||||
header.write(f" ")
|
||||
|
||||
header.write("\n};\nunsigned int "f"{file_name}_len = {file_size};\n")
|
||||
|
||||
except FileNotFoundError:
|
||||
print(f"Error: The file {input_name} was not found.")
|
||||
except Exception as e:
|
||||
print(f"An error occurred: {e}")
|
||||
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
header_name = sys.argv[1];
|
||||
with open(header_name, 'w') as header:
|
||||
header.write("//==============================================================================\n")
|
||||
header.write("// This file is automatically generated during build process, don't modify it\n")
|
||||
header.write("//==============================================================================\n\n")
|
||||
header.write("namespace rocr {\n")
|
||||
header.write("namespace AMD {\n\n")
|
||||
|
||||
for i, arg in enumerate(sys.argv):
|
||||
if (i > 1):
|
||||
#print(f"File {i}: {arg}\n")
|
||||
DumpFile(header, arg)
|
||||
header.write("} // namespace AMD\n")
|
||||
header.write("} // namespace rocr\n\n")
|
||||
|
||||
else:
|
||||
print("Empty arguments!")
|
||||
|
||||
@@ -835,12 +835,14 @@ hsa_status_t hsa_amd_agent_iterate_memory_pools(
|
||||
reinterpret_cast<hsa_status_t (*)(hsa_region_t memory_pool,
|
||||
void *data)>(callback),
|
||||
data);
|
||||
#if defined(__linux__)
|
||||
case core::Agent::kAmdAieDevice:
|
||||
return reinterpret_cast<const AMD::AieAgent *>(agent)->VisitRegion(
|
||||
false,
|
||||
reinterpret_cast<hsa_status_t (*)(hsa_region_t memory_pool,
|
||||
void *data)>(callback),
|
||||
data);
|
||||
#endif
|
||||
case core::Agent::kAmdGpuDevice:
|
||||
return reinterpret_cast<const AMD::GpuAgentInt *>(agent)->VisitRegion(
|
||||
false,
|
||||
|
||||
+3
-2
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -42,8 +42,9 @@
|
||||
|
||||
#include "core/inc/hsa_ven_amd_loader_impl.h"
|
||||
|
||||
#include "core/inc/amd_hsa_loader.hpp"
|
||||
#include "core/inc/runtime.h"
|
||||
#include "core/inc/amd_gpu_agent.h"
|
||||
#include "core/inc/amd_hsa_loader.hpp"
|
||||
|
||||
namespace rocr {
|
||||
|
||||
|
||||
@@ -48,13 +48,19 @@
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <list>
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#include <dlfcn.h>
|
||||
#include <amdgpu_drm.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/socket.h>
|
||||
#include <sys/un.h>
|
||||
#else
|
||||
#define debug_warning(__VA_ARGS__)
|
||||
#endif
|
||||
#include <iostream>
|
||||
#include <thread>
|
||||
#include <chrono>
|
||||
|
||||
#include "core/inc/runtime.h"
|
||||
#include "core/inc/hsa_table_interface.h"
|
||||
@@ -97,8 +103,12 @@
|
||||
ROCPROFILER_REGISTER_DEFINE_IMPORT(hsa, ROCP_REG_VERSION)
|
||||
#endif
|
||||
|
||||
#if defined(__linux__)
|
||||
const char rocrbuildid[] __attribute__((used)) = "ROCR BUILD ID: " STRING(ROCR_BUILD_ID);
|
||||
|
||||
#else
|
||||
#include "loader/executable.hpp"
|
||||
const char rocrbuildid[] = "ROCR BUILD ID: " STRING(ROCR_BUILD_ID);
|
||||
#endif
|
||||
extern r_debug _amdgpu_r_debug;
|
||||
|
||||
namespace rocr {
|
||||
@@ -591,7 +601,7 @@ hsa_status_t Runtime::CopyMemoryOnEngine(void* dst, core::Agent* dst_agent, cons
|
||||
core::Agent* copy_agent = (src_gpu) ? src_agent : dst_agent;
|
||||
|
||||
// engine_id is single bitset unique.
|
||||
int engine_offset = ffs(engine_id);
|
||||
int engine_offset = rocr::os::Ffs(engine_id);
|
||||
if (!engine_id || !!((engine_id >> engine_offset))) {
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
}
|
||||
@@ -1178,6 +1188,7 @@ hsa_status_t Runtime::SetPtrInfoData(const void* ptr, void* userptr) {
|
||||
|
||||
// Send the dmabuf_fd to from process via Unix socket
|
||||
static int SendDmaBufFd(int socket, int dmabuf_fd) {
|
||||
#if defined(__linux__)
|
||||
char iov_buf[1];
|
||||
struct msghdr msg = {0};
|
||||
char buf[CMSG_SPACE(sizeof(dmabuf_fd))];
|
||||
@@ -1205,10 +1216,15 @@ static int SendDmaBufFd(int socket, int dmabuf_fd) {
|
||||
ssize_t sent = sendmsg(socket, &msg, 0);
|
||||
|
||||
return (sent < 0) ? -1 : 0;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
// Receive the dmabuf_fd to from process via Unix socket
|
||||
static int ReceiveDmaBufFd(int socket) {
|
||||
#if defined(__linux__)
|
||||
struct msghdr msg = {0};
|
||||
|
||||
// The struct iovec is needed, even if it points to minimal data
|
||||
@@ -1233,6 +1249,10 @@ static int ReceiveDmaBufFd(int socket) {
|
||||
memcpy(&fd, CMSG_DATA(cmsg), sizeof(fd));
|
||||
|
||||
return fd;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
#define IPC_SOCK_SERVER_DMABUF_FD_HANDLE_LENGTH 64
|
||||
@@ -1363,6 +1383,7 @@ hsa_status_t Runtime::IPCCreate(void* ptr, size_t len, hsa_amd_ipc_memory_t* han
|
||||
close(dmabuf_fd);
|
||||
|
||||
ScopedAcquire<KernelMutex> lock(&ipc_sock_server_lock_);
|
||||
#if defined(__linux__)
|
||||
if (!ipc_sock_server_conns_.size()) { // create new runtime socket server
|
||||
struct sockaddr_un address;
|
||||
ipc_sock_server_fd_ = socket(AF_UNIX, SOCK_STREAM, 0);
|
||||
@@ -1393,7 +1414,9 @@ hsa_status_t Runtime::IPCCreate(void* ptr, size_t len, hsa_amd_ipc_memory_t* han
|
||||
// as the attach life cycle is unknown.
|
||||
os::CreateThread(AsyncIPCSockServerConnLoop, NULL);
|
||||
}
|
||||
|
||||
#else
|
||||
assert(!"Unimplemented! Do we really need this?");
|
||||
#endif
|
||||
ipc_sock_server_conns_[reinterpret_cast<uint64_t>(ptr)] = len;
|
||||
|
||||
// TODO: fragment block discard for better memory performance causes memory violations
|
||||
@@ -1406,7 +1429,6 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
|
||||
amdgpu_bo_import_result *res,
|
||||
unsigned int numNodes, HSAuint32 *nodes,
|
||||
void **importAddress, HSAuint64 *importSize) {
|
||||
struct sockaddr_un address;
|
||||
int dmabuf_fd = -1, socket_fd = socket(AF_UNIX, SOCK_STREAM, 0);
|
||||
assert(socket_fd > -1 && "DMA buffer could not be imported for IPC!");
|
||||
if (socket_fd == -1) return -1;
|
||||
@@ -1420,13 +1442,15 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
|
||||
if (status) return -1;
|
||||
|
||||
char buf[IPC_SOCK_SERVER_DMABUF_FD_HANDLE_LENGTH];
|
||||
memset(&address, 0, sizeof(struct sockaddr_un));
|
||||
memset(buf, 0, sizeof(buf));
|
||||
int timeoutLimitMs = 10000, timeoutMs = 0, timeoutIntervalMs = 1;
|
||||
#if defined(__linux__)
|
||||
struct sockaddr_un address;
|
||||
memset(&address, 0, sizeof(struct sockaddr_un));
|
||||
address.sun_family = AF_UNIX;
|
||||
snprintf(address.sun_path, IPC_SOCK_SERVER_NAME_LENGTH, "xhsa%i", conn_handle);
|
||||
address.sun_path[0] = 0; // first NULL char creates unlisted abstract socket
|
||||
|
||||
int timeoutLimitMs = 10000, timeoutMs = 0, timeoutIntervalMs = 1;
|
||||
while (timeoutMs < timeoutLimitMs) {
|
||||
if (connect(socket_fd, (struct sockaddr *) &address, sizeof(struct sockaddr_un))) {
|
||||
timeoutMs += timeoutIntervalMs;
|
||||
@@ -1435,7 +1459,9 @@ int Runtime::IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
#else
|
||||
assert(!"Unimplmented!");
|
||||
#endif
|
||||
MAKE_SCOPE_GUARD([&]() { close(socket_fd); });
|
||||
|
||||
if (timeoutMs >= timeoutLimitMs) return -1;
|
||||
@@ -1545,6 +1571,7 @@ hsa_status_t Runtime::IPCAttach(const hsa_amd_ipc_memory_t* handle, size_t len,
|
||||
dmaBufFDHandle = (dmaBufFDHandleHi << 32) | dmaBufFDHandleLo;
|
||||
}
|
||||
|
||||
#if defined(__linux__)
|
||||
if (num_agents == 0) {
|
||||
amdgpu_bo_import_result res;
|
||||
bool isDmabufSysMem = ipc_dmabuf_supported_ && importHandle.handle[3];
|
||||
@@ -1575,6 +1602,9 @@ hsa_status_t Runtime::IPCAttach(const hsa_amd_ipc_memory_t* handle, size_t len,
|
||||
*mapped_ptr = importAddress;
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
|
||||
HSAuint32* nodes = nullptr;
|
||||
if (num_agents > tinyArraySize)
|
||||
@@ -1602,6 +1632,7 @@ hsa_status_t Runtime::IPCDetach(void* ptr) {
|
||||
const auto& it = allocation_map_.find(ptr);
|
||||
if (it != allocation_map_.end()) {
|
||||
if (it->second.region != nullptr) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
#if defined(__linux__)
|
||||
if (it->second.ldrm_bo) {
|
||||
if (DRM_CALL(amdgpu_bo_va_op(it->second.ldrm_bo, 0, it->second.size,
|
||||
reinterpret_cast<uint64_t>(ptr), 0, AMDGPU_VA_OP_UNMAP)))
|
||||
@@ -1610,6 +1641,9 @@ hsa_status_t Runtime::IPCDetach(void* ptr) {
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
ldrmImportCleaned = true;
|
||||
}
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
allocation_map_.erase(it);
|
||||
lock.Release(); // Can't hold memory lock when using pointer info.
|
||||
|
||||
@@ -2325,6 +2359,7 @@ int fn_amdgpu_device_get_fd_nosupport(HsaAMDGPUDeviceHandle device_handle) {
|
||||
|
||||
int Runtime::GetAmdgpuDeviceArgs(Agent *agent, ShareableHandle handle,
|
||||
int *drm_fd, uint64_t *cpu_addr) {
|
||||
#if defined(__linux__)
|
||||
int renderFd = fn_amdgpu_device_get_fd(static_cast<AMD::GpuAgent*>(agent)->libDrmDev());
|
||||
if (renderFd < 0) return HSA_STATUS_ERROR;
|
||||
|
||||
@@ -2343,6 +2378,9 @@ int Runtime::GetAmdgpuDeviceArgs(Agent *agent, ShareableHandle handle,
|
||||
|
||||
*drm_fd = renderFd;
|
||||
*cpu_addr = args.out.addr_ptr;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
@@ -2353,6 +2391,7 @@ void Runtime::CheckVirtualMemApiSupport() {
|
||||
if (kfd_version.KernelInterfaceMajorVersion > 1 ||
|
||||
(kfd_version.KernelInterfaceMajorVersion == 1 &&
|
||||
kfd_version.KernelInterfaceMinorVersion >= 15)) {
|
||||
#if defined(__linux__)
|
||||
char* error;
|
||||
|
||||
fn_amdgpu_device_get_fd =
|
||||
@@ -2365,6 +2404,9 @@ void Runtime::CheckVirtualMemApiSupport() {
|
||||
} else {
|
||||
virtual_mem_api_supported_ = true;
|
||||
}
|
||||
#else
|
||||
virtual_mem_api_supported_ = false;
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2379,7 +2421,7 @@ void Runtime::InitIPCDmaBufSupport() {
|
||||
|
||||
GetSystemInfo(HSA_AMD_SYSTEM_INFO_DMABUF_SUPPORTED, &dmabuf_supported);
|
||||
if (!dmabuf_supported) return;
|
||||
|
||||
#if defined(__linux__)
|
||||
char* error;
|
||||
fn_amdgpu_device_get_fd =
|
||||
(int (*)(HsaAMDGPUDeviceHandle device_handle))dlsym(
|
||||
@@ -2391,6 +2433,9 @@ void Runtime::InitIPCDmaBufSupport() {
|
||||
} else {
|
||||
ipc_dmabuf_supported_ = !flag().enable_ipc_mode_legacy();
|
||||
}
|
||||
#else
|
||||
ipc_dmabuf_supported_ = false;
|
||||
#endif
|
||||
}
|
||||
|
||||
void Runtime::LoadTools() {
|
||||
@@ -3237,15 +3282,14 @@ hsa_status_t Runtime::VMemoryAddressReserve(void** va, size_t size, uint64_t add
|
||||
void* addr = (void*)address;
|
||||
HsaMemFlags memFlags = {};
|
||||
|
||||
if (!alignment)
|
||||
alignment = sysconf(_SC_PAGE_SIZE);
|
||||
if (!alignment) alignment = rocr::os::PageSize();
|
||||
|
||||
ScopedAcquire<KernelSharedMutex> lock(&memory_lock_);
|
||||
|
||||
if (flags & HSA_AMD_VMEM_ADDRESS_NO_REGISTER) {
|
||||
size_t requested = size + alignment - sysconf(_SC_PAGE_SIZE);
|
||||
auto mem = mmap(addr, requested, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE | MAP_NORESERVE, -1, 0);
|
||||
if (mem == MAP_FAILED)
|
||||
size_t requested = size + alignment - rocr::os::PageSize();
|
||||
auto mem = rocr::os::ReserveMemory(addr, requested, alignment, rocr::os::MEM_PROT_RW);
|
||||
if (mem == nullptr)
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
|
||||
auto aligned = AlignUp(mem, alignment);
|
||||
@@ -3253,8 +3297,10 @@ hsa_status_t Runtime::VMemoryAddressReserve(void** va, size_t size, uint64_t add
|
||||
// Hint to enable THP for large host allocations which can help in performance gain
|
||||
constexpr size_t kLargePageSize = 2*1024*1024;
|
||||
if (size >= kLargePageSize) {
|
||||
#if defined(__linux__)
|
||||
if (madvise(aligned, size, MADV_HUGEPAGE))
|
||||
debug_warning(false && "madvise with MADV_HUGEPAGE failed");
|
||||
#endif
|
||||
}
|
||||
|
||||
reserved_address_map_[aligned] = AddressHandle(mem, size, false);
|
||||
@@ -3292,10 +3338,11 @@ hsa_status_t Runtime::VMemoryAddressFree(void* va, size_t size) {
|
||||
if (it->second.use_count > 0) return HSA_STATUS_ERROR_RESOURCE_FREE;
|
||||
|
||||
if (it->second.registered) {
|
||||
if (HSAKMT_CALL(hsaKmtFreeMemory(it->second.os_addr, size)) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
|
||||
} else {
|
||||
if (munmap(it->second.os_addr, size)) return HSA_STATUS_ERROR;
|
||||
if (HSAKMT_CALL(hsaKmtFreeMemory(it->second.os_addr, size)) != HSAKMT_STATUS_SUCCESS)
|
||||
return HSA_STATUS_ERROR;
|
||||
}
|
||||
else if (!rocr::os::ReleaseMemory(it->second.os_addr, size))
|
||||
return HSA_STATUS_ERROR;
|
||||
|
||||
reserved_address_map_.erase(it);
|
||||
return HSA_STATUS_SUCCESS;
|
||||
@@ -3488,10 +3535,10 @@ hsa_status_t Runtime::VMemoryHandleUnmap(void* va, size_t size) {
|
||||
}
|
||||
|
||||
Runtime::MappedHandleAllowedAgent::MappedHandleAllowedAgent(
|
||||
MappedHandle *mappedHandle, Agent *targetAgent, void *va, size_t size,
|
||||
MappedHandle* _mappedHandle, Agent *targetAgent, void *va, size_t size,
|
||||
hsa_access_permission_t perms)
|
||||
: va(va), size(size), targetAgent(targetAgent), permissions(perms),
|
||||
mappedHandle(mappedHandle) {
|
||||
mappedHandle(_mappedHandle) {
|
||||
|
||||
// CPU agents have access as the memory is already mapped to the host.
|
||||
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) return;
|
||||
@@ -3527,6 +3574,7 @@ Runtime::MappedHandleAllowedAgent::~MappedHandleAllowedAgent() {
|
||||
|
||||
hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permission_t perms) {
|
||||
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) {
|
||||
#if defined(__linux__)
|
||||
void* mapped_ptr =
|
||||
mmap(va, size, PermissionsToMmapFlags(perms), MAP_SHARED | MAP_FIXED, mappedHandle->drm_fd,
|
||||
reinterpret_cast<uint64_t>(mappedHandle->drm_cpu_addr));
|
||||
@@ -3537,6 +3585,9 @@ hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permissi
|
||||
shareable_handle, va, mappedHandle->offset, size, perms);
|
||||
if (status != HSA_STATUS_SUCCESS)
|
||||
return status;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
}
|
||||
permissions = perms;
|
||||
return HSA_STATUS_SUCCESS;
|
||||
@@ -3544,8 +3595,12 @@ hsa_status_t Runtime::MappedHandleAllowedAgent::EnableAccess(hsa_access_permissi
|
||||
|
||||
hsa_status_t Runtime::MappedHandleAllowedAgent::RemoveAccess() {
|
||||
if (targetAgent->device_type() == core::Agent::DeviceType::kAmdCpuDevice) {
|
||||
#if defined(__linux__)
|
||||
if (munmap(va, size) != 0)
|
||||
return HSA_STATUS_ERROR;
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
} else {
|
||||
return targetAgent->driver().Unmap(
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -51,6 +51,9 @@
|
||||
|
||||
#include "core/util/timer.h"
|
||||
#include "core/inc/runtime.h"
|
||||
#if defined(_WIN32)
|
||||
#include "malloc.h"
|
||||
#endif
|
||||
|
||||
namespace rocr {
|
||||
namespace core {
|
||||
@@ -234,8 +237,11 @@ uint32_t Signal::WaitMultiple(uint32_t signal_count, const hsa_signal_t* hsa_sig
|
||||
MAKE_SCOPE_GUARD([&]() {
|
||||
if (signal_count > small_size) delete[] evts;
|
||||
});
|
||||
|
||||
#if defined(__linux__)
|
||||
uint64_t event_age[unique_evts];
|
||||
#else
|
||||
auto event_age = reinterpret_cast<uint64_t*>(_alloca(unique_evts * sizeof(unique_evts)));
|
||||
#endif
|
||||
memset(event_age, 0, unique_evts * sizeof(uint64_t));
|
||||
if (core::Runtime::runtime_singleton_->KfdVersion().supports_event_age)
|
||||
for (uint32_t i = 0; i < unique_evts; i++)
|
||||
@@ -367,8 +373,11 @@ uint32_t Signal::WaitAnyExceptions(uint32_t signal_count, const hsa_signal_t* hs
|
||||
std::sort(evts, evts + signal_count);
|
||||
HsaEvent** end = std::unique(evts, evts + signal_count);
|
||||
unique_evts = uint32_t(end - evts);
|
||||
|
||||
#if defined(__linux__)
|
||||
uint64_t event_age[unique_evts];
|
||||
#else
|
||||
auto event_age = reinterpret_cast<uint64_t*>(_alloca(unique_evts * sizeof(unique_evts)));
|
||||
#endif
|
||||
memset(event_age, 0, unique_evts * sizeof(uint64_t));
|
||||
if (core::Runtime::runtime_singleton_->KfdVersion().supports_event_age)
|
||||
for (uint32_t i = 0; i < unique_evts; i++)
|
||||
|
||||
@@ -44,8 +44,17 @@
|
||||
|
||||
#include <stdint.h>
|
||||
#include <algorithm>
|
||||
#if defined(__linux__)
|
||||
#include <sys/eventfd.h>
|
||||
#include <poll.h>
|
||||
#else
|
||||
struct pollfd {
|
||||
int fd;
|
||||
short int events;
|
||||
short int revents;
|
||||
};
|
||||
#define POLLIN 0x001 // from poll.h...
|
||||
#endif
|
||||
|
||||
#include "core/util/utils.h"
|
||||
#include "core/inc/runtime.h"
|
||||
@@ -171,7 +180,12 @@ void SvmProfileControl::PollSmi() {
|
||||
};
|
||||
|
||||
while (!exit) {
|
||||
#if defined(__linux__)
|
||||
int ready = poll(&files[0], files.size(), -1);
|
||||
#else
|
||||
int ready = 0;
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
if (ready < 1) {
|
||||
assert(false && "poll failed!");
|
||||
return;
|
||||
@@ -345,9 +359,10 @@ void SvmProfileControl::PollSmi() {
|
||||
}
|
||||
|
||||
SvmProfileControl::SvmProfileControl() : event(-1), exit(false) {
|
||||
#if defined(__linux__)
|
||||
event = eventfd(0, EFD_CLOEXEC);
|
||||
if (event == -1) return;
|
||||
|
||||
#endif
|
||||
poll_smi_thread_ = os::CreateThread(PollSmiRun, (void*)this);
|
||||
if (poll_smi_thread_ == NULL) {
|
||||
assert(false && "Poll SMI thread creation error.");
|
||||
@@ -356,10 +371,12 @@ SvmProfileControl::SvmProfileControl() : event(-1), exit(false) {
|
||||
}
|
||||
|
||||
SvmProfileControl::~SvmProfileControl() {
|
||||
#if defined(__linux__)
|
||||
if (event != -1) {
|
||||
eventfd_write(event, 1);
|
||||
close(event);
|
||||
}
|
||||
#endif
|
||||
if (poll_smi_thread_ != NULL) {
|
||||
exit = true;
|
||||
os::WaitForThread(poll_smi_thread_);
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -43,9 +43,12 @@
|
||||
#include "core/inc/thunk_loader.h"
|
||||
#include "core/inc/runtime.h"
|
||||
|
||||
#include <dlfcn.h>
|
||||
#include <core/util/os.h>
|
||||
#include <iostream>
|
||||
#if defined(__linux__)
|
||||
#include <dlfcn.h>
|
||||
#include <fcntl.h>
|
||||
#endif
|
||||
|
||||
namespace rocr {
|
||||
namespace core {
|
||||
@@ -57,6 +60,7 @@ namespace core {
|
||||
return "libdtif.so";
|
||||
}
|
||||
|
||||
#if defined(__linux__)
|
||||
if (core::Runtime::runtime_singleton_->flag().enable_dxg_detection()) {
|
||||
int fd = open("/dev/dxg", O_RDWR);
|
||||
if (fd >= 0) {
|
||||
@@ -65,6 +69,9 @@ namespace core {
|
||||
return "librocdxg.so";
|
||||
}
|
||||
}
|
||||
#else
|
||||
is_dxg_ = true;
|
||||
#endif
|
||||
|
||||
return "";
|
||||
}
|
||||
@@ -74,10 +81,10 @@ namespace core {
|
||||
library_name(whoami()),
|
||||
is_loaded_(false) {
|
||||
if (!library_name.empty()) {
|
||||
dlerror(); // Clear any existing error messages
|
||||
thunk_handle = dlopen(library_name.c_str(), RTLD_LAZY);
|
||||
rocr::os::DlError(); // Clear any existing error messages
|
||||
thunk_handle = rocr::os::LoadLib(library_name.c_str());
|
||||
if (thunk_handle == NULL) {
|
||||
fprintf(stderr, "Cannot load %s, failed:%s\n", library_name.c_str(), dlerror());
|
||||
fprintf(stderr, "Cannot load %s, failed:%s\n", library_name.c_str(), rocr::os::DlError());
|
||||
} else {
|
||||
debug_print("Load %s successully!\n", library_name.c_str());
|
||||
}
|
||||
@@ -88,8 +95,8 @@ namespace core {
|
||||
ThunkLoader::~ThunkLoader() {
|
||||
if (IsSharedLibraryLoaded()
|
||||
&& (thunk_handle != NULL)) {
|
||||
if (dlclose(thunk_handle) != 0) {
|
||||
fprintf(stderr, "Cannot unload %s, failed:%s\n", library_name.c_str(), dlerror());
|
||||
if (!rocr::os::CloseLib(thunk_handle)) {
|
||||
fprintf(stderr, "Cannot unload %s, failed:%s\n", library_name.c_str(), rocr::os::DlError());
|
||||
} else {
|
||||
debug_print("Unload %s successully!\n", library_name.c_str());
|
||||
}
|
||||
@@ -98,6 +105,7 @@ namespace core {
|
||||
|
||||
void ThunkLoader::LoadThunkApiTable() {
|
||||
if (IsSharedLibraryLoaded()) {
|
||||
#if defined(__linux__)
|
||||
dlerror(); // Clear any existing error messages
|
||||
|
||||
HSAKMT_PFN(hsaKmtOpenKFD) = (HSAKMT_DEF(hsaKmtOpenKFD)*)dlsym(thunk_handle, "hsaKmtOpenKFD");
|
||||
@@ -402,12 +410,12 @@ namespace core {
|
||||
|
||||
DRM_PFN(drmCommandWriteRead) = (DRM_DEF(drmCommandWriteRead)*)dlsym(thunk_handle, "drmCommandWriteRead");
|
||||
if (DRM_PFN(drmCommandWriteRead) == NULL) goto ERROR;
|
||||
|
||||
debug_print("Load all DTIF APIs OK!\n");
|
||||
return;
|
||||
|
||||
ERROR:
|
||||
fprintf(stderr, "dlsym failed: %s\n", dlerror());
|
||||
#endif
|
||||
} else {
|
||||
HSAKMT_PFN(hsaKmtOpenKFD) = (HSAKMT_DEF(hsaKmtOpenKFD)*)(&hsaKmtOpenKFD);
|
||||
HSAKMT_PFN(hsaKmtCloseKFD) = (HSAKMT_DEF(hsaKmtCloseKFD)*)(&hsaKmtCloseKFD);
|
||||
@@ -499,6 +507,9 @@ ERROR:
|
||||
HSAKMT_PFN(hsaKmtPcSamplingStart) = (HSAKMT_DEF(hsaKmtPcSamplingStart)*)(&hsaKmtPcSamplingStart);
|
||||
HSAKMT_PFN(hsaKmtPcSamplingStop) = (HSAKMT_DEF(hsaKmtPcSamplingStop)*)(&hsaKmtPcSamplingStop);
|
||||
HSAKMT_PFN(hsaKmtPcSamplingSupport) = (HSAKMT_DEF(hsaKmtPcSamplingSupport)*)(&hsaKmtPcSamplingSupport);
|
||||
#if defined(_WIN32)
|
||||
HSAKMT_PFN(hsaKmtQueueRingDoorbell) = (HSAKMT_DEF(hsaKmtQueueRingDoorbell)*)(&hsaKmtQueueRingDoorbell);
|
||||
#endif
|
||||
HSAKMT_PFN(hsaKmtModelEnabled) = (HSAKMT_DEF(hsaKmtModelEnabled)*)(&hsaKmtModelEnabled);
|
||||
|
||||
DRM_PFN(amdgpu_device_initialize) = (DRM_DEF(amdgpu_device_initialize)*)(&amdgpu_device_initialize);
|
||||
@@ -509,7 +520,9 @@ ERROR:
|
||||
DRM_PFN(amdgpu_bo_export) = (DRM_DEF(amdgpu_bo_export)*)(&amdgpu_bo_export);
|
||||
DRM_PFN(amdgpu_bo_import) = (DRM_DEF(amdgpu_bo_import)*)(&amdgpu_bo_import);
|
||||
DRM_PFN(amdgpu_bo_va_op) = (DRM_DEF(amdgpu_bo_va_op)*)(&amdgpu_bo_va_op);
|
||||
#if defined(__linux__)
|
||||
DRM_PFN(drmCommandWriteRead) = (DRM_DEF(drmCommandWriteRead)*)(&drmCommandWriteRead);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
|
||||
@@ -517,7 +530,8 @@ ERROR:
|
||||
if (!IsDTIF())
|
||||
return true;
|
||||
|
||||
DtifCreateFunc* pfnDtifCreate = (DtifCreateFunc*)dlsym(thunk_handle, "DtifCreate");
|
||||
DtifCreateFunc* pfnDtifCreate =
|
||||
(DtifCreateFunc*)rocr::os::GetExportAddress(thunk_handle, "DtifCreate");
|
||||
if (pfnDtifCreate != NULL) {
|
||||
if (pfnDtifCreate("HSA") != NULL) {
|
||||
debug_print("DtifCreate OK!\n");
|
||||
@@ -537,7 +551,8 @@ ERROR:
|
||||
if (thunk_handle == NULL)
|
||||
return false;
|
||||
|
||||
DtifDestroyFunc* pfnDtifDestroy = (DtifDestroyFunc*)dlsym(thunk_handle, "DtifDestroy");
|
||||
DtifDestroyFunc* pfnDtifDestroy =
|
||||
(DtifDestroyFunc*)rocr::os::GetExportAddress(thunk_handle, "DtifDestroy");
|
||||
if (pfnDtifDestroy != NULL) {
|
||||
pfnDtifDestroy();
|
||||
debug_print("DtifDestroy OK!\n");
|
||||
|
||||
+12
-1
@@ -3,7 +3,7 @@
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2022, Advanced Micro Devices, Inc. All rights reserved.
|
||||
## Copyright (c) 2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
@@ -139,10 +139,21 @@ function(generate_bytecodeStrm HeaderFILE)
|
||||
|
||||
## Add a custom command that generates amd_trap_handler_v2.h
|
||||
## This depends on all the generated code object files and the C++ generator script.
|
||||
|
||||
if (UNIX)
|
||||
add_custom_command(OUTPUT ${HeaderFILE}.h
|
||||
COMMAND ${CMAKE_CURRENT_SOURCE_DIR}/create_trap_handler_header.sh ${ARG_LIST}
|
||||
COMMENT "Collating trap handlers..."
|
||||
DEPENDS ${HSACO_TARG_LIST} create_trap_handler_header.sh )
|
||||
else()
|
||||
find_package(Python3 COMPONENTS Interpreter REQUIRED)
|
||||
add_custom_command(
|
||||
OUTPUT ${HeaderFILE}.h
|
||||
COMMAND ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/create_trap_handler_header.py ${ARG_LIST}
|
||||
COMMENT "Collating blit shaders..."
|
||||
DEPENDS ${HSACO_TARG_LIST} create_trap_handler_header.py)
|
||||
endif()
|
||||
|
||||
|
||||
## Export a target that builds (and depends on) amd_trap_handler_v2.h
|
||||
add_custom_target( ${HeaderFILE} DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/${HeaderFILE}.h )
|
||||
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
################################################################################
|
||||
##
|
||||
## Copyright (c) Advanced Micro Devices, Inc., or its affiliates.
|
||||
##
|
||||
## SPDX-License-Identifier: MIT
|
||||
##
|
||||
################################################################################
|
||||
import sys
|
||||
|
||||
def GetSize(fileobject):
|
||||
fileobject.seek(0,2) # move the cursor to the end of the file
|
||||
size = fileobject.tell()
|
||||
return size
|
||||
|
||||
def DumpFile(header, input_name):
|
||||
try:
|
||||
with open(input_name, "rb") as binary_file:
|
||||
# Read the entire content of the file as bytes
|
||||
binary_data = binary_file.read()
|
||||
file_size = GetSize(binary_file)
|
||||
#print(f"Binary size: {file_size}")
|
||||
# Reset file pointer
|
||||
binary_file.seek(0)
|
||||
parts = input_name.split('.')
|
||||
file_name = parts[0]
|
||||
content = f"unsigned char {file_name}""[] = {\n "
|
||||
|
||||
header.write(content)
|
||||
line = 0
|
||||
count = 0
|
||||
for byte_value in binary_data:
|
||||
count += 1
|
||||
padded_hex = '{:02x}'.format(byte_value)
|
||||
if (count != file_size):
|
||||
header.write(f"0x{padded_hex},")
|
||||
else:
|
||||
header.write(f"0x{padded_hex}")
|
||||
line += 1
|
||||
if (line == 12):
|
||||
header.write(f"\n ")
|
||||
line = 0
|
||||
else:
|
||||
header.write(f" ")
|
||||
|
||||
header.write("\n};\nunsigned int "f"{file_name}_len = {file_size};\n")
|
||||
|
||||
except FileNotFoundError:
|
||||
print(f"Error: The file {input_name} was not found.")
|
||||
except Exception as e:
|
||||
print(f"An error occurred: {e}")
|
||||
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
header_name = sys.argv[1];
|
||||
with open(header_name, 'w') as header:
|
||||
header.write("//==============================================================================\n")
|
||||
header.write("// This file is automatically generated during build process, don't modify it\n")
|
||||
header.write("//==============================================================================\n\n")
|
||||
header.write("namespace rocr {\n")
|
||||
header.write("namespace AMD {\n\n")
|
||||
|
||||
for i, arg in enumerate(sys.argv):
|
||||
if (i > 1):
|
||||
#print(f"File {i}: {arg}\n")
|
||||
DumpFile(header, arg)
|
||||
header.write("} // namespace AMD\n")
|
||||
header.write("} // namespace rocr\n\n")
|
||||
|
||||
else:
|
||||
print("Empty arguments!")
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -48,6 +48,133 @@
|
||||
#ifndef HSA_RUNTIME_CORE_UTIL_ATOMIC_HELPERS_H_
|
||||
#define HSA_RUNTIME_CORE_UTIL_ATOMIC_HELPERS_H_
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define WIN32_NO_STATUS
|
||||
#include <Windows.h>
|
||||
#undef WIN32_NO_STATUS
|
||||
|
||||
template <class T>
|
||||
void __atomic_load(const T* object, typename std::remove_volatile<T>::type* ret, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
*ret = InterlockedOr64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
0);
|
||||
} else {
|
||||
*ret = InterlockedOr(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
0);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
void __atomic_store(const T* object, typename std::remove_volatile<T>::type* val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
InterlockedExchange64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
*val);
|
||||
} else {
|
||||
InterlockedExchange(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
*val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_or(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedOr64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
} else {
|
||||
return InterlockedOr(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_and(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedAnd64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
} else {
|
||||
return InterlockedAnd(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_xor(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedXor64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
} else {
|
||||
return InterlockedXor(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_add(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedExchangeAdd64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
} else {
|
||||
return InterlockedExchangeAdd(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
typename std::remove_volatile<T>::type __atomic_fetch_sub(
|
||||
const T* object, typename std::remove_volatile<T>::type val, int arg) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
return InterlockedExchangeAdd64(
|
||||
reinterpret_cast<volatile LONG64*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val * (-1));
|
||||
} else {
|
||||
return InterlockedExchangeAdd(
|
||||
reinterpret_cast<volatile LONG*>(const_cast<typename std::remove_const<T>::type*>(object)),
|
||||
val * (-1));
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
void __atomic_compare_exchange(
|
||||
T* object, typename std::remove_volatile<T>::type* expected,
|
||||
typename std::remove_volatile<T>::type* val, int arg0, int arg1, int arg2) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
InterlockedCompareExchange64(reinterpret_cast<volatile LONG64*>(object),
|
||||
*val, *expected);
|
||||
} else {
|
||||
InterlockedCompareExchange(reinterpret_cast<volatile LONG*>(object),
|
||||
*val, *expected);
|
||||
}
|
||||
}
|
||||
|
||||
template <class T>
|
||||
void __atomic_exchange(T* object, typename std::remove_volatile<T>::type* val,
|
||||
typename std::remove_volatile<T>::type* ret, int arg0) {
|
||||
if constexpr (sizeof(T) == 8) {
|
||||
*ret = InterlockedExchange64(reinterpret_cast<volatile LONG64*>(object), *val);
|
||||
} else {
|
||||
*ret = InterlockedExchange(reinterpret_cast<volatile LONG*>(object), *val);
|
||||
}
|
||||
}
|
||||
|
||||
#define __ATOMIC_RELAXED 0
|
||||
#endif
|
||||
|
||||
#include <atomic>
|
||||
|
||||
//ALWAYS_CONSERVATIVE will very likely overfence your code.
|
||||
@@ -145,14 +272,18 @@ static __forceinline void Fence(std::memory_order order=std::memory_order_seq_cs
|
||||
|
||||
template <class T>
|
||||
static __forceinline void BasicCheck(const T* ptr) {
|
||||
#if defined(__linux__)
|
||||
constexpr bool value = __atomic_always_lock_free(sizeof(T), 0);
|
||||
static_assert(value, "Atomic type may not be compatible with peripheral atomics.");
|
||||
#endif
|
||||
};
|
||||
|
||||
template <class T>
|
||||
static __forceinline void BasicCheck(const volatile T* ptr) {
|
||||
#if defined(__linux__)
|
||||
constexpr bool value = __atomic_always_lock_free(sizeof(T), 0);
|
||||
static_assert(value, "Atomic type may not be compatible with peripheral atomics.");
|
||||
#endif
|
||||
};
|
||||
|
||||
/// @brief: Load value of type T atomically with specified memory order.
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -61,6 +61,7 @@
|
||||
#include <utility>
|
||||
#include <semaphore.h>
|
||||
#include "core/inc/runtime.h"
|
||||
#include <sys/mman.h>
|
||||
#if defined(__i386__) || defined(__x86_64__)
|
||||
#include <cpuid.h>
|
||||
#endif
|
||||
@@ -294,7 +295,7 @@ void* GetExportAddress(LibHandle lib, std::string export_name) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void CloseLib(LibHandle lib) { dlclose(*(void**)&lib); }
|
||||
bool CloseLib(LibHandle lib) { return (dlclose(*(void**)&lib) == 0) ? true : false; }
|
||||
|
||||
/*
|
||||
* @brief Look for a symbol called "HSA_AMD_TOOL_PRIORITY" across all loaded
|
||||
@@ -579,7 +580,7 @@ int WaitForOsEvent(EventHandle event, unsigned int milli_seconds) {
|
||||
}
|
||||
|
||||
int ret_code = 0;
|
||||
|
||||
|
||||
if (!eventDescrp->state) {
|
||||
if (milli_seconds == 0) {
|
||||
ret_code = 1;
|
||||
@@ -816,6 +817,124 @@ bool ParseCpuID(cpuid_t* cpuinfo) {
|
||||
#endif
|
||||
}
|
||||
|
||||
uint64_t TimeNanos() {
|
||||
struct timespec tp;
|
||||
::clock_gettime(CLOCK_MONOTONIC, &tp);
|
||||
return (uint64_t)tp.tv_sec * (1000ULL * 1000ULL * 1000ULL) + (uint64_t)tp.tv_nsec;
|
||||
}
|
||||
|
||||
static inline int MemProtToOsProt(MemProt prot) {
|
||||
switch (prot) {
|
||||
case MEM_PROT_NONE:
|
||||
return PROT_NONE;
|
||||
case MEM_PROT_READ:
|
||||
return PROT_READ;
|
||||
case MEM_PROT_RW:
|
||||
return PROT_READ | PROT_WRITE;
|
||||
case MEM_PROT_RWX:
|
||||
return PROT_READ | PROT_WRITE | PROT_EXEC;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
size_t PageSize() {
|
||||
static size_t g_page_size_ = 0; //!< The default os page size
|
||||
if (g_page_size_ == 0) {
|
||||
g_page_size_ = (size_t)::sysconf(_SC_PAGESIZE);
|
||||
}
|
||||
return g_page_size_;
|
||||
}
|
||||
|
||||
void* ReserveMemory(void* start, size_t size, size_t alignment, MemProt prot) {
|
||||
size = AlignUp(size, PageSize());
|
||||
// check for invalid input size
|
||||
if (size == 0) {
|
||||
return NULL;
|
||||
}
|
||||
alignment = std::max(PageSize(), AlignUp(alignment, PageSize()));
|
||||
assert(IsPowerOfTwo(alignment) && "not a power of 2");
|
||||
|
||||
size_t requested = size + alignment - PageSize();
|
||||
address mem = (address)::mmap(start, requested, MemProtToOsProt(prot),
|
||||
MAP_PRIVATE | MAP_NORESERVE | MAP_ANONYMOUS, 0, 0);
|
||||
|
||||
// check for out of memory
|
||||
if (mem == MAP_FAILED) return NULL;
|
||||
|
||||
address aligned = AlignUp(mem, alignment);
|
||||
|
||||
// return the unused leading pages to the free state
|
||||
if (&aligned[0] != &mem[0]) {
|
||||
assert(&aligned[0] > &mem[0] && "check this code");
|
||||
if (::munmap(&mem[0], &aligned[0] - &mem[0]) != 0) {
|
||||
assert(!"::munmap failed");
|
||||
}
|
||||
}
|
||||
// return the unused trailing pages to the free state
|
||||
if (&aligned[size] != &mem[requested]) {
|
||||
assert(&aligned[size] < &mem[requested] && "check this code");
|
||||
if (::munmap(&aligned[size], &mem[requested] - &aligned[size]) != 0) {
|
||||
assert(!"::munmap failed");
|
||||
}
|
||||
}
|
||||
|
||||
// Hint to enable THP for large host allocations which can help in performance gain
|
||||
constexpr size_t kLargePageSize = 2 * 1024 * 1024;
|
||||
if (size >= kLargePageSize) {
|
||||
int status = madvise(aligned, size, MADV_HUGEPAGE);
|
||||
if (status) {
|
||||
LogPrint(HSA_AMD_LOG_FLAG_INFO,
|
||||
"madvise with advice MADV_HUGEPAGE"
|
||||
" starting at address %p and page size 0x%zx, returned %d, errno: %s",
|
||||
aligned, size, status, strerror(errno));
|
||||
}
|
||||
}
|
||||
|
||||
return aligned;
|
||||
}
|
||||
|
||||
bool ReleaseMemory(void* addr, size_t size) {
|
||||
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
|
||||
size = AlignUp(size, PageSize());
|
||||
|
||||
return 0 == ::munmap(addr, size);
|
||||
}
|
||||
|
||||
bool CommitMemory(void* addr, size_t size, MemProt prot) {
|
||||
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
|
||||
size = AlignUp(size, PageSize());
|
||||
|
||||
return ::mmap(addr, size, MemProtToOsProt(prot), MAP_PRIVATE | MAP_FIXED | MAP_ANONYMOUS, -1,
|
||||
0) != MAP_FAILED;
|
||||
}
|
||||
|
||||
bool UncommitMemory(void* addr, size_t size) {
|
||||
assert(IsMultipleOf(addr, PageSize()) && "not page aligned!");
|
||||
size = AlignUp(size, PageSize());
|
||||
|
||||
return ::mmap(addr, size, PROT_NONE, MAP_PRIVATE | MAP_FIXED | MAP_NORESERVE | MAP_ANONYMOUS, -1,
|
||||
0) != MAP_FAILED;
|
||||
}
|
||||
|
||||
uint64_t HostTotalPhysicalMemory() {
|
||||
static uint64_t totalPhys = 0;
|
||||
|
||||
if (totalPhys != 0) {
|
||||
return totalPhys;
|
||||
}
|
||||
|
||||
totalPhys = sysconf(_SC_PAGESIZE) * sysconf(_SC_PHYS_PAGES);
|
||||
return totalPhys;
|
||||
}
|
||||
|
||||
int Ffs(int i) { return ffs(i); }
|
||||
|
||||
int Ctz(uint64_t i) { return __builtin_ctz(i); }
|
||||
|
||||
char* DlError() { return dlerror(); }
|
||||
|
||||
} // namespace os
|
||||
} // namespace rocr
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -91,7 +91,7 @@ void* GetExportAddress(LibHandle lib, std::string export_name);
|
||||
|
||||
/// @brief: Unloads the dynamic library.
|
||||
/// @param: lib(Input), library handle which will be unloaded.
|
||||
void CloseLib(LibHandle lib);
|
||||
bool CloseLib(LibHandle lib);
|
||||
|
||||
/// @brief: Lists loaded tool libraries that contain
|
||||
/// symbol HSA_AMD_TOOL_PRIORITY
|
||||
@@ -106,6 +106,7 @@ std::string GetLibraryName(LibHandle lib);
|
||||
/// @brief: Creates a Semaphore, will return NULL if failed.
|
||||
/// @param: void.
|
||||
/// @return: Semaphore.
|
||||
#undef CreateSemaphore
|
||||
Semaphore CreateSemaphore();
|
||||
|
||||
/// @brief: Waits for the semaphore. This is a blocking wait.
|
||||
@@ -127,6 +128,7 @@ void DestroySemaphore(Semaphore sem);
|
||||
/// @brief: Creates a mutex, will return NULL if failed.
|
||||
/// @param: void.
|
||||
/// @return: Mutex.
|
||||
#undef CreateMutex
|
||||
Mutex CreateMutex();
|
||||
|
||||
/// @brief: Tries to acquire the mutex once, if successed, return true.
|
||||
@@ -319,15 +321,48 @@ uint64_t ReadSystemClock();
|
||||
/// @brief read the system clock frequency
|
||||
uint64_t SystemClockFrequency();
|
||||
|
||||
typedef struct cpuid_s {
|
||||
struct cpuid_t {
|
||||
char ManufacturerID[13]; // 12 char, NULL terminated
|
||||
bool mwaitx;
|
||||
} cpuid_t;
|
||||
};
|
||||
|
||||
/// @brief parse CPUID
|
||||
/// @param: cpuinfo struct to be filled
|
||||
bool ParseCpuID(cpuid_t* cpuinfo);
|
||||
|
||||
//! Return the default os page size.
|
||||
size_t PageSize();
|
||||
|
||||
/// @brief CPU time in nanoseconds
|
||||
/// @param: None
|
||||
uint64_t TimeNanos();
|
||||
|
||||
using address = char*;
|
||||
enum MemProt { MEM_PROT_NONE = 0, MEM_PROT_READ, MEM_PROT_RW, MEM_PROT_RWX };
|
||||
|
||||
/// @brief Reserves a chunk of memory (priv | anon | noreserve)
|
||||
/// @param:
|
||||
void* ReserveMemory(void* start, size_t size, size_t alignment = 0,
|
||||
MemProt prot = MEM_PROT_NONE);
|
||||
|
||||
/// Release a chunk of memory reserved with reserveMemory.
|
||||
bool ReleaseMemory(void* addr, size_t size);
|
||||
/// Commit a chunk of memory previously reserved with reserveMemory.
|
||||
bool CommitMemory(void* addr, size_t size, MemProt prot = MEM_PROT_NONE);
|
||||
/// Uncommit a chunk of memory previously committed with commitMemory.
|
||||
bool UncommitMemory(void* addr, size_t size);
|
||||
|
||||
uint64_t HostTotalPhysicalMemory();
|
||||
|
||||
/// Find First Set for any OS
|
||||
int Ffs(int i);
|
||||
|
||||
/// Find the count of leading zeros
|
||||
int Ctz(uint64_t i);
|
||||
|
||||
/// Shared library or DLL load error
|
||||
char* DlError();
|
||||
|
||||
} // namespace os
|
||||
} // namespace rocr
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -196,6 +196,35 @@ template <typename Allocator> class SimpleHeap {
|
||||
return reinterpret_cast<void*>(base);
|
||||
}
|
||||
|
||||
/* Return block-base the ptr belongs to if the ptr is a valid ptr which is allocated
|
||||
* from this simpleheap and the block-base is allocated from block_allocator_*/
|
||||
void* block_base(void* ptr) {
|
||||
if (ptr == nullptr)
|
||||
return nullptr;
|
||||
|
||||
uintptr_t base = reinterpret_cast<uintptr_t>(ptr);
|
||||
|
||||
// Find fragment and validate.
|
||||
auto frag_map_it = block_list_.upper_bound(base);
|
||||
if (frag_map_it == block_list_.begin())
|
||||
return nullptr;
|
||||
frag_map_it--;
|
||||
auto& frag_map = frag_map_it->second;
|
||||
auto fragment = frag_map.find(base);
|
||||
if (fragment == frag_map.end() || isFree(fragment->second))
|
||||
return nullptr;
|
||||
|
||||
return reinterpret_cast<void*>(frag_map_it->first);
|
||||
}
|
||||
|
||||
void reset() {
|
||||
free_list_.clear();
|
||||
block_list_.clear();
|
||||
block_cache_.clear();
|
||||
in_use_size_ = 0;
|
||||
cache_size_ = 0;
|
||||
}
|
||||
|
||||
bool free(void* ptr) {
|
||||
if (ptr == nullptr) return true;
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -49,13 +49,16 @@
|
||||
#include "stddef.h"
|
||||
#include "stdlib.h"
|
||||
#include "stdarg.h"
|
||||
#if defined(__linux__)
|
||||
#include "unistd.h"
|
||||
#endif
|
||||
#include <assert.h>
|
||||
#include <iostream>
|
||||
#include <string>
|
||||
#include <algorithm>
|
||||
#include <sstream>
|
||||
#include <thread>
|
||||
#include <locale>
|
||||
|
||||
namespace rocr {
|
||||
extern FILE* log_file;
|
||||
@@ -64,6 +67,14 @@ extern uint8_t log_flags[8];
|
||||
typedef unsigned int uint;
|
||||
typedef uint64_t uint64;
|
||||
|
||||
// 2MB huge page size
|
||||
#define GPU_HUGE_PAGE_SIZE (2 << 20)
|
||||
|
||||
// 4KB page size
|
||||
#define DEFAULT_GPU_PAGE_SIZE (1 << 12)
|
||||
|
||||
void log_printf(const char* file, int line, const char* format, ...);
|
||||
|
||||
#if defined(__GNUC__)
|
||||
#if defined(__i386__) || defined(__x86_64__)
|
||||
#include <x86intrin.h>
|
||||
@@ -75,8 +86,6 @@ typedef uint64_t uint64;
|
||||
#define __stdcall // __attribute__((__stdcall__))
|
||||
#define __ALIGNED__(x) __attribute__((aligned(x)))
|
||||
|
||||
void log_printf(const char* file, int line, const char* format, ...);
|
||||
|
||||
static __forceinline void* _aligned_malloc(size_t size, size_t alignment) {
|
||||
#ifdef _ISOC11_SOURCE
|
||||
return aligned_alloc(alignment, size);
|
||||
@@ -114,6 +123,7 @@ static __forceinline unsigned long long int strtoull(const char* str,
|
||||
do { \
|
||||
} while (false)
|
||||
#else
|
||||
#if defined(__linux__)
|
||||
#define debug_warning_n(exp, limit) \
|
||||
do { \
|
||||
static std::atomic<int> count(0); \
|
||||
@@ -123,6 +133,18 @@ static __forceinline unsigned long long int strtoull(const char* str,
|
||||
count++; \
|
||||
} \
|
||||
} while (false)
|
||||
#else
|
||||
#define debug_warning_n(exp, limit) \
|
||||
do { \
|
||||
static std::atomic<int> count(0); \
|
||||
if (!(exp) && (limit == 0 || count < limit)) { \
|
||||
fprintf(stderr, "Warning: " STRING(exp) " in %s, " __FILE__ ":" STRING(__LINE__) "\n" \
|
||||
); \
|
||||
count++; \
|
||||
} \
|
||||
} while (false)
|
||||
|
||||
#endif
|
||||
#endif
|
||||
#define debug_warning(exp) debug_warning_n((exp), 0)
|
||||
|
||||
@@ -369,10 +391,15 @@ inline void FlushCpuCache(const void* base, size_t offset, size_t len) {
|
||||
static long cacheline_size = 0;
|
||||
|
||||
if (!cacheline_size) {
|
||||
#ifdef _SC_LEVEL1_DCACHE_LINESIZE
|
||||
long sz = sysconf(_SC_LEVEL1_DCACHE_LINESIZE);
|
||||
long sz = 64;
|
||||
#if defined(__linux__)
|
||||
#ifdef _SC_LEVEL1_DCACHE_LINESIZE
|
||||
sz = sysconf(_SC_LEVEL1_DCACHE_LINESIZE);
|
||||
#else
|
||||
sz = 0;
|
||||
#endif
|
||||
#else
|
||||
long sz = 0;
|
||||
//@todo abstract GetLogicalProcessorInformation call
|
||||
#endif
|
||||
if (sz <= 0) return;
|
||||
cacheline_size = sz;
|
||||
@@ -421,6 +448,17 @@ inline uint32_t PtrHigh64Shift40(const void* p) {
|
||||
return (uint32_t)((ptr & 0xFFFFFF0000000000ULL) >> 40);
|
||||
}
|
||||
|
||||
static inline uint8_t Ptr48High8(const void* p) {
|
||||
uintptr_t ptr = reinterpret_cast<uintptr_t>(p);
|
||||
return (uint8_t)((ptr & 0xFF0000000000ULL) >> 40);
|
||||
}
|
||||
|
||||
static inline uint32_t Ptr48Low32(const void* p) {
|
||||
uintptr_t ptr = reinterpret_cast<uintptr_t>(p);
|
||||
assert((ptr & 0xFFFFFFFFFF00ULL) == ptr);
|
||||
return (uint32_t)((ptr & 0xFFFFFFFFFFULL) >> 8);
|
||||
}
|
||||
|
||||
inline uint32_t PtrLow32(const void* p) {
|
||||
return static_cast<uint32_t>(reinterpret_cast<uintptr_t>(p));
|
||||
}
|
||||
@@ -433,6 +471,10 @@ inline uint32_t PtrHigh32(const void* p) {
|
||||
return ptr;
|
||||
}
|
||||
|
||||
inline uint32_t HighPart(uint64_t value) { return (value & 0xFFFFFFFF00000000) >> 32; }
|
||||
|
||||
inline uint32_t LowPart(uint64_t value) { return (value & 0x00000000FFFFFFFF); }
|
||||
|
||||
/// @brief: Concatenates two numbers of type InType to a number of type OutType
|
||||
/// @param: hi(Input), To be placed in the upper bits of the output
|
||||
/// @param: lo(Input), To be placed in the lower bits of the output
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -41,7 +41,6 @@
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#ifdef _WIN32 // Are we compiling for windows?
|
||||
#define NOMINMAX
|
||||
|
||||
#include "core/util/os.h"
|
||||
|
||||
@@ -49,10 +48,13 @@
|
||||
#include <process.h>
|
||||
#include <string>
|
||||
#include <windows.h>
|
||||
#include <ntstatus.h>
|
||||
#include <psapi.h>
|
||||
|
||||
#include <emmintrin.h>
|
||||
#include <pmmintrin.h>
|
||||
#include <xmmintrin.h>
|
||||
#include <shared_mutex>
|
||||
|
||||
#undef Yield
|
||||
#undef CreateMutex
|
||||
@@ -82,37 +84,37 @@ void* GetExportAddress(LibHandle lib, std::string export_name) {
|
||||
return GetProcAddress(*(HMODULE*)&lib, export_name.c_str());
|
||||
}
|
||||
|
||||
void CloseLib(LibHandle lib) { FreeLibrary(*(::HMODULE*)&lib); }
|
||||
bool CloseLib(LibHandle lib) { return FreeLibrary(*(::HMODULE*)&lib); }
|
||||
|
||||
std::vector<LibHandle> GetLoadedLibs() {
|
||||
// Use EnumProcessModulesEx
|
||||
static_assert(false, "Not implemented.");
|
||||
assert(!"Not implemented.");
|
||||
return std::vector<LibHandle>{};
|
||||
}
|
||||
|
||||
std::string GetLibraryName(LibHandle lib) {
|
||||
static_assert(false, "Not implemented.");
|
||||
assert(!"Not implemented.");
|
||||
return std::string{};
|
||||
}
|
||||
|
||||
Semaphore CreateSemaphore() {
|
||||
sem = static_cast<void*>(CreateSemaphore(NULL, 0, LONG_MAX, NULL));
|
||||
assert(sem != NULL && "CreateSemaphore failed");
|
||||
|
||||
auto sem = static_cast<void*>(CreateSemaphoreA(nullptr, 0, LONG_MAX, nullptr));
|
||||
assert(sem != nullptr && "CreateSemaphore failed");
|
||||
return *(Semaphore*)&sem;
|
||||
}
|
||||
|
||||
bool WaitSemaphore(Semaphore sem) {
|
||||
return WaitForSingleObject(*(::HANDLE*)&lock, INFINITE) == WAIT_OBJECT_0;
|
||||
return WaitForSingleObject(sem, INFINITE) == WAIT_OBJECT_0;
|
||||
}
|
||||
|
||||
void PostSemaphore(Semaphore sem) {
|
||||
ReleaseSemaphore(static_cast<HANDLE>(*sem), 1, NULL);
|
||||
ReleaseSemaphore(sem, 1, nullptr);
|
||||
}
|
||||
|
||||
void DestroySemaphore(Semaphore sem) {
|
||||
if (!CloseHandle(static_cast<HANDLE>(*sem))) {
|
||||
if (!CloseHandle(sem)) {
|
||||
assert("CloseHandle() failed");
|
||||
}
|
||||
*sem = NULL;
|
||||
}
|
||||
|
||||
Mutex CreateMutex() { return CreateEvent(NULL, false, true, NULL); }
|
||||
@@ -259,48 +261,37 @@ uint64_t AccurateClockFrequency() {
|
||||
}
|
||||
|
||||
SharedMutex CreateSharedMutex() {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return nullptr;
|
||||
return reinterpret_cast<SharedMutex>(new std::shared_mutex());
|
||||
}
|
||||
|
||||
bool TryAcquireSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
return reinterpret_cast<std::shared_mutex*>(lock)->try_lock();
|
||||
}
|
||||
|
||||
bool AcquireSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
reinterpret_cast<std::shared_mutex*>(lock)->lock();
|
||||
return true;
|
||||
}
|
||||
|
||||
void ReleaseSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
reinterpret_cast<std::shared_mutex*>(lock)->unlock();
|
||||
}
|
||||
|
||||
bool TrySharedAcquireSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
return reinterpret_cast<std::shared_mutex*>(lock)->try_lock_shared();
|
||||
}
|
||||
|
||||
bool SharedAcquireSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
reinterpret_cast<std::shared_mutex*>(lock)->lock_shared();
|
||||
return true;
|
||||
}
|
||||
|
||||
void SharedReleaseSharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
reinterpret_cast<std::shared_mutex*>(lock)->unlock_shared();
|
||||
}
|
||||
|
||||
void DestroySharedMutex(SharedMutex lock) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
delete reinterpret_cast<std::shared_mutex*>(lock);
|
||||
}
|
||||
|
||||
uint64_t ReadSystemClock() {
|
||||
@@ -310,17 +301,183 @@ uint64_t ReadSystemClock() {
|
||||
}
|
||||
|
||||
uint64_t SystemClockFrequency() {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return 0;
|
||||
LARGE_INTEGER frequency;
|
||||
QueryPerformanceFrequency(&frequency);
|
||||
return frequency.QuadPart;
|
||||
}
|
||||
|
||||
bool ParseCpuID(cpuid_t* cpuinfo) {
|
||||
assert(false && "Not implemented.");
|
||||
abort();
|
||||
return false;
|
||||
int regs[4] = {};
|
||||
int info{};
|
||||
|
||||
__cpuid(regs, info);
|
||||
memset(cpuinfo->ManufacturerID, 0, sizeof(cpuinfo->ManufacturerID));
|
||||
*reinterpret_cast<int*>(cpuinfo->ManufacturerID) = regs[1];
|
||||
*reinterpret_cast<int*>(cpuinfo->ManufacturerID + 4) = regs[3];
|
||||
*reinterpret_cast<int*>(cpuinfo->ManufacturerID + 8) = regs[2];
|
||||
// @todo fill the rest of CPU info
|
||||
return true;
|
||||
}
|
||||
|
||||
bool IsEnvVarSet(std::string env_var_name) {
|
||||
char* buff = NULL;
|
||||
buff = getenv(env_var_name.c_str());
|
||||
return (buff != NULL);
|
||||
}
|
||||
|
||||
std::vector<LibHandle> GetLoadedToolsLib() {
|
||||
std::vector<LibHandle> ret;
|
||||
std::vector<std::string> names;
|
||||
HMODULE hMods[1024];
|
||||
HANDLE hProcess = GetCurrentProcess();
|
||||
DWORD cbNeeded;
|
||||
unsigned int i;
|
||||
|
||||
if (EnumProcessModules(hProcess, hMods, sizeof(hMods), &cbNeeded)) {
|
||||
for (i = 0; i < (cbNeeded / sizeof(HMODULE)); i++) {
|
||||
TCHAR szModName[MAX_PATH];
|
||||
|
||||
// Get the full path to the module's file.
|
||||
|
||||
if (GetModuleFileNameEx(hProcess, hMods[i], szModName, sizeof(szModName) / sizeof(TCHAR))) {
|
||||
// Print the module name and handle value.
|
||||
names.push_back(szModName);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!names.empty()) {
|
||||
for (auto& name : names) ret.push_back(LoadLib(name));
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
int GetProcessId() { return ::_getpid(); }
|
||||
|
||||
uint64_t TimeNanos() {
|
||||
static double PerformanceFrequency = 0.f;
|
||||
if (PerformanceFrequency == 0) {
|
||||
LARGE_INTEGER frequency;
|
||||
QueryPerformanceFrequency(&frequency);
|
||||
PerformanceFrequency = (double)frequency.QuadPart;
|
||||
}
|
||||
LARGE_INTEGER current;
|
||||
QueryPerformanceCounter(¤t);
|
||||
return (uint64_t)((double)current.QuadPart / PerformanceFrequency * 1e9);
|
||||
}
|
||||
|
||||
static inline int memProtToOsProt(MemProt prot) {
|
||||
switch (prot) {
|
||||
case MEM_PROT_NONE:
|
||||
return PAGE_NOACCESS;
|
||||
case MEM_PROT_READ:
|
||||
return PAGE_READONLY;
|
||||
case MEM_PROT_RW:
|
||||
return PAGE_READWRITE;
|
||||
case MEM_PROT_RWX:
|
||||
return PAGE_EXECUTE_READWRITE;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
static size_t g_page_size_ = 0; //!< The default os page size
|
||||
static int processorCount_; //!< The number of active processors
|
||||
static size_t allocationGranularity_;
|
||||
|
||||
//! Return the default os page size.
|
||||
size_t PageSize() {
|
||||
if (g_page_size_ == 0) {
|
||||
SYSTEM_INFO si{};
|
||||
::GetSystemInfo(&si);
|
||||
g_page_size_ = si.dwPageSize;
|
||||
}
|
||||
return g_page_size_;
|
||||
}
|
||||
|
||||
void* ReserveMemory(void* start, size_t size, size_t alignment, MemProt prot) {
|
||||
size = AlignUp(size, PageSize());
|
||||
if (allocationGranularity_ == 0) {
|
||||
SYSTEM_INFO si;
|
||||
::GetSystemInfo(&si);
|
||||
g_page_size_ = si.dwPageSize;
|
||||
allocationGranularity_ = (size_t)si.dwAllocationGranularity;
|
||||
}
|
||||
alignment = std::max(allocationGranularity_, AlignUp(alignment, allocationGranularity_));
|
||||
assert(IsPowerOfTwo(alignment) && "not a power of 2");
|
||||
|
||||
size_t requested = size + alignment - allocationGranularity_;
|
||||
address mem, aligned;
|
||||
do {
|
||||
mem = reinterpret_cast<address>(VirtualAlloc(start, requested, MEM_RESERVE, memProtToOsProt(prot)));
|
||||
|
||||
// check for out of memory.
|
||||
if (mem == NULL) return NULL;
|
||||
|
||||
aligned = AlignUp(mem, alignment);
|
||||
|
||||
// check for already aligned memory.
|
||||
if (aligned == mem && size == requested) {
|
||||
return mem;
|
||||
}
|
||||
|
||||
// try to reserve the aligned address.
|
||||
if (VirtualFree(mem, 0, MEM_RELEASE) == 0) {
|
||||
assert(!"VirtualFree failed");
|
||||
}
|
||||
|
||||
mem = (address)VirtualAlloc(aligned, size, MEM_RESERVE, memProtToOsProt(prot));
|
||||
assert((mem == NULL || mem == aligned) && "VirtualAlloc failed");
|
||||
|
||||
} while (mem != aligned);
|
||||
|
||||
return mem;
|
||||
}
|
||||
bool ReleaseMemory(void* addr, size_t size) { return VirtualFree(addr, 0, MEM_RELEASE) != 0; }
|
||||
|
||||
bool CommitMemory(void* addr, size_t size, MemProt prot) {
|
||||
return VirtualAlloc(addr, size, MEM_COMMIT, memProtToOsProt(prot)) != NULL;
|
||||
}
|
||||
|
||||
bool UncommitMemory(void* addr, size_t size) { return VirtualFree(addr, size, MEM_DECOMMIT) != 0; }
|
||||
|
||||
uint64_t HostTotalPhysicalMemory() {
|
||||
static uint64_t totalPhys = 0;
|
||||
|
||||
if (totalPhys != 0) {
|
||||
return totalPhys;
|
||||
}
|
||||
|
||||
MEMORYSTATUSEX mstatus;
|
||||
mstatus.dwLength = sizeof(mstatus);
|
||||
|
||||
::GlobalMemoryStatusEx(&mstatus);
|
||||
|
||||
totalPhys = mstatus.ullTotalPhys;
|
||||
return totalPhys;
|
||||
}
|
||||
|
||||
int Ffs(int i) {
|
||||
int res = 0;
|
||||
unsigned long index;
|
||||
if (_BitScanForward(&index, i) != 0) {
|
||||
res = index + 1;
|
||||
}
|
||||
return res;
|
||||
}
|
||||
|
||||
int Ctz(uint64_t i) {
|
||||
unsigned long index;
|
||||
if (_BitScanReverse64(&index, i)) {
|
||||
return sizeof(i) * 8 - 1 - index;
|
||||
} else {
|
||||
return sizeof(i) * 8;
|
||||
}
|
||||
}
|
||||
|
||||
char* DlError() { return nullptr; }
|
||||
} // namespace os
|
||||
} // namespace rocr
|
||||
|
||||
|
||||
@@ -51,8 +51,9 @@ if( NOT _is_hsa_runtime_dynamic )
|
||||
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} "${CMAKE_CURRENT_LIST_DIR}")
|
||||
|
||||
find_dependency(hsakmt 1.0)
|
||||
find_dependency(LibElf)
|
||||
|
||||
if (UNIX)
|
||||
find_dependency(LibElf)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
include( "${CMAKE_CURRENT_LIST_DIR}/@CORE_RUNTIME_NAME@Targets.cmake" )
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
EXPORTS
|
||||
hsa_init
|
||||
hsa_shut_down
|
||||
hsa_system_get_info
|
||||
hsa_extension_get_name
|
||||
hsa_system_extension_supported
|
||||
hsa_system_major_extension_supported
|
||||
hsa_system_get_extension_table
|
||||
hsa_system_get_major_extension_table
|
||||
hsa_iterate_agents
|
||||
hsa_agent_get_info
|
||||
hsa_agent_get_exception_policies
|
||||
hsa_cache_get_info
|
||||
hsa_agent_iterate_caches
|
||||
hsa_agent_extension_supported
|
||||
hsa_agent_major_extension_supported
|
||||
hsa_queue_create
|
||||
hsa_soft_queue_create
|
||||
hsa_queue_destroy
|
||||
hsa_queue_inactivate
|
||||
hsa_queue_load_read_index_scacquire
|
||||
hsa_queue_load_read_index_relaxed
|
||||
hsa_queue_load_write_index_scacquire
|
||||
hsa_queue_load_write_index_relaxed
|
||||
hsa_queue_store_write_index_relaxed
|
||||
hsa_queue_store_write_index_screlease
|
||||
hsa_queue_cas_write_index_scacq_screl
|
||||
hsa_queue_cas_write_index_scacquire
|
||||
hsa_queue_cas_write_index_relaxed
|
||||
hsa_queue_cas_write_index_screlease
|
||||
hsa_queue_add_write_index_scacq_screl
|
||||
hsa_queue_add_write_index_scacquire
|
||||
hsa_queue_add_write_index_relaxed
|
||||
hsa_queue_add_write_index_screlease
|
||||
hsa_queue_store_read_index_relaxed
|
||||
hsa_queue_store_read_index_screlease
|
||||
hsa_agent_iterate_regions
|
||||
hsa_region_get_info
|
||||
hsa_memory_register
|
||||
hsa_memory_deregister
|
||||
hsa_memory_allocate
|
||||
hsa_memory_free
|
||||
hsa_memory_copy
|
||||
hsa_memory_assign_agent
|
||||
hsa_signal_create
|
||||
hsa_signal_destroy
|
||||
hsa_signal_load_relaxed
|
||||
hsa_signal_load_scacquire
|
||||
hsa_signal_store_relaxed
|
||||
hsa_signal_store_screlease
|
||||
hsa_signal_silent_store_relaxed
|
||||
hsa_signal_silent_store_screlease
|
||||
hsa_signal_wait_relaxed
|
||||
hsa_signal_wait_scacquire
|
||||
hsa_signal_group_create
|
||||
hsa_signal_group_destroy
|
||||
hsa_signal_group_wait_any_scacquire
|
||||
hsa_signal_group_wait_any_relaxed
|
||||
hsa_signal_and_relaxed
|
||||
hsa_signal_and_scacquire
|
||||
hsa_signal_and_screlease
|
||||
hsa_signal_and_scacq_screl
|
||||
hsa_signal_or_relaxed
|
||||
hsa_signal_or_scacquire
|
||||
hsa_signal_or_screlease
|
||||
hsa_signal_or_scacq_screl
|
||||
hsa_signal_xor_relaxed
|
||||
hsa_signal_xor_scacquire
|
||||
hsa_signal_xor_screlease
|
||||
hsa_signal_xor_scacq_screl
|
||||
hsa_signal_exchange_relaxed
|
||||
hsa_signal_exchange_scacquire
|
||||
hsa_signal_exchange_screlease
|
||||
hsa_signal_exchange_scacq_screl
|
||||
hsa_signal_add_relaxed
|
||||
hsa_signal_add_scacquire
|
||||
hsa_signal_add_screlease
|
||||
hsa_signal_add_scacq_screl
|
||||
hsa_signal_subtract_relaxed
|
||||
hsa_signal_subtract_scacquire
|
||||
hsa_signal_subtract_screlease
|
||||
hsa_signal_subtract_scacq_screl
|
||||
hsa_signal_cas_relaxed
|
||||
hsa_signal_cas_scacquire
|
||||
hsa_signal_cas_screlease
|
||||
hsa_signal_cas_scacq_screl
|
||||
hsa_isa_from_name
|
||||
hsa_agent_iterate_isas
|
||||
hsa_isa_get_info
|
||||
hsa_isa_get_info_alt
|
||||
hsa_isa_get_exception_policies
|
||||
hsa_isa_get_round_method
|
||||
hsa_wavefront_get_info
|
||||
hsa_isa_iterate_wavefronts
|
||||
hsa_isa_compatible
|
||||
hsa_code_object_serialize
|
||||
hsa_code_object_deserialize
|
||||
hsa_code_object_destroy
|
||||
hsa_code_object_get_info
|
||||
hsa_code_object_get_symbol
|
||||
hsa_code_object_get_symbol_from_name
|
||||
hsa_code_symbol_get_info
|
||||
hsa_code_object_iterate_symbols
|
||||
hsa_code_object_reader_create_from_file
|
||||
hsa_code_object_reader_create_from_memory
|
||||
hsa_code_object_reader_destroy
|
||||
hsa_executable_create
|
||||
hsa_executable_create_alt
|
||||
hsa_executable_destroy
|
||||
hsa_executable_load_code_object
|
||||
hsa_executable_load_program_code_object
|
||||
hsa_executable_load_agent_code_object
|
||||
hsa_executable_freeze
|
||||
hsa_executable_get_info
|
||||
hsa_executable_global_variable_define
|
||||
hsa_executable_agent_global_variable_define
|
||||
hsa_executable_readonly_variable_define
|
||||
hsa_executable_validate
|
||||
hsa_executable_validate_alt
|
||||
hsa_executable_get_symbol
|
||||
hsa_executable_get_symbol_by_name
|
||||
hsa_executable_symbol_get_info
|
||||
hsa_executable_iterate_symbols
|
||||
hsa_executable_iterate_agent_symbols
|
||||
hsa_executable_iterate_program_symbols
|
||||
hsa_status_string
|
||||
hsa_ext_program_create
|
||||
hsa_ext_program_destroy
|
||||
hsa_ext_program_add_module
|
||||
hsa_ext_program_iterate_modules
|
||||
hsa_ext_program_get_info
|
||||
hsa_ext_program_finalize
|
||||
hsa_amd_coherency_get_type
|
||||
hsa_amd_coherency_set_type
|
||||
hsa_amd_profiling_set_profiler_enabled
|
||||
hsa_amd_profiling_get_dispatch_time
|
||||
hsa_amd_profiling_async_copy_enable
|
||||
hsa_amd_profiling_get_async_copy_time
|
||||
hsa_amd_profiling_convert_tick_to_system_domain
|
||||
hsa_amd_signal_create
|
||||
hsa_amd_signal_wait_any
|
||||
hsa_amd_signal_async_handler
|
||||
hsa_amd_async_function
|
||||
hsa_amd_image_get_info_max_dim
|
||||
hsa_amd_queue_cu_set_mask
|
||||
hsa_amd_queue_cu_get_mask
|
||||
hsa_amd_memory_fill
|
||||
hsa_amd_memory_async_copy
|
||||
hsa_amd_memory_async_copy_on_engine
|
||||
hsa_amd_memory_copy_engine_status
|
||||
hsa_amd_memory_get_preferred_copy_engine
|
||||
hsa_amd_memory_async_copy_rect
|
||||
hsa_amd_memory_lock
|
||||
hsa_amd_memory_lock_to_pool
|
||||
hsa_amd_memory_unlock
|
||||
hsa_amd_agent_iterate_memory_pools
|
||||
hsa_amd_agent_memory_pool_get_info
|
||||
hsa_amd_agents_allow_access
|
||||
hsa_amd_memory_pool_get_info
|
||||
hsa_amd_memory_pool_allocate
|
||||
hsa_amd_memory_pool_free
|
||||
hsa_amd_memory_pool_can_migrate
|
||||
hsa_amd_memory_migrate
|
||||
hsa_amd_interop_map_buffer
|
||||
hsa_amd_interop_unmap_buffer
|
||||
hsa_amd_image_create
|
||||
hsa_ext_image_get_capability
|
||||
hsa_ext_image_data_get_info
|
||||
hsa_ext_image_create
|
||||
hsa_ext_image_import
|
||||
hsa_ext_image_export
|
||||
hsa_ext_image_copy
|
||||
hsa_ext_image_clear
|
||||
hsa_ext_image_destroy
|
||||
hsa_ext_sampler_create
|
||||
hsa_ext_sampler_create_v2
|
||||
hsa_ext_sampler_destroy
|
||||
hsa_ext_image_get_capability_with_layout
|
||||
hsa_ext_image_data_get_info_with_layout
|
||||
hsa_ext_image_create_with_layout
|
||||
hsa_amd_pointer_info
|
||||
hsa_amd_pointer_info_set_userdata
|
||||
hsa_amd_ipc_memory_create
|
||||
hsa_amd_ipc_memory_attach
|
||||
hsa_amd_ipc_memory_detach
|
||||
hsa_amd_ipc_signal_create
|
||||
hsa_amd_ipc_signal_attach
|
||||
hsa_amd_register_system_event_handler
|
||||
hsa_amd_queue_set_priority
|
||||
hsa_amd_register_deallocation_callback
|
||||
hsa_amd_deregister_deallocation_callback
|
||||
hsa_amd_signal_value_pointer
|
||||
_amdgpu_r_debug
|
||||
hsa_amd_svm_attributes_set
|
||||
hsa_amd_svm_attributes_get
|
||||
hsa_amd_svm_prefetch_async
|
||||
hsa_amd_spm_acquire
|
||||
hsa_amd_spm_release
|
||||
hsa_amd_spm_set_dest_buffer
|
||||
hsa_amd_portable_export_dmabuf
|
||||
hsa_amd_portable_close_dmabuf
|
||||
hsa_amd_vmem_address_reserve
|
||||
hsa_amd_vmem_address_reserve_align
|
||||
hsa_amd_vmem_address_free
|
||||
hsa_amd_vmem_handle_create
|
||||
hsa_amd_vmem_handle_release
|
||||
hsa_amd_vmem_map
|
||||
hsa_amd_vmem_unmap
|
||||
hsa_amd_vmem_set_access
|
||||
hsa_amd_vmem_get_access
|
||||
hsa_amd_vmem_export_shareable_handle
|
||||
hsa_amd_vmem_import_shareable_handle
|
||||
hsa_amd_vmem_retain_alloc_handle
|
||||
hsa_amd_vmem_get_alloc_properties_from_handle
|
||||
hsa_amd_agent_set_async_scratch_limit
|
||||
hsa_ven_amd_pcs_iterate_configuration
|
||||
hsa_ven_amd_pcs_create
|
||||
hsa_ven_amd_pcs_create_from_id
|
||||
hsa_ven_amd_pcs_destroy
|
||||
hsa_ven_amd_pcs_start
|
||||
hsa_ven_amd_pcs_stop
|
||||
hsa_ven_amd_pcs_flush
|
||||
hsa_amd_queue_get_info
|
||||
hsa_amd_enable_logging
|
||||
hsa_amd_signal_wait_all
|
||||
hsa_amd_portable_export_dmabuf_v2
|
||||
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2023, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2023-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -40,10 +40,16 @@
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#if defined(__linux__)
|
||||
#include <unistd.h>
|
||||
#include <elf.h>
|
||||
#include <fcntl.h>
|
||||
#include <sys/resource.h>
|
||||
#include <elf.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#include <stdio.h>
|
||||
#include <win32/elf.h>
|
||||
#endif
|
||||
#include <fcntl.h>
|
||||
#include <cstring>
|
||||
#include <vector>
|
||||
#include <sstream>
|
||||
@@ -270,11 +276,14 @@ struct LoadSegmentBuilder : public SegmentBuilder {
|
||||
if (fd_ == -1) return HSA_STATUS_ERROR;
|
||||
|
||||
size_t done = 0;
|
||||
ssize_t read;
|
||||
size_t read;
|
||||
do {
|
||||
#if defined(__linux__)
|
||||
read = pread(fd_, static_cast<char *>(buf) + done, buf_size - done,
|
||||
offset + done);
|
||||
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
if (read == -1 && errno != EINTR) {
|
||||
perror("Failed to read GPU memory");
|
||||
return HSA_STATUS_ERROR;
|
||||
@@ -305,6 +314,7 @@ hsa_status_t build_core_dump(const std::string& filename, const SegmentsInfo& se
|
||||
debug_print("Core file size over limit\n");
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
#if defined(__linux__)
|
||||
int fd = open(filename.c_str(), O_WRONLY | O_CREAT | O_EXCL, S_IRUSR | S_IWUSR);
|
||||
if (fd == -1) {
|
||||
perror("Failed to create GPU coredump");
|
||||
@@ -423,6 +433,9 @@ hsa_status_t build_core_dump(const std::string& filename, const SegmentsInfo& se
|
||||
}
|
||||
printf("GPU core dump created: %s\n", filename.c_str());
|
||||
close(fd);
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
#endif
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
} // namespace impl
|
||||
@@ -431,7 +444,7 @@ hsa_status_t dump_gpu_core() {
|
||||
impl::NoteSegmentBuilder nbuilder;
|
||||
impl::LoadSegmentBuilder lbuilder;
|
||||
impl::SegmentsInfo segments;
|
||||
|
||||
#if defined(__linux__)
|
||||
struct rlimit rlimit;
|
||||
|
||||
if (getrlimit(RLIMIT_CORE, &rlimit)) {
|
||||
@@ -452,6 +465,10 @@ hsa_status_t dump_gpu_core() {
|
||||
std::stringstream st;
|
||||
st << PREFIX_FILE_NAME << "." << getpid();
|
||||
return build_core_dump(st.str(), segments, rlimit.rlim_cur);
|
||||
#else
|
||||
assert(!"Unimplemented!");
|
||||
return HSA_STATUS_SUCCESS;
|
||||
#endif
|
||||
}
|
||||
} // namespace coredump
|
||||
} // namespace amd
|
||||
|
||||
Разница между файлами не показана из-за своего большого размера
Загрузить разницу
@@ -3,7 +3,7 @@
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
|
||||
// Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
@@ -40,12 +40,14 @@
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include "executable.hpp"
|
||||
|
||||
#include <libelf.h>
|
||||
#include <limits.h>
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <cstdint>
|
||||
#endif
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstddef>
|
||||
@@ -61,6 +63,8 @@
|
||||
#include "amd_options.hpp"
|
||||
#include "core/util/utils.h"
|
||||
|
||||
#include "executable.hpp"
|
||||
|
||||
#include "AMDHSAKernelDescriptor.h"
|
||||
|
||||
using namespace rocr::amd::hsa;
|
||||
@@ -88,7 +92,12 @@ static __forceinline link_map*& r_debug_tail() {
|
||||
namespace rocr {
|
||||
// Having a side effect prevents call site optimization that allows removal of a noinline function call
|
||||
// with no side effect.
|
||||
__attribute__((noinline)) void _loader_debug_state() {
|
||||
#if defined(__linux__)
|
||||
__attribute__((noinline))
|
||||
#else
|
||||
__declspec(noinline)
|
||||
#endif
|
||||
void _loader_debug_state() {
|
||||
static volatile int function_needs_a_side_effect = 0;
|
||||
function_needs_a_side_effect ^= 1;
|
||||
}
|
||||
|
||||
@@ -48,7 +48,9 @@
|
||||
#include <cstdint>
|
||||
#include <iostream>
|
||||
#include <libelf.h>
|
||||
#if defined(__linux__)
|
||||
#include <link.h>
|
||||
#endif
|
||||
#include <list>
|
||||
#include <string>
|
||||
#include <unordered_map>
|
||||
@@ -62,6 +64,74 @@
|
||||
#include "inc/amd_hsa_kernel_code.h"
|
||||
#include "amd_hsa_locks.hpp"
|
||||
|
||||
#if defined(_WIN32) || defined(_WIN64)
|
||||
// r_version history:
|
||||
// 1: Initial debug protocol
|
||||
// 2: New trap handler ABI. The reason for halting a wave is recorded in ttmp11[8:7].
|
||||
// 3: New trap handler ABI. A wave halted at S_ENDPGM rewinds its PC by 8 bytes, and sets
|
||||
// ttmp11[9]=1. 4: New trap handler ABI. Save the trap id in ttmp11[16:9] 5: New trap handler ABI.
|
||||
// Save the PC in ttmp11[22:7] ttmp6[31:0], and park the wave if stopped 6: New trap handler ABI.
|
||||
// ttmp6[25:0] contains dispatch index modulo queue size 7: New trap handler ABI. Send interrupts as
|
||||
// a bitmask, coalescing concurrent exceptions. 8: New trap handler ABI. for gfx942: Initialize
|
||||
// ttmp[4:5] if ttmp11[31] == 0. 9: New trap handler ABI. For gfx11: Save PC in ttmp11[22:7]
|
||||
// ttmp6[31:0], and park the wave if stopped. 10: New trap handler ABI. Set status.skip_export when
|
||||
// halting the wave.
|
||||
// For gfx942, set ttmp6[31] = 0 if ttmp11[31] == 0.
|
||||
#if _WIN64
|
||||
#define __WORDSIZE 64
|
||||
#else
|
||||
#define __WORDSIZE 32
|
||||
#endif
|
||||
|
||||
#define __ELF_NATIVE_CLASS __WORDSIZE
|
||||
|
||||
/* We use this macro to refer to ELF types independent of the native wordsize.
|
||||
`ElfW(TYPE)' is used in place of `Elf32_TYPE' or `Elf64_TYPE'. */
|
||||
#define _ElfW_1(e, w, t) e##w##t
|
||||
#define _ElfW(e, w, t) _ElfW_1(e, w, _##t)
|
||||
#define ElfW(type) _ElfW(Elf, __ELF_NATIVE_CLASS, type)
|
||||
|
||||
/* Structure describing a loaded shared object. The `l_next' and `l_prev'
|
||||
members form a chain of all the shared objects loaded at startup.
|
||||
|
||||
These data structures exist in space used by the run-time dynamic linker;
|
||||
modifying them may have disastrous results. */
|
||||
|
||||
struct link_map {
|
||||
/* These first few members are part of the protocol with the debugger.
|
||||
This is the same format used in SVR4. */
|
||||
|
||||
ElfW(Addr) l_addr; /* Difference between the address in the ELF
|
||||
file and the addresses in memory. */
|
||||
char* l_name; /* Absolute file name object was found in. */
|
||||
ElfW(Dyn) * l_ld; /* Dynamic section of the shared object. */
|
||||
struct link_map *l_next, *l_prev; /* Chain of loaded objects. */
|
||||
};
|
||||
|
||||
struct r_debug {
|
||||
/* Version number for this protocol. It should be greater than 0. */
|
||||
int r_version;
|
||||
|
||||
struct link_map* r_map; /* Head of the chain of loaded objects. */
|
||||
|
||||
/* This is the address of a function internal to the run-time linker,
|
||||
that will always be called when the linker begins to map in a
|
||||
library or unmap it, and again when the mapping change is complete.
|
||||
The debugger can set a breakpoint at this address if it wants to
|
||||
notice shared object mapping changes. */
|
||||
ElfW(Addr) r_brk;
|
||||
enum RT {
|
||||
/* This state value describes the mapping change taking place when
|
||||
the `r_brk' address is called. */
|
||||
RT_CONSISTENT, /* Mapping change is complete. */
|
||||
RT_ADD, /* Beginning to add a new object. */
|
||||
RT_DELETE /* Beginning to remove an object mapping. */
|
||||
} r_state;
|
||||
|
||||
ElfW(Addr) r_ldbase; /* Base address the linker is loaded at. */
|
||||
};
|
||||
#endif
|
||||
|
||||
namespace rocr {
|
||||
namespace amd {
|
||||
namespace hsa {
|
||||
@@ -604,7 +674,7 @@ public:
|
||||
hsa_status_t QuerySegmentDescriptors(
|
||||
hsa_ven_amd_loader_segment_descriptor_t *segment_descriptors,
|
||||
size_t *num_segment_descriptors) override;
|
||||
|
||||
#undef FindExecutable
|
||||
hsa_executable_t FindExecutable(uint64_t device_address) override;
|
||||
|
||||
uint64_t FindHostAddress(uint64_t device_address) override;
|
||||
|
||||
@@ -306,7 +306,7 @@ hsa_status_t PcsRuntime::PcSamplingCreateInternal(
|
||||
|
||||
hsa_status_t PcsRuntime::PcSamplingDestroy(hsa_ven_amd_pcs_t handle) {
|
||||
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
|
||||
if (pcSamplingSessionIt == pc_sampling_.end()) {
|
||||
debug_warning(false && "Cannot find PcSampling session");
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
@@ -320,7 +320,7 @@ hsa_status_t PcsRuntime::PcSamplingDestroy(hsa_ven_amd_pcs_t handle) {
|
||||
|
||||
hsa_status_t PcsRuntime::PcSamplingStart(hsa_ven_amd_pcs_t handle) {
|
||||
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
|
||||
if (pcSamplingSessionIt == pc_sampling_.end()) {
|
||||
debug_warning(false && "Cannot find PcSampling session");
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
@@ -332,7 +332,7 @@ hsa_status_t PcsRuntime::PcSamplingStart(hsa_ven_amd_pcs_t handle) {
|
||||
|
||||
hsa_status_t PcsRuntime::PcSamplingStop(hsa_ven_amd_pcs_t handle) {
|
||||
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
|
||||
if (pcSamplingSessionIt == pc_sampling_.end()) {
|
||||
debug_warning(false && "Cannot find PcSampling session");
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
@@ -344,7 +344,7 @@ hsa_status_t PcsRuntime::PcSamplingStop(hsa_ven_amd_pcs_t handle) {
|
||||
|
||||
hsa_status_t PcsRuntime::PcSamplingFlush(hsa_ven_amd_pcs_t handle) {
|
||||
ScopedAcquire<KernelMutex> lock(&pc_sampling_lock_);
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(reinterpret_cast<uint64_t>(handle.handle));
|
||||
auto pcSamplingSessionIt = pc_sampling_.find(static_cast<uint64_t>(handle.handle));
|
||||
if (pcSamplingSessionIt == pc_sampling_.end()) {
|
||||
debug_warning(false && "Cannot find PcSampling session");
|
||||
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
|
||||
|
||||
Ссылка в новой задаче
Block a user