Add 'projects/rocr-runtime/' from commit '72061a9024139fa0a99f73f9d3d4deb275670095'
git-subtree-dir: projects/rocr-runtime git-subtree-mainline:ad0fb25ed5git-subtree-split:72061a9024
This commit is contained in:
@@ -0,0 +1,38 @@
|
||||
resources:
|
||||
repositories:
|
||||
- repository: pipelines_repo
|
||||
type: github
|
||||
endpoint: ROCm
|
||||
name: ROCm/ROCm
|
||||
|
||||
variables:
|
||||
- group: common
|
||||
- template: /.azuredevops/variables-global.yml@pipelines_repo
|
||||
|
||||
trigger:
|
||||
batch: true
|
||||
branches:
|
||||
include:
|
||||
- amd-staging
|
||||
- amd-mainline
|
||||
paths:
|
||||
exclude:
|
||||
- .github
|
||||
- LICENSE.txt
|
||||
- '*.md'
|
||||
|
||||
pr:
|
||||
autoCancel: true
|
||||
branches:
|
||||
include:
|
||||
- amd-staging
|
||||
- amd-mainline
|
||||
paths:
|
||||
exclude:
|
||||
- .github
|
||||
- LICENSE.txt
|
||||
- '*.md'
|
||||
drafts: false
|
||||
|
||||
jobs:
|
||||
- template: ${{ variables.CI_COMPONENT_PATH }}/ROCR-Runtime.yml@pipelines_repo
|
||||
+8
@@ -0,0 +1,8 @@
|
||||
# Default code owners
|
||||
@kentrussell @fxkamd @dayatsin-amd
|
||||
|
||||
*.md @ROCm/rocm-documentation @kentrussell @dayatsin-amd
|
||||
*.rst @ROCm/rocm-documentation @kentrussell @dayatsin-amd
|
||||
|
||||
# Header directory for Doxygen documentation
|
||||
inc/* @ROCm/rocm-documentation @kentrussell @fxkamd @dayatsin-amd
|
||||
+5
@@ -0,0 +1,5 @@
|
||||
disabled: false
|
||||
scmId: gh-emu-rocm
|
||||
branchesToScan:
|
||||
- amd-staging
|
||||
- amd-mainline
|
||||
@@ -0,0 +1,15 @@
|
||||
name: Rocm Validation Suite KWS
|
||||
on:
|
||||
push:
|
||||
branches: [amd-staging, amd-mainline]
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
workflow_dispatch:
|
||||
jobs:
|
||||
kws:
|
||||
if: ${{ github.event_name == 'pull_request' }}
|
||||
uses: AMD-ROCm-Internal/rocm_ci_infra/.github/workflows/kws.yml@mainline
|
||||
secrets: inherit
|
||||
with:
|
||||
pr_number: ${{github.event.pull_request.number}}
|
||||
base_branch: ${{github.base_ref}}
|
||||
@@ -0,0 +1,25 @@
|
||||
name: ROCm CI Caller
|
||||
on:
|
||||
pull_request:
|
||||
branches: [amd-staging, amd-npi, release/rocm-rel-*, amd-master]
|
||||
types: [opened, reopened, synchronize]
|
||||
push:
|
||||
branches: [amd-mainline]
|
||||
workflow_dispatch:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
jobs:
|
||||
call-workflow:
|
||||
if: github.event_name != 'issue_comment' ||(github.event_name == 'issue_comment' && github.event.issue.pull_request && (startsWith(github.event.comment.body, '!verify') || startsWith(github.event.comment.body, '!verify release') || startsWith(github.event.comment.body, '!verify retest')))
|
||||
uses: AMD-ROCm-Internal/rocm_ci_infra/.github/workflows/rocm_ci.yml@mainline
|
||||
secrets: inherit
|
||||
with:
|
||||
input_sha: ${{github.event_name == 'pull_request' && github.event.pull_request.head.sha || (github.event_name == 'push' && github.sha) || (github.event_name == 'issue_comment' && github.event.issue.pull_request.head.sha) || github.sha}}
|
||||
input_pr_num: ${{github.event_name == 'pull_request' && github.event.pull_request.number || (github.event_name == 'issue_comment' && github.event.issue.number) || 0}}
|
||||
input_pr_url: ${{github.event_name == 'pull_request' && github.event.pull_request.html_url || (github.event_name == 'issue_comment' && github.event.issue.pull_request.html_url) || ''}}
|
||||
input_pr_title: ${{github.event_name == 'pull_request' && github.event.pull_request.title || (github.event_name == 'issue_comment' && github.event.issue.pull_request.title) || ''}}
|
||||
repository_name: ${{ github.repository }}
|
||||
base_ref: ${{github.event_name == 'pull_request' && github.event.pull_request.base.ref || (github.event_name == 'issue_comment' && github.event.issue.pull_request.base.ref) || github.ref}}
|
||||
trigger_event_type: ${{ github.event_name }}
|
||||
comment_text: ${{ github.event_name == 'issue_comment' && github.event.comment.body || '' }}
|
||||
@@ -0,0 +1,22 @@
|
||||
.*
|
||||
|
||||
#
|
||||
# git files that we don't want to ignore even it they are dot-files
|
||||
#
|
||||
!.gitignore
|
||||
!.mailmap
|
||||
.github*
|
||||
patches-*
|
||||
build/
|
||||
outgoing/
|
||||
Makefile
|
||||
|
||||
# documentation artifacts
|
||||
_build/
|
||||
_doxygen/
|
||||
_images/
|
||||
_static/
|
||||
_templates/
|
||||
_toc.yml
|
||||
doxygen
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
# Read the Docs configuration file
|
||||
# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details
|
||||
|
||||
version: 2
|
||||
|
||||
sphinx:
|
||||
configuration: runtime/docs/conf.py
|
||||
|
||||
formats: [htmlzip, pdf, epub]
|
||||
|
||||
python:
|
||||
install:
|
||||
- requirements: runtime/docs/sphinx/requirements.txt
|
||||
|
||||
build:
|
||||
os: ubuntu-22.04
|
||||
tools:
|
||||
python: "3.10"
|
||||
@@ -0,0 +1,315 @@
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
|
||||
cmake_minimum_required(VERSION 3.7)
|
||||
|
||||
# Set the project name
|
||||
project("rocr")
|
||||
|
||||
set(CMAKE_VERBOSE_MAKEFILE ON)
|
||||
## Expose static library option
|
||||
if ( NOT DEFINED BUILD_SHARED_LIBS )
|
||||
set ( BUILD_SHARED_LIBS ON )
|
||||
endif()
|
||||
set ( BUILD_SHARED_LIBS ${BUILD_SHARED_LIBS} CACHE BOOL "Build shared library (.so) or not.")
|
||||
|
||||
if (NOT DEFINED BUILD_ROCR)
|
||||
set(BUILD_ROCR ON)
|
||||
endif()
|
||||
|
||||
function(add_rocm_subdir subdir subdir_assigns)
|
||||
message("add_rocm_subdir() -- " ${subdir})
|
||||
# message(" subdir_assigns before:" ${subdir_assigns} "EOM")
|
||||
string(STRIP "${subdir_assigns}" subdir_assigns)
|
||||
message(" subdir_assigns:" ${subdir_assigns} "EOM")
|
||||
|
||||
# if the subdir_assigns is defined and non-empty, then..
|
||||
|
||||
if(NOT "${subdir_assigns}" STREQUAL "")
|
||||
foreach(assignment IN LISTS subdir_assigns)
|
||||
# The format of each var should be VARNAME=VALUE
|
||||
message("assignment: " ${assignment})
|
||||
string(REPLACE "=" ";" pair ${assignment})
|
||||
list(GET pair 0 var_name)
|
||||
list(GET pair 1 var_value)
|
||||
|
||||
# Set variable locally for this function and for the subdirectory
|
||||
set(${var_name} "${var_value}")
|
||||
message("The value of ${var_name} is: ${${var_name}}")
|
||||
endforeach()
|
||||
endif()
|
||||
add_subdirectory(${subdir})
|
||||
endfunction()
|
||||
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules")
|
||||
include(utils)
|
||||
|
||||
|
||||
## Get version strings
|
||||
get_version("1.18.0")
|
||||
if (${ROCM_PATCH_VERSION})
|
||||
set(VERSION_PATCH ${ROCM_PATCH_VERSION})
|
||||
endif()
|
||||
set(SO_VERSION_STRING "${VERSION_MAJOR}.${VERSION_MINOR}.${VERSION_PATCH}")
|
||||
set(PACKAGE_VERSION_STRING "${VERSION_MAJOR}.${VERSION_MINOR}.${VERSION_COMMIT_COUNT}")
|
||||
|
||||
if (NOT DEFINED BUILD_SHARED_LIBS)
|
||||
set(BUILD_SHARED_LIBS ON)
|
||||
endif()
|
||||
|
||||
# Set hsa pkg dependency with rocprofiler-register package
|
||||
# for Shared Library Only.
|
||||
if (BUILD_SHARED_LIBS)
|
||||
set(HSA_DEP_ROCPROFILER_REGISTER ON CACHE INTERNAL "")
|
||||
endif()
|
||||
|
||||
if (HSA_DEP_ROCPROFILER_REGISTER)
|
||||
string(APPEND CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS ", rocprofiler-register")
|
||||
string(APPEND CPACK_RPM_BINARY_PACKAGE_REQUIRES " rocprofiler-register")
|
||||
endif()
|
||||
|
||||
add_rocm_subdir(libhsakmt "${THUNK_DEFINITIONS}")
|
||||
set_target_properties(hsakmt PROPERTIES
|
||||
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/archive"
|
||||
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/lib"
|
||||
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/runtime")
|
||||
|
||||
if (BUILD_ROCR)
|
||||
add_rocm_subdir(runtime/hsa-runtime "${ROCR_DEFINITIONS}")
|
||||
set_target_properties(hsa-runtime64 PROPERTIES
|
||||
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/archive"
|
||||
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/lib"
|
||||
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/rocr/runtime")
|
||||
|
||||
if (BUILD_SHARED_LIBS)
|
||||
add_dependencies(hsa-runtime64 hsakmt)
|
||||
else()
|
||||
add_dependencies(hsa-runtime64 hsakmt-staticdrm)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Optionally record the package's find module in the user's package cache.
|
||||
if ( NOT DEFINED EXPORT_TO_USER_PACKAGE_REGISTRY )
|
||||
set ( EXPORT_TO_USER_PACKAGE_REGISTRY "off")
|
||||
endif()
|
||||
set ( EXPORT_TO_USER_PACKAGE_REGISTRY ${EXPORT_TO_USER_PACKAGE_REGISTRY} CACHE BOOL "Add cmake package config location to the user's cmake package registry.")
|
||||
if(${EXPORT_TO_USER_PACKAGE_REGISTRY})
|
||||
# Enable writing to the registry
|
||||
set(CMAKE_EXPORT_PACKAGE_REGISTRY ON)
|
||||
# Generate a target file for the build
|
||||
export(TARGETS ${CORE_RUNTIME_NAME} NAMESPACE ${CORE_RUNTIME_NAME}:: FILE ${CORE_RUNTIME_NAME}Targets.cmake)
|
||||
# Record the package in the user's cache.
|
||||
export(PACKAGE ${CORE_RUNTIME_NAME})
|
||||
endif()
|
||||
|
||||
## Packaging directives
|
||||
set(CPACK_VERBOSE 1)
|
||||
set(CPACK_GENERATOR "DEB;RPM" CACHE STRING "Package types to build")
|
||||
set(ENABLE_LDCONFIG ON CACHE BOOL "Set library links and caches using ldconfig.")
|
||||
|
||||
# From libhsakmt:
|
||||
set(CPACK_PACKAGING_INSTALL_PREFIX "${CMAKE_INSTALL_PREFIX}" CACHE STRING "Default packaging prefix.")
|
||||
|
||||
if(DEFINED CPACK_PACKAGING_INSTALL_PREFIX)
|
||||
set(CPACK_RPM_EXCLUDE_FROM_AUTO_FILELIST_ADDITION "${CPACK_PACKAGING_INSTALL_PREFIX} ${CPACK_PACKAGING_INSTALL_PREFIX}/${CMAKE_INSTALL_BINDIR}")
|
||||
endif()
|
||||
|
||||
# ASAN Package will have libraries and license file
|
||||
if (ENABLE_ASAN_PACKAGING)
|
||||
# ASAN Package requires only asan component with libraries and license file
|
||||
set(CPACK_COMPONENTS_ALL asan)
|
||||
else()
|
||||
set(CPACK_COMPONENTS_ALL binary dev)
|
||||
endif()
|
||||
set(CPACK_DEB_COMPONENT_INSTALL ON)
|
||||
set(CPACK_RPM_COMPONENT_INSTALL ON)
|
||||
set(CPACK_PACKAGE_VENDOR "Advanced Micro Devices, Inc.")
|
||||
set(CPACK_PACKAGE_VERSION ${PACKAGE_VERSION_STRING})
|
||||
set(CPACK_PACKAGE_CONTACT "AMD HSA Support <dl.HSA-Runtime-Support@amd.com>")
|
||||
set(CPACK_COMPONENT_DESCRIPTION "AMD Heterogeneous System Architecture HSA - Linux HSA Runtime for Boltzmann (ROCm) platforms\nIncludes HSAKMT, the user-mode API interfaces used to interact with the ROCk driver.\n Contains the headers, pkgonfig and\n cmake files for ROCT.")
|
||||
set(CPACK_COMPONENT_BINARY_DESCRIPTION "AMD Heterogeneous System Architecture HSA - Linux HSA Runtime for Boltzmann (ROCm) platforms")
|
||||
set(CPACK_COMPONENT_DEV_DESCRIPTION "AMD Heterogeneous System Architecture HSA development package.\n This package contains the headers and cmake files for the rocr-runtime package.")
|
||||
set(CPACK_COMPONENT_ASAN_DESCRIPTION "AMD Heterogeneous System Architecture HSA - Linux HSA instrumented libraries for Boltzmann (ROCm) platforms")
|
||||
|
||||
if (DEFINED ENV{ROCM_LIBPATCH_VERSION})
|
||||
set(CPACK_PACKAGE_VERSION "${CPACK_PACKAGE_VERSION}.$ENV{ROCM_LIBPATCH_VERSION}")
|
||||
message("Using CPACK_PACKAGE_VERSION ${CPACK_PACKAGE_VERSION}")
|
||||
endif()
|
||||
|
||||
# Debian package specific variables
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_NAME "hsa-rocr")
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_NAME "hsa-rocr-dev")
|
||||
set(CPACK_DEBIAN_ASAN_PACKAGE_NAME "hsa-rocr-asan")
|
||||
if (DEFINED ENV{CPACK_DEBIAN_PACKAGE_RELEASE})
|
||||
set(CPACK_DEBIAN_PACKAGE_RELEASE $ENV{CPACK_DEBIAN_PACKAGE_RELEASE})
|
||||
else()
|
||||
set(CPACK_DEBIAN_PACKAGE_RELEASE "local")
|
||||
endif()
|
||||
message("Using CPACK_DEBIAN_PACKAGE_RELEASE ${CPACK_DEBIAN_PACKAGE_RELEASE}")
|
||||
set(CPACK_DEBIAN_FILE_NAME "DEB-DEFAULT")
|
||||
set(CPACK_DEBIAN_PACKAGE_HOMEPAGE "https://github.com/RadeonOpenCompute/ROCR-Runtime")
|
||||
|
||||
## Process the Debian install/remove scripts to update the CPACK variables
|
||||
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/DEBIAN/Binary/postinst.in DEBIAN/Binary/postinst @ONLY)
|
||||
configure_file(${CMAKE_CURRENT_SOURCE_DIR}/DEBIAN/Binary/prerm.in DEBIAN/Binary/prerm @ONLY)
|
||||
file(COPY ${CMAKE_CURRENT_SOURCE_DIR}/DEBIAN/preinst DESTINATION ${CMAKE_CURRENT_BINARY_DIR}/DEBIAN)
|
||||
set (CPACK_DEBIAN_BINARY_PACKAGE_CONTROL_EXTRA "DEBIAN/preinst;DEBIAN/Binary/postinst;DEBIAN/Binary/prerm")
|
||||
# Needed since some packages still say they need hsakmt-roct
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_REPLACES "hsakmt-roct,hsakmt-roct-dev,hsa-ext-rocr-dev")
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_PROVIDES "hsakmt-roct,hsakmt-roct-dev,hsa-ext-rocr-dev")
|
||||
#TODO: hsa-ext-rocr-dev can be added to conflicts list and remove CPACK_DEBIAN_DEV_PACKAGE_BREAKS
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_CONFLICTS "hsakmt-roct,hsakmt-roct-dev")
|
||||
# package dependencies
|
||||
set(CPACK_DEBIAN_PACKAGE_DEPENDS "libdrm-amdgpu-dev | libdrm-dev, rocm-core")
|
||||
set(CPACK_DEBIAN_PACKAGE_RECOMMENDS "libdrm-amdgpu-dev")
|
||||
# Setting devel package dependendent version
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_DEPENDS "libdrm-amdgpu-dev | libdrm-dev, rocm-core, hsa-rocr")
|
||||
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_RECOMMENDS "libdrm-amdgpu-dev")
|
||||
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS "libdrm-amdgpu-amdgpu1 | libdrm-amdgpu1, libnuma1, libelf1")
|
||||
set(CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS "libdrm-amdgpu-dev | libdrm-dev, rocm-core-asan, libdrm-amdgpu-amdgpu1 | libdrm-amdgpu1, libnuma1, libelf1")
|
||||
set(CPACK_DEBIAN_ASAN_PACKAGE_RECOMMENDS "libdrm-amdgpu-dev")
|
||||
|
||||
set(CPACK_DEBIAN_BINARY_PACKAGE_RECOMMENDS "libdrm-amdgpu-amdgpu1")
|
||||
if (ROCM_DEP_ROCMCORE)
|
||||
string(APPEND CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS ", rocm-core")
|
||||
string(APPEND CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS ", rocm-core-asan")
|
||||
endif()
|
||||
if (HSA_DEP_ROCPROFILER_REGISTER)
|
||||
string(APPEND CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS ", rocprofiler-register")
|
||||
endif()
|
||||
# Declare package relationships (hsa-ext-rocr-dev is a legacy package that we subsume)
|
||||
set(CPACK_DEBIAN_DEV_PACKAGE_BREAKS "hsa-ext-rocr-dev")
|
||||
|
||||
# RPM package specific variables
|
||||
set(EL7_DISTRO "FALSE")
|
||||
Checksetel7(EL7_DISTRO)
|
||||
set(CPACK_RPM_BINARY_PACKAGE_NAME "hsa-rocr")
|
||||
# Since we changed the package name to match RPM specs, take care of older builds that had -dev installed
|
||||
# Also cover the fact that this now replaces the old binary package hsakmt-roct
|
||||
set(CPACK_RPM_DEV_PACKAGE_PROVIDES "hsakmt-roct,hsakmt-roct-devel,hsakmt-roct-dev,hsa-ext-rocr-dev")
|
||||
set(CPACK_RPM_DEV_PACKAGE_OBSOLETES "hsakmt-roct,hsakmt-roct-devel,hsakmt-roct-dev,hsa-ext-rocr-dev")
|
||||
|
||||
set(CPACK_RPM_DEV_PACKAGE_NAME "hsa-rocr-devel")
|
||||
set(CPACK_RPM_ASAN_PACKAGE_NAME "hsa-rocr-asan")
|
||||
if (DEFINED ENV{CPACK_RPM_PACKAGE_RELEASE})
|
||||
set(CPACK_RPM_PACKAGE_RELEASE $ENV{CPACK_RPM_PACKAGE_RELEASE})
|
||||
else()
|
||||
set(CPACK_RPM_PACKAGE_RELEASE "local")
|
||||
endif()
|
||||
|
||||
string(APPEND CPACK_RPM_PACKAGE_RELEASE "%{?dist}")
|
||||
set(CPACK_RPM_FILE_NAME "RPM-DEFAULT")
|
||||
message("CPACK_RPM_PACKAGE_RELEASE: ${CPACK_RPM_PACKAGE_RELEASE}")
|
||||
set(CPACK_RPM_PACKAGE_LICENSE "NCSA")
|
||||
|
||||
## Process the Rpm install/remove scripts to update the CPACK variables
|
||||
configure_file("${CMAKE_CURRENT_SOURCE_DIR}/RPM/Binary/post.in" RPM/Binary/post @ONLY)
|
||||
configure_file("${CMAKE_CURRENT_SOURCE_DIR}/RPM/Binary/postun.in" RPM/Binary/postun @ONLY)
|
||||
file(COPY ${CMAKE_CURRENT_SOURCE_DIR}/RPM/preinst DESTINATION ${CMAKE_CURRENT_BINARY_DIR}/RPM)
|
||||
set (CPACK_RPM_PRE_INSTALL_SCRIPT_FILE "${CMAKE_CURRENT_BINARY_DIR}/RPM/preinst")
|
||||
|
||||
set(CPACK_RPM_BINARY_POST_INSTALL_SCRIPT_FILE "${CMAKE_CURRENT_BINARY_DIR}/RPM/Binary/post")
|
||||
set(CPACK_RPM_BINARY_POST_UNINSTALL_SCRIPT_FILE "${CMAKE_CURRENT_BINARY_DIR}/RPM/Binary/postun")
|
||||
|
||||
# package dependencies
|
||||
set(CPACK_RPM_DEV_PACKAGE_REQUIRES "rocm-core , hsa-rocr")
|
||||
|
||||
#
|
||||
if (${EL7_DISTRO} STREQUAL "TRUE")
|
||||
set(CPACK_RPM_BINARY_PACKAGE_REQUIRES "libdrm-amdgpu, numactl-libs")
|
||||
set(CPACK_RPM_ASAN_PACKAGE_REQUIRES "libdrm-amdgpu, numactl-libs, libdrm-amdgpu-devel")
|
||||
set(CPACK_RPM_PACKAGE_REQUIRES "libdrm-amdgpu-devel")
|
||||
string(APPEND CPACK_RPM_DEV_PACKAGE_REQUIRES ", libdrm-amdgpu-devel")
|
||||
else()
|
||||
set(CPACK_RPM_BINARY_PACKAGE_REQUIRES "(libdrm-amdgpu or libdrm or libdrm_amdgpu1), (libnuma1 or numactl-libs)")
|
||||
set(CPACK_RPM_ASAN_PACKAGE_REQUIRES "(libdrm-amdgpu or libdrm or libdrm_amdgpu1), (libnuma1 or numactl-libs), (libdrm-amdgpu-devel or libdrm-devel)")
|
||||
set(CPACK_RPM_USER_BINARY_SPECFILE "${CMAKE_CURRENT_SOURCE_DIR}/RPM/hsa-rocr.spec.in")
|
||||
set(CPACK_RPM_PACKAGE_RECOMMENDS "libdrm-amdgpu, libdrm-amdgpu-devel")
|
||||
|
||||
set(CPACK_RPM_PACKAGE_REQUIRES "(libdrm-amdgpu-devel or libdrm-devel)")
|
||||
string(APPEND CPACK_RPM_DEV_PACKAGE_REQUIRES ", (libdrm-amdgpu-devel or libdrm-devel)")
|
||||
set(CPACK_RPM_DEV_PACKAGE_RECOMMENDS "libdrm-amdgpu-devel")
|
||||
set(CPACK_RPM_ASAN_PACKAGE_RECOMMENDS "libdrm-amdgpu-devel")
|
||||
|
||||
endif()
|
||||
|
||||
if (ROCM_DEP_ROCMCORE)
|
||||
string(APPEND CPACK_RPM_BINARY_PACKAGE_REQUIRES " rocm-core")
|
||||
string(APPEND CPACK_RPM_ASAN_PACKAGE_REQUIRES " rocm-core-asan")
|
||||
else()
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_PACKAGE_REQUIRES ${CPACK_RPM_PACKAGE_REQUIRES})
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_DEBIAN_PACKAGE_DEPENDS ${CPACK_DEBIAN_PACKAGE_DEPENDS})
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_DEV_PACKAGE_REQUIRES ${CPACK_RPM_DEV_PACKAGE_REQUIRES})
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_DEBIAN_DEV_PACKAGE_DEPENDS ${CPACK_DEBIAN_DEV_PACKAGE_DEPENDS})
|
||||
string(REGEX REPLACE ",? ?rocm-core-asan" "" CPACK_RPM_ASAN_PACKAGE_REQUIRES ${CPACK_RPM_ASAN_PACKAGE_REQUIRES})
|
||||
string(REGEX REPLACE ",? ?rocm-core-asan" "" CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS ${CPACK_DEBIAN_ASAN_PACKAGE_DEPENDS})
|
||||
endif()
|
||||
if (HSA_DEP_ROCPROFILER_REGISTER)
|
||||
string(APPEND CPACK_RPM_BINARY_PACKAGE_REQUIRES " rocprofiler-register")
|
||||
endif()
|
||||
|
||||
if(NOT BUILD_SHARED_LIBS)
|
||||
# Suffix package name with static
|
||||
set(CPACK_RPM_STATIC_PACKAGE_NAME "hsa-rocr-static-devel")
|
||||
set(CPACK_DEBIAN_STATIC_PACKAGE_NAME "hsa-rocr-static-dev")
|
||||
set(CPACK_COMPONENT_STATIC_DESCRIPTION "HSA (Heterogenous System Architecture) core runtime - Linux static libraries")
|
||||
set(CPACK_RPM_STATIC_PACKAGE_REQUIRES "${CPACK_RPM_BINARY_PACKAGE_REQUIRES}")
|
||||
set(CPACK_DEBIAN_STATIC_PACKAGE_DEPENDS "${CPACK_DEBIAN_BINARY_PACKAGE_DEPENDS}")
|
||||
endif()
|
||||
|
||||
## Include packaging
|
||||
include(CPack)
|
||||
|
||||
# static package generation
|
||||
# Group binary and dev component to single package
|
||||
if(NOT BUILD_SHARED_LIBS)
|
||||
cpack_add_component_group("static")
|
||||
cpack_add_component(binary GROUP static)
|
||||
cpack_add_component(dev GROUP static)
|
||||
endif()
|
||||
|
||||
cpack_add_component(asan
|
||||
DISPLAY_NAME "ASAN"
|
||||
DESCRIPTION "ASAN libraries for rocr-runtime")
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/bin/bash
|
||||
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2020-2021, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
set -e
|
||||
|
||||
# left-hand term originates from @ENABLE_LDCONFIG@ = ON/OFF at package build
|
||||
do_ldconfig() {
|
||||
if [ "@ENABLE_LDCONFIG@" == "ON" ]; then
|
||||
echo @CPACK_PACKAGING_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@ > /etc/ld.so.conf.d/rocr-runtime.conf
|
||||
ldconfig
|
||||
fi
|
||||
}
|
||||
|
||||
case "$1" in
|
||||
( configure )
|
||||
do_ldconfig
|
||||
;;
|
||||
( abort-upgrade | abort-remove | abort-deconfigure )
|
||||
echo "$1"
|
||||
;;
|
||||
( * )
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,64 @@
|
||||
#!/bin/bash
|
||||
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2020-2021, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
set -e
|
||||
|
||||
# left-hand term originates from @ENABLE_LDCONFIG@ = ON/OFF at package build
|
||||
rm_ldconfig() {
|
||||
if [ "@ENABLE_LDCONFIG@" == "ON" ]; then
|
||||
rm -f /etc/ld.so.conf.d/rocr-runtime.conf
|
||||
ldconfig
|
||||
fi
|
||||
}
|
||||
|
||||
case "$1" in
|
||||
( remove | upgrade)
|
||||
rm_ldconfig
|
||||
;;
|
||||
( purge )
|
||||
;;
|
||||
( * )
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,56 @@
|
||||
#!/bin/bash
|
||||
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2020-2021, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
set -e
|
||||
|
||||
case "$1" in
|
||||
( configure )
|
||||
# Workaround for CPACK directory symlink handling error.
|
||||
mkdir -p @CPACK_PACKAGING_INSTALL_PREFIX@/hsa/include
|
||||
ln -sf ../../@CMAKE_INSTALL_INCLUDEDIR@/hsa @CPACK_PACKAGING_INSTALL_PREFIX@/hsa/include/hsa
|
||||
;;
|
||||
( * )
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,57 @@
|
||||
#!/bin/bash
|
||||
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2020-2021, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
set -e
|
||||
|
||||
case "$1" in
|
||||
( remove | upgrade )
|
||||
# Workaround for CPACK directory symlink handling error.
|
||||
# Needed for remove and upgrade scenarios since
|
||||
# upgrade installs to new folder and old folders need to be cleaned
|
||||
rm -rf @CPACK_PACKAGING_INSTALL_PREFIX@/hsa
|
||||
;;
|
||||
( * )
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,23 @@
|
||||
#!/bin/bash
|
||||
|
||||
echo "Pre-install check for ROCr."
|
||||
|
||||
# Check for old installations...
|
||||
if ls /usr/lib/libhsa-runtime* 1> /dev/null 2>&1; then
|
||||
echo "An old version of libhsa-runtime was found in /usr/lib."
|
||||
echo "This must be uninstalled before proceeding with the installation"
|
||||
echo "to avoid potential incompatibilities."
|
||||
|
||||
read -r -p "Do you want to uninstall the old version? [y/N] " response
|
||||
if [ "$response" = "y" ]; then
|
||||
if ! rm -rf /usr/lib/libhsa-runtime*; then
|
||||
echo "Failed to remove /usr/lib/libhsa-runtime* files."
|
||||
echo "Try to uninstall these files manually."
|
||||
exit 1
|
||||
fi
|
||||
echo "Old version uninstalled."
|
||||
else
|
||||
echo "The old and new versions of ROCm are incompatible. Installation aborted."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
@@ -0,0 +1,37 @@
|
||||
The University of Illinois/NCSA
|
||||
Open Source License (NCSA)
|
||||
|
||||
Copyright (c) 2014-2025, Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
Developed by:
|
||||
|
||||
AMD Research and AMD HSA Software Development
|
||||
|
||||
Advanced Micro Devices, Inc.
|
||||
|
||||
www.amd.com
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to
|
||||
deal with the Software without restriction, including without limitation
|
||||
the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
and/or sell copies of the Software, and to permit persons to whom the
|
||||
Software is furnished to do so, subject to the following conditions:
|
||||
|
||||
- Redistributions of source code must retain the above copyright notice,
|
||||
this list of conditions and the following disclaimers.
|
||||
- Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimers in
|
||||
the documentation and/or other materials provided with the distribution.
|
||||
- Neither the names of Advanced Micro Devices, Inc,
|
||||
nor the names of its contributors may be used to endorse or promote
|
||||
products derived from this Software without specific prior written
|
||||
permission.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
DEALINGS WITH THE SOFTWARE.
|
||||
@@ -0,0 +1,172 @@
|
||||
# ROCR Runtime
|
||||
|
||||
This ROCm Runtime (ROCr) repo combines 2 previously separate repos into a single repo:
|
||||
- The HSA Runtime (`hsa-runtime`) for AMD GPU application development and
|
||||
- The ROCt Thunk Library (`libhsakmt`), a "thunk" interface to the ROCm kernel driver (ROCk), used by the runtime.
|
||||
|
||||
## Infrastructure
|
||||
|
||||
The HSA runtime is a thin, user-mode API that exposes the necessary interfaces to access and interact with graphics hardware driven by the AMDGPU driver set and the ROCK kernel driver. Together they enable programmers to directly harness the power of AMD discrete graphics devices by allowing host applications to launch compute kernels directly to the graphics hardware.
|
||||
|
||||
The capabilities expressed by the HSA Runtime API are:
|
||||
|
||||
* Error handling
|
||||
* Runtime initialization and shutdown
|
||||
* System and agent information
|
||||
* Signals and synchronization
|
||||
* Architected dispatch
|
||||
* Memory management
|
||||
* HSA runtime fits into a typical software architecture stack.
|
||||
|
||||
The HSA runtime provides direct access to the graphics hardware to give the programmer more control of the execution. An example of low level hardware access is the support of one or more user mode queues provides programmers with a low-latency kernel dispatch interface, allowing them to develop customized dispatch algorithms specific to their application.
|
||||
|
||||
The HSA Architected Queuing Language is an open standard, defined by the HSA Foundation, specifying the packet syntax used to control supported AMD/ATI Radeon (c) graphics devices. The AQL language supports several packet types, including packets that can command the hardware to automatically resolve inter-packet dependencies (barrier AND & barrier OR packet), kernel dispatch packets and agent dispatch packets.
|
||||
|
||||
In addition to user mode queues and AQL, the HSA runtime exposes various virtual address ranges that can be accessed by one or more of the system's graphics devices, and possibly the host. The exposed virtual address ranges either support a fine grained or a coarse grained access. Updates to memory in a fine grained region are immediately visible to all devices that can access it, but only one device can have access to a coarse grained allocation at a time. Ownership of a coarse grained region can be changed using the HSA runtime memory APIs, but this transfer of ownership must be explicitly done by the host application.
|
||||
|
||||
Programmers should consult the HSA Runtime Programmer's Reference Manual for a full description of the HSA Runtime APIs, AQL and the HSA memory policy.
|
||||
|
||||
## Known issues
|
||||
|
||||
* Each HSA process creates an internal DMA queue, but there is a system-wide limit of four DMA queues. When the limit is reached HSA processes will use internal kernels for copies.
|
||||
|
||||
## Artifacts produced by the build
|
||||
|
||||
- **libhsakmt (ROCt)** - User-mode API interfaces for interacting with the ROCk driver
|
||||
- **Runtime (ROCr)** - Core runtime supporting HSA standards
|
||||
- **rocrtst** - Runtime test suites for HSA implementation validation and performance testing
|
||||
- **kfdtest** - Validation tests for ROCt
|
||||
|
||||
## Building the ROCR Runtime
|
||||
|
||||
### Target platform requirements
|
||||
Please see the [ROCm System requirements (Linux)](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/reference/system-requirements.html).
|
||||
|
||||
Ensure you have the following installed:
|
||||
|
||||
- CMake 3.7 or higher
|
||||
- `libelf-dev`
|
||||
- `g++`
|
||||
- `libdrm-amdgpu-dev` or `libdrm-dev`
|
||||
- `pkg-config`
|
||||
- `rocm-core`
|
||||
- `rocm-llvm-dev`
|
||||
|
||||
### ROCr & ROCt Build Instructions
|
||||
1. **Clone this repository and cd into its root**
|
||||
2. **Prepare the build directory**
|
||||
```sh
|
||||
mkdir build && cd build
|
||||
```
|
||||
3. **Configure the build (example)**
|
||||
```sh
|
||||
cmake -DCMAKE_INSTALL_PREFIX=<rocm install dir> ..
|
||||
```
|
||||
e.g:
|
||||
```
|
||||
cmake -DCMAKE_INSTALL_PREFIX=/opt/rocm ..
|
||||
```
|
||||
4. **Compile the project**
|
||||
```sh
|
||||
make
|
||||
```
|
||||
5. **Install the runtime**
|
||||
```sh
|
||||
make install
|
||||
```
|
||||
6. **(Optional) Build packages**
|
||||
```sh
|
||||
make package
|
||||
```
|
||||
#### Non-default CMake Build Options
|
||||
- *Produce a release build instead of debug*
|
||||
```sh
|
||||
-DCMAKE_BUILD_TYPE=Release
|
||||
```
|
||||
|
||||
- *Control whether libhsakmt and libhsa-runtime are shared or static*
|
||||
libhsakmt is always built as a static library that gets linked into libhsa-runtime, so there is a single library generated called libhsa-runtime64{.so/.a}. If `BUILD_SHARED_LIBS` is not set, this is a shared library by default. Setting `BUILD_SHARED_LIBS` to `OFF` will make it static.
|
||||
```sh
|
||||
-DBUILD_SHARED_LIBS=ON # or OFF for static lib
|
||||
```
|
||||
### Building the tests
|
||||
#### rocrtst
|
||||
1. **Go to rocrtst root**
|
||||
```sh
|
||||
cd <rocr-runtime>/rocrtst/suites/test_common
|
||||
```
|
||||
2. **Prepare the build directory**
|
||||
```sh
|
||||
mkdir build && cd build
|
||||
```
|
||||
3. **Configure the build**
|
||||
Example configuration:
|
||||
```sh
|
||||
cmake \
|
||||
-DCMAKE_PREFIX_PATH="<rocm install root>;<llvm install root>" \
|
||||
-DROCM_DIR="$ROCM_INSTALL_PATH" \
|
||||
-DOPENCL_DIR="<rocm install root>" \
|
||||
..
|
||||
```
|
||||
4. **Compile the project**
|
||||
```sh
|
||||
make
|
||||
make rocrtst_kernels
|
||||
```
|
||||
5. ** Run the tests
|
||||
Make sure libhsa-runtime.so is in the library path; e.g.,
|
||||
```sh
|
||||
$ LD_LIBRARY_PATH=<rocm install root> ./rocrtst -h # See help options
|
||||
```
|
||||
#### kfdtest
|
||||
1. **Go to kfdtest root**
|
||||
```sh
|
||||
cd <rocr-runtime>/libhsakmt/tests/kfdtest
|
||||
```
|
||||
2. **Prepare the build directory**
|
||||
```sh
|
||||
mkdir build && cd build
|
||||
```
|
||||
3. **Configure the build**
|
||||
Example configuration:
|
||||
```sh
|
||||
cmake \
|
||||
-DCMAKE_PREFIX_PATH="<rocm install root>" \
|
||||
-DROCM_DIR="$ROCM_INSTALL_PATH" \
|
||||
..
|
||||
```
|
||||
4. **Compile the project**
|
||||
```sh
|
||||
make
|
||||
```
|
||||
## Using the ROCR Runtime
|
||||
|
||||
After installation, you can link against the runtime by using the provided CMake package configurations. For example, to use the ROCR runtime in your project:
|
||||
|
||||
```cmake
|
||||
find_package(hsa-runtime64 1.0 REQUIRED)
|
||||
add_executable(MyApp main.cpp)
|
||||
target_link_libraries(MyApp PRIVATE hsa-runtime64::hsa-runtime64)
|
||||
```
|
||||
## Disclaimer
|
||||
The information contained herein is for informational purposes only, and is
|
||||
subject to change without notice. While every precaution has been taken in the
|
||||
preparation of this document, it may contain technical inaccuracies, omissions
|
||||
and typographical errors, and AMD is under no obligation to update or otherwise
|
||||
correct this information. Advanced Micro Devices, Inc. makes no representations
|
||||
or warranties with respect to the accuracy or completeness of the contents of
|
||||
this document, and assumes no liability of any kind, including the implied
|
||||
warranties of noninfringement, merchantability or fitness for particular
|
||||
purposes, with respect to the operation or use of AMD hardware, software or
|
||||
other products described herein. No license, including implied or arising by
|
||||
estoppel, to any intellectual property rights is granted by this document.
|
||||
Terms and limitations applicable to the purchase or use of AMD's products are
|
||||
as set forth in a signed agreement between the parties or in AMD's Standard
|
||||
Terms and Conditions of Sale.
|
||||
|
||||
AMD, the AMD Arrow logo, and combinations thereof are trademarks of Advanced
|
||||
Micro Devices, Inc. Other product names used in this publication are for
|
||||
identification purposes only and may be trademarks of their respective
|
||||
companies.
|
||||
|
||||
Copyright © 2014-2024 Advanced Micro Devices, Inc. All rights reserved.
|
||||
@@ -0,0 +1,47 @@
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2016-2021, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
# left-hand term originates from @ENABLE_LDCONFIG@ = ON/OFF at package build
|
||||
if [ "@ENABLE_LDCONFIG@" == "ON" ]; then
|
||||
echo @CPACK_PACKAGING_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@ > /etc/ld.so.conf.d/hsa-rocr.conf
|
||||
ldconfig
|
||||
fi
|
||||
@@ -0,0 +1,48 @@
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2016-2021, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
# left-hand term originates from @ENABLE_LDCONFIG@ = ON/OFF at package build
|
||||
if [ $1 -le 1 ] && [ "@ENABLE_LDCONFIG@" == "ON" ]; then
|
||||
# perform the below actions for rpm remove($1=0) or upgrade($1=1) operations
|
||||
rm -f /etc/ld.so.conf.d/hsa-rocr.conf
|
||||
ldconfig
|
||||
fi
|
||||
@@ -0,0 +1,45 @@
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2016-2021, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
# Workaround for CPACK directory symlink handling error.
|
||||
mkdir -p @CPACK_PACKAGING_INSTALL_PREFIX@/hsa/include
|
||||
ln -sf ../../@CMAKE_INSTALL_INCLUDEDIR@/hsa @CPACK_PACKAGING_INSTALL_PREFIX@/hsa/include/hsa
|
||||
@@ -0,0 +1,48 @@
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2016-2021, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and/or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and/or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
if [ $1 -le 1 ]; then
|
||||
# Workaround for CPACK directory symlink handling error.
|
||||
# Needed for uninstall and upgrade scenarios since
|
||||
# upgrade install to new folder and old folders need to be cleaned
|
||||
rm -rf @CPACK_PACKAGING_INSTALL_PREFIX@/hsa
|
||||
fi
|
||||
@@ -0,0 +1,97 @@
|
||||
# Restore old style debuginfo creation for rpm >= 4.14.
|
||||
%undefine _debugsource_packages
|
||||
%undefine _debuginfo_subpackages
|
||||
|
||||
# -*- rpm-spec -*-
|
||||
BuildRoot: %_topdir/@CPACK_PACKAGE_FILE_NAME@@CPACK_RPM_PACKAGE_COMPONENT_PART_PATH@
|
||||
Summary: @CPACK_RPM_PACKAGE_SUMMARY@
|
||||
Name: @CPACK_RPM_PACKAGE_NAME@
|
||||
Version: @CPACK_RPM_PACKAGE_VERSION@
|
||||
Release: @CPACK_RPM_PACKAGE_RELEASE@
|
||||
License: @CPACK_RPM_PACKAGE_LICENSE@
|
||||
Group: @CPACK_RPM_PACKAGE_GROUP@
|
||||
Vendor: @CPACK_RPM_PACKAGE_VENDOR@
|
||||
|
||||
# Modifications to allow recommends to be used (not implemented in cpack):
|
||||
%if "@CPACK_RPM_PACKAGE_RECOMMENDS@" != ""
|
||||
Recommends: @CPACK_RPM_PACKAGE_RECOMMENDS@
|
||||
%endif
|
||||
# End of modifications
|
||||
|
||||
@TMP_RPM_URL@
|
||||
@TMP_RPM_REQUIRES@
|
||||
@TMP_RPM_REQUIRES_PRE@
|
||||
@TMP_RPM_REQUIRES_POST@
|
||||
@TMP_RPM_REQUIRES_PREUN@
|
||||
@TMP_RPM_REQUIRES_POSTUN@
|
||||
@TMP_RPM_PROVIDES@
|
||||
@TMP_RPM_OBSOLETES@
|
||||
@TMP_RPM_CONFLICTS@
|
||||
@TMP_RPM_SUGGESTS@
|
||||
@TMP_RPM_AUTOPROV@
|
||||
@TMP_RPM_AUTOREQ@
|
||||
@TMP_RPM_AUTOREQPROV@
|
||||
@TMP_RPM_BUILDARCH@
|
||||
@TMP_RPM_PREFIXES@
|
||||
@TMP_RPM_EPOCH@
|
||||
|
||||
@TMP_RPM_DEBUGINFO@
|
||||
|
||||
%define _rpmdir %_topdir/RPMS
|
||||
%define _srcrpmdir %_topdir/SRPMS
|
||||
@FILE_NAME_DEFINE@
|
||||
%define _unpackaged_files_terminate_build 0
|
||||
@TMP_RPM_SPEC_INSTALL_POST@
|
||||
@CPACK_RPM_SPEC_MORE_DEFINE@
|
||||
@CPACK_RPM_COMPRESSION_TYPE_TMP@
|
||||
|
||||
%description
|
||||
@CPACK_RPM_PACKAGE_DESCRIPTION@
|
||||
|
||||
# This is a shortcutted spec file generated by CMake RPM generator
|
||||
# we skip _install step because CPack does that for us.
|
||||
# We do only save CPack installed tree in _prepr
|
||||
# and then restore it in build.
|
||||
%prep
|
||||
mv $RPM_BUILD_ROOT %_topdir/tmpBBroot
|
||||
|
||||
%install
|
||||
if [ -e $RPM_BUILD_ROOT ];
|
||||
then
|
||||
rm -rf $RPM_BUILD_ROOT
|
||||
fi
|
||||
mv %_topdir/tmpBBroot $RPM_BUILD_ROOT
|
||||
|
||||
@TMP_RPM_DEBUGINFO_INSTALL@
|
||||
|
||||
%clean
|
||||
|
||||
%post
|
||||
@RPM_SYMLINK_POSTINSTALL@
|
||||
@CPACK_RPM_SPEC_POSTINSTALL@
|
||||
|
||||
%posttrans
|
||||
@CPACK_RPM_SPEC_POSTTRANS@
|
||||
|
||||
%postun
|
||||
@CPACK_RPM_SPEC_POSTUNINSTALL@
|
||||
|
||||
%pre
|
||||
@CPACK_RPM_SPEC_PREINSTALL@
|
||||
|
||||
%pretrans
|
||||
@CPACK_RPM_SPEC_PRETRANS@
|
||||
|
||||
%preun
|
||||
@CPACK_RPM_SPEC_PREUNINSTALL@
|
||||
|
||||
%files
|
||||
%defattr(@TMP_DEFAULT_FILE_PERMISSIONS@,@TMP_DEFAULT_USER@,@TMP_DEFAULT_GROUP@,@TMP_DEFAULT_DIR_PERMISSIONS@)
|
||||
@CPACK_RPM_INSTALL_FILES@
|
||||
@CPACK_RPM_ABSOLUTE_INSTALL_FILES@
|
||||
@CPACK_RPM_USER_INSTALL_FILES@
|
||||
|
||||
%changelog
|
||||
@CPACK_RPM_SPEC_CHANGELOG@
|
||||
|
||||
@TMP_OTHER_COMPONENTS@
|
||||
@@ -0,0 +1,23 @@
|
||||
#!/bin/bash
|
||||
|
||||
echo "Pre-install check for ROCr."
|
||||
|
||||
# Check for old installations...
|
||||
if ls /usr/lib/libhsa-runtime* 1> /dev/null 2>&1; then
|
||||
echo "An old version of libhsa-runtime was found in /usr/lib."
|
||||
echo "This must be uninstalled before proceeding with the installation"
|
||||
echo "to avoid potential incompatibilities."
|
||||
|
||||
read -r -p "Do you want to uninstall the old version? [y/N] " response
|
||||
if [ "$response" = "y" ]; then
|
||||
if ! rm -rf /usr/lib/libhsa-runtime*; then
|
||||
echo "Failed to remove /usr/lib/libhsa-runtime* files."
|
||||
echo "Try to uninstall these files manually."
|
||||
exit 1
|
||||
fi
|
||||
echo "Old version uninstalled."
|
||||
else
|
||||
echo "The old and new versions of ROCm are incompatible. Installation aborted."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
@@ -0,0 +1,60 @@
|
||||
---
|
||||
Language: Cpp
|
||||
# BasedOnStyle: Google
|
||||
AccessModifierOffset: -1
|
||||
ConstructorInitializerIndentWidth: 4
|
||||
AlignEscapedNewlinesLeft: false
|
||||
AlignTrailingComments: true
|
||||
AlignConsecutiveAssignments: false
|
||||
AlignOperands: false
|
||||
AllowAllParametersOfDeclarationOnNextLine: true
|
||||
AllowShortBlocksOnASingleLine: false
|
||||
AllowShortIfStatementsOnASingleLine: true
|
||||
AllowShortLoopsOnASingleLine: true
|
||||
AllowShortFunctionsOnASingleLine: All
|
||||
AlwaysBreakAfterDefinitionReturnType: false
|
||||
AlwaysBreakTemplateDeclarations: false
|
||||
AlwaysBreakBeforeMultilineStrings: true
|
||||
BreakBeforeBinaryOperators: false
|
||||
BreakBeforeTernaryOperators: true
|
||||
BreakConstructorInitializersBeforeComma: false
|
||||
BinPackParameters: true
|
||||
ColumnLimit: 100
|
||||
ConstructorInitializerAllOnOneLineOrOnePerLine: true
|
||||
ExperimentalAutoDetectBinPacking: false
|
||||
IndentCaseLabels: true
|
||||
IndentWrappedFunctionNames: false
|
||||
IndentFunctionDeclarationAfterType: false
|
||||
MaxEmptyLinesToKeep: 2
|
||||
KeepEmptyLinesAtTheStartOfBlocks: false
|
||||
NamespaceIndentation: None
|
||||
ObjCSpaceAfterProperty: false
|
||||
ObjCSpaceBeforeProtocolList: false
|
||||
PenaltyBreakBeforeFirstCallParameter: 1
|
||||
PenaltyBreakComment: 300
|
||||
PenaltyBreakString: 1000
|
||||
PenaltyBreakFirstLessLess: 120
|
||||
PenaltyExcessCharacter: 1000000
|
||||
PenaltyReturnTypeOnItsOwnLine: 200
|
||||
DerivePointerAlignment: false
|
||||
PointerAlignment: Left
|
||||
SpacesBeforeTrailingComments: 2
|
||||
Cpp11BracedListStyle: true
|
||||
Standard: Auto
|
||||
IndentWidth: 2
|
||||
TabWidth: 8
|
||||
UseTab: Never
|
||||
BreakBeforeBraces: Attach
|
||||
SpacesInParentheses: false
|
||||
SpacesInAngles: false
|
||||
SpaceInEmptyParentheses: false
|
||||
SpacesInCStyleCastParentheses: false
|
||||
SpacesInContainerLiterals: true
|
||||
SpaceBeforeAssignmentOperators: true
|
||||
ContinuationIndentWidth: 4
|
||||
CommentPragmas: '^ IWYU pragma:'
|
||||
ForEachMacros: [ foreach, Q_FOREACH, BOOST_FOREACH ]
|
||||
SpaceBeforeParens: ControlStatements
|
||||
DisableFormat: false
|
||||
SortIncludes: false
|
||||
...
|
||||
Executable
+135
@@ -0,0 +1,135 @@
|
||||
#!/usr/bin/env python3
|
||||
#
|
||||
#===- clang-format-diff.py - ClangFormat Diff Reformatter ----*- python -*--===#
|
||||
#
|
||||
# Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
# See https://llvm.org/LICENSE.txt for license information.
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
#
|
||||
#===------------------------------------------------------------------------===#
|
||||
|
||||
"""
|
||||
This script reads input from a unified diff and reformats all the changed
|
||||
lines. This is useful to reformat all the lines touched by a specific patch.
|
||||
Example usage for git/svn users:
|
||||
|
||||
git diff -U0 --no-color --relative HEAD^ | clang-format-diff.py -p1 -i
|
||||
svn diff --diff-cmd=diff -x-U0 | clang-format-diff.py -i
|
||||
|
||||
It should be noted that the filename contained in the diff is used unmodified
|
||||
to determine the source file to update. Users calling this script directly
|
||||
should be careful to ensure that the path in the diff is correct relative to the
|
||||
current working directory.
|
||||
"""
|
||||
from __future__ import absolute_import, division, print_function
|
||||
|
||||
import argparse
|
||||
import difflib
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
if sys.version_info.major >= 3:
|
||||
from io import StringIO
|
||||
else:
|
||||
from io import BytesIO as StringIO
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__,
|
||||
formatter_class=
|
||||
argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument('-i', action='store_true', default=False,
|
||||
help='apply edits to files instead of displaying a diff')
|
||||
parser.add_argument('-p', metavar='NUM', default=0,
|
||||
help='strip the smallest prefix containing P slashes')
|
||||
parser.add_argument('-regex', metavar='PATTERN', default=None,
|
||||
help='custom pattern selecting file paths to reformat '
|
||||
'(case sensitive, overrides -iregex)')
|
||||
parser.add_argument('-iregex', metavar='PATTERN', default=
|
||||
r'.*\.(cpp|cc|c\+\+|cxx|c|cl|h|hh|hpp|hxx|m|mm|inc|js|ts'
|
||||
r'|proto|protodevel|java|cs)',
|
||||
help='custom pattern selecting file paths to reformat '
|
||||
'(case insensitive, overridden by -regex)')
|
||||
parser.add_argument('-sort-includes', action='store_true', default=False,
|
||||
help='let clang-format sort include blocks')
|
||||
parser.add_argument('-v', '--verbose', action='store_true',
|
||||
help='be more verbose, ineffective without -i')
|
||||
parser.add_argument('-style',
|
||||
help='formatting style to apply (LLVM, GNU, Google, Chromium, '
|
||||
'Microsoft, Mozilla, WebKit)')
|
||||
parser.add_argument('-binary', default='clang-format',
|
||||
help='location of binary to use for clang-format')
|
||||
args = parser.parse_args()
|
||||
|
||||
# Extract changed lines for each file.
|
||||
filename = None
|
||||
lines_by_file = {}
|
||||
for line in sys.stdin:
|
||||
match = re.search(r'^\+\+\+\ (.*?/){%s}(\S*)' % args.p, line)
|
||||
if match:
|
||||
filename = match.group(2)
|
||||
if filename is None:
|
||||
continue
|
||||
|
||||
if args.regex is not None:
|
||||
if not re.match('^%s$' % args.regex, filename):
|
||||
continue
|
||||
else:
|
||||
if not re.match('^%s$' % args.iregex, filename, re.IGNORECASE):
|
||||
continue
|
||||
|
||||
match = re.search(r'^@@.*\+(\d+)(,(\d+))?', line)
|
||||
if match:
|
||||
start_line = int(match.group(1))
|
||||
line_count = 1
|
||||
if match.group(3):
|
||||
line_count = int(match.group(3))
|
||||
if line_count == 0:
|
||||
continue
|
||||
end_line = start_line + line_count - 1
|
||||
lines_by_file.setdefault(filename, []).extend(
|
||||
['-lines', str(start_line) + ':' + str(end_line)])
|
||||
|
||||
# Reformat files containing changes in place.
|
||||
for filename, lines in lines_by_file.items():
|
||||
if args.i and args.verbose:
|
||||
print('Formatting {}'.format(filename))
|
||||
command = [args.binary, filename]
|
||||
if args.i:
|
||||
command.append('-i')
|
||||
if args.sort_includes:
|
||||
command.append('-sort-includes')
|
||||
command.extend(lines)
|
||||
if args.style:
|
||||
command.extend(['-style', args.style])
|
||||
|
||||
try:
|
||||
p = subprocess.Popen(command,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=None,
|
||||
stdin=subprocess.PIPE,
|
||||
universal_newlines=True)
|
||||
except OSError as e:
|
||||
# Give the user more context when clang-format isn't
|
||||
# found/isn't executable, etc.
|
||||
raise RuntimeError(
|
||||
'Failed to run "%s" - %s"' % (" ".join(command), e.strerror))
|
||||
|
||||
stdout, stderr = p.communicate()
|
||||
if p.returncode != 0:
|
||||
sys.exit(p.returncode)
|
||||
|
||||
if not args.i:
|
||||
with open(filename) as f:
|
||||
code = f.readlines()
|
||||
formatted_code = StringIO(stdout).readlines()
|
||||
diff = difflib.unified_diff(code, formatted_code,
|
||||
filename, filename,
|
||||
'(before formatting)', '(after formatting)')
|
||||
diff_string = ''.join(diff)
|
||||
if len(diff_string) > 0:
|
||||
sys.stdout.write(diff_string)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,233 @@
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2014-2017, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and#or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and#or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
function( get_path LIB CACHED_PATH HELP )
|
||||
|
||||
set( options "")
|
||||
set( oneValueArgs RESULT )
|
||||
set( multiValueArgs HINTS NAMES )
|
||||
cmake_parse_arguments(ARGS "${options}" "${oneValueArgs}" "${multiValueArgs}" ${ARGN} )
|
||||
|
||||
# Search for canary file.
|
||||
if( ${LIB} )
|
||||
find_library( FULLPATH NAMES ${ARGS_NAMES} HINTS ${${CACHED_PATH}} ${ARGS_HINTS} )
|
||||
else()
|
||||
find_file( FULLPATH NAMES ${ARGS_NAMES} HINTS ${${CACHED_PATH}} ${ARGS_HINTS} )
|
||||
endif()
|
||||
set( RESULT (NOT ${FULLPATH} MATCHES NOTFOUND) )
|
||||
|
||||
# Extract path
|
||||
get_filename_component ( DIRPATH ${FULLPATH} DIRECTORY )
|
||||
|
||||
# Check path against cache
|
||||
if( NOT "${${CACHED_PATH}}" STREQUAL "" )
|
||||
if ( NOT "${${CACHED_PATH}}" STREQUAL "${DIRPATH}" )
|
||||
message(WARNING "${CACHED_PATH} may be incorrect." )
|
||||
set( DIRPATH ${${CACHED_PATH}} )
|
||||
endif()
|
||||
elseif(NOT ${RESULT})
|
||||
message(WARNING "${CACHED_PATH} not located during path search.")
|
||||
endif()
|
||||
|
||||
# Set cache variable and help text
|
||||
set( ${CACHED_PATH} ${DIRPATH} CACHE PATH ${HELP} FORCE )
|
||||
unset( FULLPATH CACHE )
|
||||
|
||||
# Return success flag
|
||||
if( NOT ${ARGS_RESULT} STREQUAL "" )
|
||||
set( ${ARGS_RESULT} ${RESULT} PARENT_SCOPE)
|
||||
endif()
|
||||
|
||||
endfunction()
|
||||
|
||||
## Searches for a file using include paths and stores the path to that file in the cache
|
||||
## using the cached value if set. Search paths are optional. Returns success in RESULT.
|
||||
## get_include_path(<VAR> NAMES name1 [name2...] [HINTS path1 [path2 ... ENV var]] [RESULT <var>]
|
||||
macro( get_include_path CACHED_PATH HELP )
|
||||
get_path( 0 ${ARGV} )
|
||||
endmacro()
|
||||
|
||||
## Searches for a file using library paths and stores the path to that file in the cache
|
||||
## using the cached value if set. Search paths are optional. Returns success in RESULT.
|
||||
## get_library_path(<VAR> NAMES name1 [name2...] [HINTS path1 [path2 ... ENV var]] [RESULT <var>]
|
||||
macro( get_library_path CACHED_PATH HELP )
|
||||
get_path( 1 ${ARGV} )
|
||||
endmacro()
|
||||
|
||||
## Parses the VERSION_STRING variable and places
|
||||
## the first, second and third number values in
|
||||
## the major, minor and patch variables.
|
||||
function( parse_version VERSION_STRING )
|
||||
|
||||
string ( FIND ${VERSION_STRING} "-" STRING_INDEX )
|
||||
|
||||
if ( ${STRING_INDEX} GREATER -1 )
|
||||
math ( EXPR STRING_INDEX "${STRING_INDEX} + 1" )
|
||||
string ( SUBSTRING ${VERSION_STRING} ${STRING_INDEX} -1 VERSION_BUILD )
|
||||
endif ()
|
||||
|
||||
string ( REGEX MATCHALL "[0123456789]+" VERSIONS ${VERSION_STRING} )
|
||||
list ( LENGTH VERSIONS VERSION_COUNT )
|
||||
|
||||
if ( ${VERSION_COUNT} GREATER 0)
|
||||
list ( GET VERSIONS 0 MAJOR )
|
||||
set ( VERSION_MAJOR ${MAJOR} PARENT_SCOPE )
|
||||
endif ()
|
||||
|
||||
if ( ${VERSION_COUNT} GREATER 1 )
|
||||
list ( GET VERSIONS 1 MINOR )
|
||||
set ( VERSION_MINOR ${MINOR} PARENT_SCOPE )
|
||||
endif ()
|
||||
|
||||
if ( ${VERSION_COUNT} GREATER 2 )
|
||||
list ( GET VERSIONS 2 PATCH )
|
||||
set ( VERSION_PATCH ${PATCH} PARENT_SCOPE )
|
||||
endif ()
|
||||
|
||||
endfunction ()
|
||||
|
||||
## Gets the current version of the repository
|
||||
## using versioning tags and git describe.
|
||||
## Passes back a packaging version string
|
||||
## and a library version string.
|
||||
function ( get_version DEFAULT_VERSION_STRING )
|
||||
|
||||
set( VERSION_JOB "local-build" )
|
||||
set( VERSION_COMMIT_COUNT 0 )
|
||||
set( VERSION_HASH "unknown" )
|
||||
|
||||
find_program( GIT NAMES git )
|
||||
|
||||
if( GIT )
|
||||
|
||||
#execute_process ( COMMAND git describe --tags --dirty --long
|
||||
# WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}
|
||||
# OUTPUT_VARIABLE GIT_TAG_STRING
|
||||
# OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
# RESULT_VARIABLE RESULT )
|
||||
|
||||
# Get branch commit (common ancestor) of current branch and master branch.
|
||||
execute_process(COMMAND git merge-base HEAD origin/HEAD
|
||||
WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}
|
||||
OUTPUT_VARIABLE GIT_MERGE_BASE
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
RESULT_VARIABLE RESULT )
|
||||
|
||||
if( ${RESULT} EQUAL 0 )
|
||||
# Count commits from branch point.
|
||||
execute_process(COMMAND git rev-list --count ${GIT_MERGE_BASE}..HEAD
|
||||
WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}
|
||||
OUTPUT_VARIABLE VERSION_COMMIT_COUNT
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
RESULT_VARIABLE RESULT )
|
||||
if(NOT ${RESULT} EQUAL 0 )
|
||||
set( VERSION_COMMIT_COUNT 0 )
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Get current short hash.
|
||||
execute_process(COMMAND git rev-parse --short HEAD
|
||||
WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}
|
||||
OUTPUT_VARIABLE VERSION_HASH
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
RESULT_VARIABLE RESULT )
|
||||
if( ${RESULT} EQUAL 0 )
|
||||
# Check for dirty workspace.
|
||||
execute_process(COMMAND git diff --quiet
|
||||
WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}
|
||||
RESULT_VARIABLE RESULT )
|
||||
if(${RESULT} EQUAL 1)
|
||||
set(VERSION_HASH "${VERSION_HASH}-dirty")
|
||||
endif()
|
||||
else()
|
||||
set( VERSION_HASH "unknown" )
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Build automation IDs
|
||||
if(DEFINED ENV{ROCM_BUILD_ID})
|
||||
set( VERSION_JOB $ENV{ROCM_BUILD_ID} )
|
||||
endif()
|
||||
|
||||
parse_version(${DEFAULT_VERSION_STRING})
|
||||
|
||||
set( VERSION_MAJOR "${VERSION_MAJOR}" PARENT_SCOPE )
|
||||
set( VERSION_MINOR "${VERSION_MINOR}" PARENT_SCOPE )
|
||||
set( VERSION_PATCH "${VERSION_PATCH}" PARENT_SCOPE )
|
||||
set( VERSION_COMMIT_COUNT "${VERSION_COMMIT_COUNT}" PARENT_SCOPE )
|
||||
set( VERSION_HASH "${VERSION_HASH}" PARENT_SCOPE )
|
||||
set( VERSION_JOB "${VERSION_JOB}" PARENT_SCOPE )
|
||||
|
||||
#message("${VERSION_MAJOR}" )
|
||||
#message("${VERSION_MINOR}" )
|
||||
#message("${VERSION_PATCH}" )
|
||||
#message("${VERSION_COMMIT_COUNT}")
|
||||
#message("${VERSION_HASH}")
|
||||
#message("${VERSION_JOB}")
|
||||
|
||||
endfunction()
|
||||
|
||||
## Collects subdirectory names and returns them in a list
|
||||
function ( listsubdirs DIRPATH SUBDIRECTORIES )
|
||||
file( GLOB CONTENTS RELATIVE ${DIRPATH} "${DIRPATH}/*" )
|
||||
set ( FOLDERS, "" )
|
||||
foreach( ITEM IN LISTS CONTENTS)
|
||||
if( IS_DIRECTORY "${DIRPATH}/${ITEM}" )
|
||||
list( APPEND FOLDERS ${ITEM} )
|
||||
endif()
|
||||
endforeach()
|
||||
set (${SUBDIRECTORIES} ${FOLDERS} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
## Sets el7 flag to be true
|
||||
function (Checksetel7 EL7_DISTRO)
|
||||
execute_process(COMMAND rpm --eval %{?dist}
|
||||
RESULT_VARIABLE PROC_RESULT
|
||||
OUTPUT_VARIABLE EVAL_RESULT
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE)
|
||||
message("RESULT_VARIABLE ${PROC_RESULT} OUTPUT_VARIABLE: ${EVAL_RESULT}")
|
||||
if (PROC_RESULT EQUAL "0" AND NOT EVAL_RESULT STREQUAL "")
|
||||
if ("${EVAL_RESULT}" STREQUAL ".el7")
|
||||
set (${EL7_DISTRO} TRUE PARENT_SCOPE)
|
||||
endif()
|
||||
endif()
|
||||
endfunction()
|
||||
@@ -0,0 +1,6 @@
|
||||
#!/bin/bash
|
||||
root=`git rev-parse --show-toplevel`
|
||||
pushd . > /dev/null
|
||||
cd $root
|
||||
git diff -U0 HEAD^ | ./clang-format-diff.py -p1 -i -style=file
|
||||
popd > /dev/null
|
||||
@@ -0,0 +1,319 @@
|
||||
################################################################################
|
||||
##
|
||||
## Copyright (c) 2016 Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## MIT LICENSE:
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
## this software and associated documentation files (the "Software"), to deal in
|
||||
## the Software without restriction, including without limitation the rights to
|
||||
## use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
## of the Software, and to permit persons to whom the Software is furnished to do
|
||||
## so, subject to the following conditions:
|
||||
##
|
||||
## The above copyright notice and this permission notice shall be included in all
|
||||
## copies or substantial portions of the Software.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
## AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
## LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
## OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
## SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
cmake_minimum_required ( VERSION 3.6.3 )
|
||||
|
||||
set(CMAKE_VERBOSE_MAKEFILE ON)
|
||||
|
||||
set ( HSAKMT "hsakmt" )
|
||||
set ( HSAKMT_PACKAGE "hsakmt-roct" )
|
||||
set ( HSAKMT_COMPONENT "lib${HSAKMT}" )
|
||||
set ( HSAKMT_TARGET "${HSAKMT}" )
|
||||
set(HSAKMT_STATIC_DRM_TARGET "${HSAKMT_TARGET}-staticdrm")
|
||||
|
||||
project ( ${HSAKMT_TARGET} VERSION 1.9.0)
|
||||
|
||||
# Optionally, build HSAKMT with ccache.
|
||||
set(ROCM_CCACHE_BUILD OFF CACHE BOOL "Set to ON for a ccache enabled build")
|
||||
if (ROCM_CCACHE_BUILD)
|
||||
find_program(CCACHE_PROGRAM ccache)
|
||||
if (CCACHE_PROGRAM)
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE ${CCACHE_PROGRAM})
|
||||
else()
|
||||
message(WARNING "Unable to find ccache. Falling back to real compiler")
|
||||
endif() # if (CCACHE_PROGRAM)
|
||||
endif() # if (ROCM_CCACHE_BUILD)
|
||||
|
||||
list( PREPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake_modules" )
|
||||
|
||||
## Include common cmake modules
|
||||
include ( utils )
|
||||
include ( GNUInstallDirs )
|
||||
|
||||
## Setup the package version.
|
||||
get_version ( "1.0.0" )
|
||||
|
||||
set ( BUILD_VERSION_MAJOR ${VERSION_MAJOR} )
|
||||
set ( BUILD_VERSION_MINOR ${VERSION_MINOR} )
|
||||
set ( BUILD_VERSION_PATCH ${VERSION_PATCH} )
|
||||
|
||||
set ( LIB_VERSION_MAJOR 1)
|
||||
set ( LIB_VERSION_MINOR 0)
|
||||
if (${ROCM_PATCH_VERSION})
|
||||
set ( LIB_VERSION_PATCH ${ROCM_PATCH_VERSION} )
|
||||
else ()
|
||||
set ( LIB_VERSION_PATCH 6)
|
||||
endif ()
|
||||
set ( LIB_VERSION_STRING "${LIB_VERSION_MAJOR}.${LIB_VERSION_MINOR}.${LIB_VERSION_PATCH}" )
|
||||
|
||||
if ( DEFINED VERSION_BUILD AND NOT ${VERSION_BUILD} STREQUAL "" )
|
||||
message ( "VERSION BUILD DEFINED ${VERSION_BUILD}" )
|
||||
set ( BUILD_VERSION_PATCH "${BUILD_VERSION_PATCH}-${VERSION_BUILD}" )
|
||||
endif ()
|
||||
set ( BUILD_VERSION_STRING "${BUILD_VERSION_MAJOR}.${BUILD_VERSION_MINOR}.${BUILD_VERSION_PATCH}" )
|
||||
|
||||
## Compiler flags
|
||||
set (HSAKMT_C_FLAGS -fPIC -W -Wall -Wextra -Wno-unused-parameter -Wformat-security -Wswitch-default -Wundef -Wshadow -Wpointer-arith -Wbad-function-cast -Wcast-qual -Wstrict-prototypes -Wmissing-prototypes -Wmissing-declarations -Wredundant-decls -Wunreachable-code -std=gnu99 -fvisibility=hidden)
|
||||
if ( CMAKE_COMPILER_IS_GNUCC )
|
||||
set ( HSAKMT_C_FLAGS "${HSAKMT_C_FLAGS}" -Wlogical-op)
|
||||
endif ()
|
||||
if ( ${HSAKMT_WERROR} )
|
||||
set ( HSAKMT_C_FLAGS "${HSAKMT_C_FLAGS}" -Werror )
|
||||
endif ()
|
||||
if ( "${CMAKE_BUILD_TYPE}" STREQUAL Release )
|
||||
set ( HSAKMT_C_FLAGS "${HSAKMT_C_FLAGS}" -O2 )
|
||||
else ()
|
||||
set ( HSAKMT_C_FLAGS "${HSAKMT_C_FLAGS}" -g )
|
||||
endif ()
|
||||
|
||||
set ( HSAKMT_LINKER_SCRIPT "${CMAKE_CURRENT_SOURCE_DIR}/src/libhsakmt.ver" )
|
||||
|
||||
## Linker Flags
|
||||
## Add --enable-new-dtags to generate DT_RUNPATH
|
||||
set (HSAKMT_LINK_FLAGS "${HSAKMT_LINK_FLAGS} -Wl,--enable-new-dtags -Wl,--version-script=${HSAKMT_LINKER_SCRIPT} -Wl,-soname=${HSAKMT_COMPONENT}.so.${LIB_VERSION_MAJOR} -Wl,-z,nodelete")
|
||||
|
||||
## Address Sanitize Flag
|
||||
if ( ${ADDRESS_SANITIZER} )
|
||||
set ( HSAKMT_C_FLAGS "${HSAKMT_C_FLAGS}" -fsanitize=address )
|
||||
set ( HSAKMT_LINK_FLAGS "${HSAKMT_LINK_FLAGS} -fsanitize=address" )
|
||||
if ( BUILD_SHARED_LIBS )
|
||||
set ( HSAKMT_LINK_FLAGS "${HSAKMT_LINK_FLAGS} -shared-libsan" )
|
||||
else ()
|
||||
set ( HSAKMT_LINK_FLAGS "${HSAKMT_LINK_FLAGS} -static-libsan" )
|
||||
endif ()
|
||||
else ()
|
||||
if ( CMAKE_COMPILER_IS_GNUCC )
|
||||
set ( HSAKMT_LINK_FLAGS "${HSAKMT_LINK_FLAGS} -Wl,-no-undefined" )
|
||||
else ()
|
||||
set ( HSAKMT_LINK_FLAGS "${HSAKMT_LINK_FLAGS} -Wl,-undefined,error" )
|
||||
endif ()
|
||||
endif ()
|
||||
|
||||
## Source files
|
||||
set ( HSAKMT_SRC "src/debug.c"
|
||||
"src/events.c"
|
||||
"src/fmm.c"
|
||||
"src/globals.c"
|
||||
"src/hsakmtmodel.c"
|
||||
"src/libhsakmt.c"
|
||||
"src/memory.c"
|
||||
"src/openclose.c"
|
||||
"src/perfctr.c"
|
||||
"src/pmc_table.c"
|
||||
"src/queues.c"
|
||||
"src/time.c"
|
||||
"src/topology.c"
|
||||
"src/rbtree.c"
|
||||
"src/spm.c"
|
||||
"src/version.c"
|
||||
"src/svm.c"
|
||||
"src/pc_sampling.c")
|
||||
|
||||
## Declare the library target name
|
||||
add_library (${HSAKMT_TARGET} STATIC "")
|
||||
|
||||
## Add sources
|
||||
target_sources ( ${HSAKMT_TARGET} PRIVATE ${HSAKMT_SRC} )
|
||||
|
||||
## Add headers. The public headers need to point at their location in both build and install
|
||||
## directory layouts. This declaration allows publishing library use data to downstream clients.
|
||||
target_include_directories( ${HSAKMT_TARGET}
|
||||
PUBLIC
|
||||
$<BUILD_INTERFACE:${CMAKE_CURRENT_SOURCE_DIR}/include>
|
||||
$<INSTALL_INTERFACE:${CMAKE_INSTALL_INCLUDEDIR}>
|
||||
PRIVATE
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/src )
|
||||
|
||||
set_property(TARGET ${HSAKMT_TARGET} PROPERTY LINK_FLAGS ${HSAKMT_LINK_FLAGS})
|
||||
|
||||
## Set the VERSION and SOVERSION values
|
||||
set_property ( TARGET ${HSAKMT_TARGET} PROPERTY VERSION "${LIB_VERSION_STRING}" )
|
||||
set_property ( TARGET ${HSAKMT_TARGET} PROPERTY SOVERSION "${LIB_VERSION_MAJOR}" )
|
||||
|
||||
find_package(PkgConfig)
|
||||
# get OS-info for OS-specific build dependencies
|
||||
get_os_info()
|
||||
|
||||
find_package(PkgConfig)
|
||||
# Check for libraries required for building
|
||||
find_library(LIBC NAMES c REQUIRED)
|
||||
find_package(NUMA)
|
||||
if(NUMA_FOUND)
|
||||
set(NUMA "${NUMA_LIBRARIES}")
|
||||
else()
|
||||
find_library(NUMA NAMES numa REQUIRED)
|
||||
endif()
|
||||
message(STATUS "LIBC: " ${LIBC})
|
||||
message(STATUS "NUMA: " ${NUMA})
|
||||
|
||||
## If environment variable DRM_DIR is set, the script
|
||||
## will pick up the corresponding libraries from that path.
|
||||
if(DRM_DIR)
|
||||
list (PREPEND CMAKE_PREFIX_PATH "${DRM_DIR}")
|
||||
endif()
|
||||
|
||||
# The module name passed to pkg_check_modules() is determined by the
|
||||
# name of file *.pc
|
||||
pkg_check_modules(DRM REQUIRED IMPORTED_TARGET libdrm)
|
||||
pkg_check_modules(DRM_AMDGPU REQUIRED IMPORTED_TARGET libdrm_amdgpu)
|
||||
include_directories(${DRM_AMDGPU_INCLUDE_DIRS})
|
||||
include_directories(${DRM_INCLUDE_DIRS})
|
||||
|
||||
target_link_libraries ( ${HSAKMT_TARGET}
|
||||
PRIVATE ${DRM_LDFLAGS} ${DRM_AMDGPU_LDFLAGS} pthread rt ${LIBC} ${NUMA} ${CMAKE_DL_LIBS}
|
||||
)
|
||||
|
||||
target_compile_options(${HSAKMT_TARGET} PRIVATE ${DRM_CFLAGS} ${HSAKMT_C_FLAGS})
|
||||
|
||||
include(CheckFunctionExists)
|
||||
set(CMAKE_REQUIRED_DEFINITIONS -D__USE_GNU=1)
|
||||
set(CMAKE_REQUIRED_INCLUDES sys/mman.h)
|
||||
check_function_exists(memfd_create HAVE_MEMFD_CREATE)
|
||||
if(HAVE_MEMFD_CREATE)
|
||||
target_compile_definitions(${HSAKMT_TARGET} PRIVATE -DHAVE_MEMFD_CREATE=1)
|
||||
endif()
|
||||
|
||||
## Define default paths and packages.
|
||||
if( CMAKE_INSTALL_PREFIX_INITIALIZED_TO_DEFAULT )
|
||||
set ( CMAKE_INSTALL_PREFIX "/opt/rocm" )
|
||||
endif()
|
||||
set ( CMAKE_INSTALL_PREFIX ${CMAKE_INSTALL_PREFIX} CACHE STRING "Default installation directory." FORCE )
|
||||
|
||||
# Installs binaries and exports the library usage data to ${HSAKMT_TARGET}Targets
|
||||
install ( TARGETS ${HSAKMT_TARGET} EXPORT ${HSAKMT_TARGET}Targets
|
||||
ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR} COMPONENT asan
|
||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} COMPONENT asan )
|
||||
install ( TARGETS ${HSAKMT_TARGET}
|
||||
ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR} COMPONENT binary
|
||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} COMPONENT binary )
|
||||
|
||||
# Install public headers
|
||||
install ( DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/include/${HSAKMT_TARGET} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}
|
||||
COMPONENT dev PATTERN "linux" EXCLUDE )
|
||||
|
||||
# Record our usage data for clients find_package calls.
|
||||
install ( EXPORT ${HSAKMT_TARGET}Targets
|
||||
FILE ${HSAKMT_TARGET}Targets.cmake
|
||||
NAMESPACE ${HSAKMT_TARGET}::
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${HSAKMT_TARGET}
|
||||
COMPONENT dev)
|
||||
|
||||
# Adds the target alias hsakmt::hsakmt to the local cmake cache.
|
||||
# This isn't necessary today. It's harmless preparation for some
|
||||
# hypothetical future in which the we might be included by add_subdirectory()
|
||||
# in some other project's cmake file. It allows uniform use of find_package
|
||||
# and target_link_library() without regard to whether a target is external or
|
||||
# a subdirectory of the current build.
|
||||
add_library( ${HSAKMT_TARGET}::${HSAKMT_TARGET} ALIAS ${HSAKMT_TARGET} )
|
||||
|
||||
# Create cmake configuration files
|
||||
include(CMakePackageConfigHelpers)
|
||||
|
||||
configure_package_config_file(${HSAKMT_TARGET}-config.cmake.in
|
||||
${HSAKMT_TARGET}-config.cmake
|
||||
INSTALL_DESTINATION
|
||||
${CMAKE_INSTALL_LIBDIR}/cmake/${HSAKMT_TARGET} )
|
||||
|
||||
write_basic_package_version_file(${HSAKMT_TARGET}-config-version.cmake
|
||||
VERSION ${BUILD_VERSION_STRING}
|
||||
COMPATIBILITY
|
||||
AnyNewerVersion)
|
||||
|
||||
install(FILES
|
||||
${CMAKE_CURRENT_BINARY_DIR}/${HSAKMT_TARGET}-config.cmake
|
||||
${CMAKE_CURRENT_BINARY_DIR}/${HSAKMT_TARGET}-config-version.cmake
|
||||
DESTINATION
|
||||
${CMAKE_INSTALL_LIBDIR}/cmake/${HSAKMT_TARGET}
|
||||
COMPONENT dev)
|
||||
|
||||
# Optionally record the package's find module in the user's package cache.
|
||||
if ( NOT DEFINED EXPORT_TO_USER_PACKAGE_REGISTRY )
|
||||
set ( EXPORT_TO_USER_PACKAGE_REGISTRY "off" )
|
||||
endif()
|
||||
set ( EXPORT_TO_USER_PACKAGE_REGISTRY ${EXPORT_TO_USER_PACKAGE_REGISTRY}
|
||||
CACHE BOOL "Add cmake package config location to the user's cmake package registry.")
|
||||
if(${EXPORT_TO_USER_PACKAGE_REGISTRY})
|
||||
# Enable writing to the registry
|
||||
set(CMAKE_EXPORT_PACKAGE_REGISTRY ON)
|
||||
# Generate a target file for the build
|
||||
export(TARGETS ${HSAKMT_TARGET} NAMESPACE ${HSAKMT_TARGET}:: FILE ${HSAKMT_TARGET}Targets.cmake)
|
||||
# Record the package in the user's cache.
|
||||
export(PACKAGE ${HSAKMT_TARGET})
|
||||
endif()
|
||||
|
||||
# CPACK_PACKAGING_INSTALL_PREFIX is needed in libhsakmt.pc.in
|
||||
# TODO: Add support for relocatable packages.
|
||||
configure_file ( libhsakmt.pc.in libhsakmt.pc @ONLY )
|
||||
|
||||
install ( FILES ${CMAKE_CURRENT_BINARY_DIR}/libhsakmt.pc DESTINATION ${CMAKE_INSTALL_LIBDIR}/pkgconfig COMPONENT dev)
|
||||
|
||||
if ( NOT BUILD_SHARED_LIBS)
|
||||
## Create separate target file for static builds
|
||||
## In static builds, libdrm and libdrm_amdgpu need to be linked statically
|
||||
add_library (${HSAKMT_STATIC_DRM_TARGET} STATIC "")
|
||||
target_sources (${HSAKMT_STATIC_DRM_TARGET} PRIVATE ${HSAKMT_SRC})
|
||||
|
||||
target_include_directories( ${HSAKMT_STATIC_DRM_TARGET}
|
||||
PUBLIC
|
||||
$<BUILD_INTERFACE:${CMAKE_CURRENT_SOURCE_DIR}/include>
|
||||
$<INSTALL_INTERFACE:${CMAKE_INSTALL_INCLUDEDIR}>
|
||||
PRIVATE
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/src )
|
||||
|
||||
## Set the VERSION and SOVERSION values
|
||||
set_property(TARGET ${HSAKMT_STATIC_DRM_TARGET} PROPERTY LINK_FLAGS ${HSAKMT_LINK_FLAGS}
|
||||
PROPERTY VERSION "${LIB_VERSION_STRING}"
|
||||
PROPERTY SOVERSION "${LIB_VERSION_MAJOR}" )
|
||||
|
||||
#Additional search path for static libraries
|
||||
if(${DISTRO_ID} MATCHES "ubuntu")
|
||||
set(AMDGPU_STATIC_LIB_PATHS "-L/opt/amdgpu/lib/x86_64-linux-gnu")
|
||||
else()
|
||||
set(AMDGPU_STATIC_LIB_PATHS "-L/opt/amdgpu/lib64" "-L/opt/amdgpu/lib")
|
||||
endif()
|
||||
# Link drm_amdgpu and drm library statically
|
||||
target_link_libraries ( ${HSAKMT_STATIC_DRM_TARGET}
|
||||
PRIVATE pthread rt c numa ${CMAKE_DL_LIBS}
|
||||
INTERFACE -Wl,-Bstatic ${AMDGPU_STATIC_LIB_PATHS} ${DRM_AMDGPU_LDFLAGS} ${DRM_LDFLAGS} -Wl,-Bdynamic
|
||||
)
|
||||
target_compile_options(${HSAKMT_STATIC_DRM_TARGET} PRIVATE ${DRM_CFLAGS} ${HSAKMT_C_FLAGS})
|
||||
|
||||
install ( TARGETS ${HSAKMT_STATIC_DRM_TARGET} EXPORT ${HSAKMT_STATIC_DRM_TARGET}Targets
|
||||
ARCHIVE DESTINATION ${CMAKE_INSTALL_LIBDIR} COMPONENT binary
|
||||
LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} COMPONENT binary)
|
||||
install ( EXPORT ${HSAKMT_STATIC_DRM_TARGET}Targets
|
||||
FILE ${HSAKMT_STATIC_DRM_TARGET}Targets.cmake
|
||||
NAMESPACE ${HSAKMT_STATIC_DRM_TARGET}::
|
||||
DESTINATION ${CMAKE_INSTALL_LIBDIR}/cmake/${HSAKMT_TARGET}
|
||||
COMPONENT dev)
|
||||
|
||||
add_library( ${HSAKMT_STATIC_DRM_TARGET}::${HSAKMT_STATIC_DRM_TARGET} ALIAS ${HSAKMT_STATIC_DRM_TARGET} )
|
||||
endif()
|
||||
|
||||
###########################
|
||||
# Packaging directives
|
||||
###########################
|
||||
# Use component packaging
|
||||
set ( ENABLE_LDCONFIG ON CACHE BOOL "Set library links and caches using ldconfig.")
|
||||
+23
@@ -0,0 +1,23 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -e
|
||||
|
||||
# left-hand term originates from ENABLE_LDCONFIG = ON/OFF at package build
|
||||
do_ldconfig() {
|
||||
if [ "@ENABLE_LDCONFIG@" == "ON" ]; then
|
||||
echo @CPACK_PACKAGING_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@ > /@CMAKE_INSTALL_SYSCONFDIR@/ld.so.conf.d/x86_64-libhsakmt.conf
|
||||
ldconfig
|
||||
fi
|
||||
}
|
||||
|
||||
case "$1" in
|
||||
( configure )
|
||||
do_ldconfig
|
||||
;;
|
||||
( abort-upgrade | abort-remove | abort-deconfigure )
|
||||
echo "$1"
|
||||
;;
|
||||
( * )
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -e
|
||||
|
||||
# left-hand term originates from ENABLE_LDCONFIG = ON/OFF at package build
|
||||
rm_ldconfig() {
|
||||
if [ "@ENABLE_LDCONFIG@" == "ON" ]; then
|
||||
rm -f /@CMAKE_INSTALL_SYSCONFDIR@/ld.so.conf.d/x86_64-libhsakmt.conf && ldconfig
|
||||
fi
|
||||
}
|
||||
|
||||
case "$1" in
|
||||
( remove | upgrade )
|
||||
rm_ldconfig
|
||||
;;
|
||||
( purge )
|
||||
;;
|
||||
( * )
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,50 @@
|
||||
ROCT-Thunk Interface LICENSE
|
||||
|
||||
Copyright (c) 2016 Advanced Micro Devices, Inc. All rights reserved.
|
||||
|
||||
MIT LICENSE:
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
|
||||
CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
|
||||
TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||
SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
This product contains software provided by Nginx, Inc. and its contributors.
|
||||
|
||||
Copyright (C) 2002-2018 Igor Sysoev
|
||||
Copyright (C) 2011-2018 Nginx, Inc.
|
||||
All rights reserved.
|
||||
|
||||
Redistribution and use in source and binary forms, with or without
|
||||
modification, are permitted provided that the following conditions
|
||||
are met:
|
||||
1. Redistributions of source code must retain the above copyright
|
||||
notice, this list of conditions and the following disclaimer.
|
||||
2. Redistributions in binary form must reproduce the above copyright
|
||||
notice, this list of conditions and the following disclaimer in the
|
||||
documentation and/or other materials provided with the distribution.
|
||||
|
||||
THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
|
||||
ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
|
||||
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
|
||||
OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
|
||||
HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
|
||||
LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
|
||||
OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
|
||||
SUCH DAMAGE.
|
||||
@@ -0,0 +1,47 @@
|
||||
# ROCt Library
|
||||
|
||||
This repository includes the user-mode API interfaces used to interact with the ROCk driver.
|
||||
|
||||
Starting at 1.7 release, ROCt uses drm render device. This requires the user to belong to video group. Add the user account to video group with "sudo usermod -a -G video _username_" command if the user if not part of video group yet.
|
||||
NOTE: Users of Ubuntu 20.04 will need to add the user to the new "render" group, as Ubuntu has changed the owner:group of /dev/kfd to render:render as of that release
|
||||
|
||||
## ROCk Driver
|
||||
|
||||
The ROCt library is not a standalone product and requires that you have the correct ROCk driver installed, or are using a compatible upstream kernel.
|
||||
Please refer to <https://rocm.docs.amd.com> under "Getting Started Guide" for a list of supported Operating Systems and kernel versions, as well as supported hardware.
|
||||
|
||||
## Building the Thunk
|
||||
|
||||
A simple cmake-based system is available for building thunk. To build the thunk from the the ROCT-Thunk-Interface directory, execute:
|
||||
|
||||
```bash
|
||||
mkdir -p build
|
||||
cd build
|
||||
cmake ..
|
||||
make
|
||||
```
|
||||
|
||||
If the hsakmt-roct and hsakmt-roct-dev packages are desired:
|
||||
|
||||
```bash
|
||||
mkdir -p build
|
||||
cd build
|
||||
cmake ..
|
||||
make package
|
||||
```
|
||||
|
||||
If you choose not to build and install packages, manual installation of the binaries and header files can be done via:
|
||||
|
||||
```bash
|
||||
make install
|
||||
```
|
||||
|
||||
NOTE: For older versions of the thunk where hsakmt-dev.txt is present, "make package-dev" and "make install-dev" are required to generate/install the developer packages. Currently, these are created via the "make package" and "make install" commands
|
||||
|
||||
## Disclaimer
|
||||
|
||||
The information contained herein is for informational purposes only, and is subject to change without notice. While every precaution has been taken in the preparation of this document, it may contain technical inaccuracies, omissions and typographical errors, and AMD is under no obligation to update or otherwise correct this information. Advanced Micro Devices, Inc. makes no representations or warranties with respect to the accuracy or completeness of the contents of this document, and assumes no liability of any kind, including the implied warranties of noninfringement, merchantability or fitness for particular purposes, with respect to the operation or use of AMD hardware, software or other products described herein. No license, including implied or arising by estoppel, to any intellectual property rights is granted by this document. Terms and limitations applicable to the purchase or use of AMD's products are as set forth in a signed agreement between the parties or in AMD's Standard Terms and Conditions of Sale.
|
||||
|
||||
AMD, the AMD Arrow logo, and combinations thereof are trademarks of Advanced Micro Devices, Inc. Other product names used in this publication are for identification purposes only and may be trademarks of their respective companies.
|
||||
|
||||
Copyright (c) 2014-2023 Advanced Micro Devices, Inc. All rights reserved.
|
||||
@@ -0,0 +1,97 @@
|
||||
# Restore old style debuginfo creation for rpm >= 4.14.
|
||||
%undefine _debugsource_packages
|
||||
%undefine _debuginfo_subpackages
|
||||
|
||||
# -*- rpm-spec -*-
|
||||
BuildRoot: %_topdir/@CPACK_PACKAGE_FILE_NAME@@CPACK_RPM_PACKAGE_COMPONENT_PART_PATH@
|
||||
Summary: @CPACK_RPM_PACKAGE_SUMMARY@
|
||||
Name: @CPACK_RPM_PACKAGE_NAME@
|
||||
Version: @CPACK_RPM_PACKAGE_VERSION@
|
||||
Release: @CPACK_RPM_PACKAGE_RELEASE@
|
||||
License: @CPACK_RPM_PACKAGE_LICENSE@
|
||||
Group: @CPACK_RPM_PACKAGE_GROUP@
|
||||
Vendor: @CPACK_RPM_PACKAGE_VENDOR@
|
||||
|
||||
@TMP_RPM_URL@
|
||||
@TMP_RPM_REQUIRES@
|
||||
@TMP_RPM_REQUIRES_PRE@
|
||||
@TMP_RPM_REQUIRES_POST@
|
||||
@TMP_RPM_REQUIRES_PREUN@
|
||||
@TMP_RPM_REQUIRES_POSTUN@
|
||||
@TMP_RPM_PROVIDES@
|
||||
@TMP_RPM_OBSOLETES@
|
||||
@TMP_RPM_CONFLICTS@
|
||||
@TMP_RPM_SUGGESTS@
|
||||
@TMP_RPM_AUTOPROV@
|
||||
@TMP_RPM_AUTOREQ@
|
||||
@TMP_RPM_AUTOREQPROV@
|
||||
@TMP_RPM_BUILDARCH@
|
||||
@TMP_RPM_PREFIXES@
|
||||
@TMP_RPM_EPOCH@
|
||||
|
||||
# Modifications to allow recommends to be used (not implemented in cpack):
|
||||
%if "@CPACK_RPM_PACKAGE_RECOMMENDS@" != ""
|
||||
Recommends: @CPACK_RPM_PACKAGE_RECOMMENDS@
|
||||
%endif
|
||||
# End of modifications
|
||||
|
||||
@TMP_RPM_DEBUGINFO@
|
||||
|
||||
%define _rpmdir %_topdir/RPMS
|
||||
%define _srcrpmdir %_topdir/SRPMS
|
||||
@FILE_NAME_DEFINE@
|
||||
%define _unpackaged_files_terminate_build 0
|
||||
@TMP_RPM_SPEC_INSTALL_POST@
|
||||
@CPACK_RPM_SPEC_MORE_DEFINE@
|
||||
@CPACK_RPM_COMPRESSION_TYPE_TMP@
|
||||
|
||||
%description
|
||||
@CPACK_RPM_PACKAGE_DESCRIPTION@
|
||||
|
||||
# This is a shortcutted spec file generated by CMake RPM generator
|
||||
# we skip _install step because CPack does that for us.
|
||||
# We do only save CPack installed tree in _prepr
|
||||
# and then restore it in build.
|
||||
%prep
|
||||
mv $RPM_BUILD_ROOT %_topdir/tmpBBroot
|
||||
|
||||
%install
|
||||
if [ -e $RPM_BUILD_ROOT ];
|
||||
then
|
||||
rm -rf $RPM_BUILD_ROOT
|
||||
fi
|
||||
mv %_topdir/tmpBBroot $RPM_BUILD_ROOT
|
||||
|
||||
@TMP_RPM_DEBUGINFO_INSTALL@
|
||||
|
||||
%clean
|
||||
|
||||
%post
|
||||
@RPM_SYMLINK_POSTINSTALL@
|
||||
@CPACK_RPM_SPEC_POSTINSTALL@
|
||||
|
||||
%posttrans
|
||||
@CPACK_RPM_SPEC_POSTTRANS@
|
||||
|
||||
%postun
|
||||
@CPACK_RPM_SPEC_POSTUNINSTALL@
|
||||
|
||||
%pre
|
||||
@CPACK_RPM_SPEC_PREINSTALL@
|
||||
|
||||
%pretrans
|
||||
@CPACK_RPM_SPEC_PRETRANS@
|
||||
|
||||
%preun
|
||||
@CPACK_RPM_SPEC_PREUNINSTALL@
|
||||
|
||||
%files
|
||||
%defattr(@TMP_DEFAULT_FILE_PERMISSIONS@,@TMP_DEFAULT_USER@,@TMP_DEFAULT_GROUP@,@TMP_DEFAULT_DIR_PERMISSIONS@)
|
||||
@CPACK_RPM_INSTALL_FILES@
|
||||
@CPACK_RPM_ABSOLUTE_INSTALL_FILES@
|
||||
@CPACK_RPM_USER_INSTALL_FILES@
|
||||
|
||||
%changelog
|
||||
@CPACK_RPM_SPEC_CHANGELOG@
|
||||
|
||||
@TMP_OTHER_COMPONENTS@
|
||||
@@ -0,0 +1,42 @@
|
||||
%define name hsakmt-rocm-dev
|
||||
%define version %{getenv:PACKAGE_VER}
|
||||
%define packageroot %{getenv:PACKAGE_DIR}
|
||||
|
||||
Name: %{name}
|
||||
Version: %{version}
|
||||
Release: 1
|
||||
Summary: Thunk libraries for AMD KFD
|
||||
|
||||
Group: System Environment/Libraries
|
||||
License: Advanced Micro Devices Inc.
|
||||
|
||||
%if 0%{?centos} == 6
|
||||
Requires: numactl
|
||||
%else
|
||||
Requires: numactl-libs
|
||||
%endif
|
||||
|
||||
|
||||
%description
|
||||
This package includes the libhsakmt (Thunk) libraries
|
||||
for AMD KFD
|
||||
|
||||
%prep
|
||||
%setup -T -D -c -n %{name}
|
||||
|
||||
%install
|
||||
cp -R %packageroot $RPM_BUILD_ROOT
|
||||
find $RPM_BUILD_ROOT \! -type d | sed "s|$RPM_BUILD_ROOT||"> thunk.list
|
||||
|
||||
%post
|
||||
ldconfig
|
||||
|
||||
%postun
|
||||
ldconfig
|
||||
|
||||
%clean
|
||||
rm -rf $RPM_BUILD_ROOT
|
||||
|
||||
%files -f thunk.list
|
||||
|
||||
%defattr(-,root,root,-)
|
||||
@@ -0,0 +1,5 @@
|
||||
# left-hand term originates from ENABLE_LDCONFIG = ON/OFF at package build
|
||||
if [ "@ENABLE_LDCONFIG@" == "ON" ]; then
|
||||
echo -e "@CPACK_PACKAGING_INSTALL_PREFIX@/@CMAKE_INSTALL_LIBDIR@" > /@CMAKE_INSTALL_SYSCONFDIR@/ld.so.conf.d/x86_64-libhsakmt.conf
|
||||
ldconfig
|
||||
fi
|
||||
@@ -0,0 +1,6 @@
|
||||
# second term originates from ENABLE_LDCONFIG = ON/OFF at package build
|
||||
if [ $1 -le 1 ] && [ "@ENABLE_LDCONFIG@" == "ON" ]; then
|
||||
# perform the below actions for rpm remove($1=0) or upgrade($1=1) operations
|
||||
rm -f /@CMAKE_INSTALL_SYSCONFDIR@/ld.so.conf.d/x86_64-libhsakmt.conf
|
||||
ldconfig
|
||||
fi
|
||||
@@ -0,0 +1,139 @@
|
||||
################################################################################
|
||||
##
|
||||
## The University of Illinois/NCSA
|
||||
## Open Source License (NCSA)
|
||||
##
|
||||
## Copyright (c) 2014-2017, Advanced Micro Devices, Inc. All rights reserved.
|
||||
##
|
||||
## Developed by:
|
||||
##
|
||||
## AMD Research and AMD HSA Software Development
|
||||
##
|
||||
## Advanced Micro Devices, Inc.
|
||||
##
|
||||
## www.amd.com
|
||||
##
|
||||
## Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
## of this software and associated documentation files (the "Software"), to
|
||||
## deal with the Software without restriction, including without limitation
|
||||
## the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
## and#or sell copies of the Software, and to permit persons to whom the
|
||||
## Software is furnished to do so, subject to the following conditions:
|
||||
##
|
||||
## - Redistributions of source code must retain the above copyright notice,
|
||||
## this list of conditions and the following disclaimers.
|
||||
## - Redistributions in binary form must reproduce the above copyright
|
||||
## notice, this list of conditions and the following disclaimers in
|
||||
## the documentation and#or other materials provided with the distribution.
|
||||
## - Neither the names of Advanced Micro Devices, Inc,
|
||||
## nor the names of its contributors may be used to endorse or promote
|
||||
## products derived from this Software without specific prior written
|
||||
## permission.
|
||||
##
|
||||
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
## IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
## FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
## THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
## OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
## ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
## DEALINGS WITH THE SOFTWARE.
|
||||
##
|
||||
################################################################################
|
||||
|
||||
## Parses the VERSION_STRING variable and places
|
||||
## the first, second and third number values in
|
||||
## the major, minor and patch variables.
|
||||
function( parse_version VERSION_STRING )
|
||||
|
||||
string ( FIND ${VERSION_STRING} "-" STRING_INDEX )
|
||||
|
||||
if ( ${STRING_INDEX} GREATER -1 )
|
||||
math ( EXPR STRING_INDEX "${STRING_INDEX} + 1" )
|
||||
string ( SUBSTRING ${VERSION_STRING} ${STRING_INDEX} -1 VERSION_BUILD )
|
||||
endif ()
|
||||
|
||||
string ( REGEX MATCHALL "[0123456789]+" VERSIONS ${VERSION_STRING} )
|
||||
list ( LENGTH VERSIONS VERSION_COUNT )
|
||||
|
||||
if ( ${VERSION_COUNT} GREATER 0)
|
||||
list ( GET VERSIONS 0 MAJOR )
|
||||
set ( VERSION_MAJOR ${MAJOR} PARENT_SCOPE )
|
||||
set ( TEMP_VERSION_STRING "${MAJOR}" )
|
||||
endif ()
|
||||
|
||||
if ( ${VERSION_COUNT} GREATER 1 )
|
||||
list ( GET VERSIONS 1 MINOR )
|
||||
set ( VERSION_MINOR ${MINOR} PARENT_SCOPE )
|
||||
set ( TEMP_VERSION_STRING "${TEMP_VERSION_STRING}.${MINOR}" )
|
||||
endif ()
|
||||
|
||||
if ( ${VERSION_COUNT} GREATER 2 )
|
||||
list ( GET VERSIONS 2 PATCH )
|
||||
set ( VERSION_PATCH ${PATCH} PARENT_SCOPE )
|
||||
set ( TEMP_VERSION_STRING "${TEMP_VERSION_STRING}.${PATCH}" )
|
||||
endif ()
|
||||
|
||||
if ( DEFINED VERSION_BUILD )
|
||||
set ( VERSION_BUILD "${VERSION_BUILD}" PARENT_SCOPE )
|
||||
endif ()
|
||||
|
||||
set ( VERSION_STRING "${TEMP_VERSION_STRING}" PARENT_SCOPE )
|
||||
|
||||
endfunction ()
|
||||
|
||||
## Gets the current version of the repository
|
||||
## using versioning tags and git describe.
|
||||
## Passes back a packaging version string
|
||||
## and a library version string.
|
||||
function ( get_version DEFAULT_VERSION_STRING )
|
||||
|
||||
parse_version ( ${DEFAULT_VERSION_STRING} )
|
||||
|
||||
find_program ( GIT NAMES git )
|
||||
|
||||
if ( GIT )
|
||||
|
||||
execute_process ( COMMAND git describe --tags --dirty --long
|
||||
WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}
|
||||
OUTPUT_VARIABLE GIT_TAG_STRING
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
RESULT_VARIABLE RESULT )
|
||||
|
||||
if ( ${RESULT} EQUAL 0 )
|
||||
|
||||
parse_version ( ${GIT_TAG_STRING} )
|
||||
|
||||
endif ()
|
||||
|
||||
endif ()
|
||||
|
||||
set( VERSION_STRING "${VERSION_STRING}" PARENT_SCOPE )
|
||||
set( VERSION_MAJOR "${VERSION_MAJOR}" PARENT_SCOPE )
|
||||
set( VERSION_MINOR "${VERSION_MINOR}" PARENT_SCOPE )
|
||||
set( VERSION_PATCH "${VERSION_PATCH}" PARENT_SCOPE )
|
||||
set( VERSION_BUILD "${VERSION_BUILD}" PARENT_SCOPE )
|
||||
|
||||
endfunction()
|
||||
|
||||
#get the OS version
|
||||
function(get_os_info)
|
||||
if( EXISTS "/etc/os-release")
|
||||
file(STRINGS "/etc/os-release" DISTRO_ID REGEX "^ID=")
|
||||
file(STRINGS "/etc/os-release" DISTRO_RELEASE REGEX "^VERSION_ID=")
|
||||
string(REPLACE "ID=" "" DISTRO_ID "${DISTRO_ID}")
|
||||
string(REPLACE "VERSION_ID=" "" DISTRO_RELEASE "${DISTRO_RELEASE}")
|
||||
message(STATUS "Detected distribution: ${DISTRO_ID}:${DISTRO_RELEASE}")
|
||||
elseif(EXISTS "/etc/centos-release" )
|
||||
# Example: CentOS release 6.10 (Final)
|
||||
file(STRINGS "/etc/centos-release" DISTRO_FULL_STR REGEX "release")
|
||||
string(REGEX MATCH "^[a-zA-Z]+" DISTRO_ID "${DISTRO_FULL_STR}")
|
||||
string(TOLOWER "${DISTRO_ID}" DISTRO_ID)
|
||||
string(REGEX MATCH "[0-9]+" DISTRO_RELEASE "${DISTRO_FULL_STR}")
|
||||
message(STATUS "Detected distribution: ${DISTRO_ID}:${DISTRO_RELEASE}")
|
||||
else()
|
||||
message(STATUS "Not able to detect OS")
|
||||
endif()
|
||||
set(DISTRO_ID "${DISTRO_ID}" PARENT_SCOPE )
|
||||
set(DISTRO_RELEASE "${DISTRO_RELEASE}" PARENT_SCOPE )
|
||||
|
||||
endfunction()
|
||||
@@ -0,0 +1,19 @@
|
||||
@PACKAGE_INIT@
|
||||
|
||||
include( CMakeFindDependencyMacro )
|
||||
|
||||
# Locate dependent packages here. Finding them propagates usage requirements,
|
||||
# if any, to our clients and ensures that their target names are in scope for
|
||||
# the build. hsakmt has no cmake project dependencies so there is nothing to
|
||||
# find. If we switch to use find_package with external (to ROCm) library
|
||||
# dependencies (ie libnuma) then those packages should be located here using
|
||||
# find_dependencies as shown below.
|
||||
#find_dependency(Bar, 2.0)
|
||||
|
||||
# If the option is ON link other dependent libraries dynamically
|
||||
# If the option is OFF, then link libdrm and libdrm_amdgpu statically
|
||||
if(@BUILD_SHARED_LIBS@)
|
||||
include( "${CMAKE_CURRENT_LIST_DIR}/@HSAKMT_TARGET@Targets.cmake" )
|
||||
else()
|
||||
include( "${CMAKE_CURRENT_LIST_DIR}/@HSAKMT_STATIC_DRM_TARGET@Targets.cmake" )
|
||||
endif()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,36 @@
|
||||
/*
|
||||
* Copyright © 2025 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#ifndef _HSAKMTMODEL_H_
|
||||
#define _HSAKMTMODEL_H_
|
||||
#include <stdbool.h>
|
||||
extern bool hsakmt_use_model;
|
||||
extern char *hsakmt_model_topology;
|
||||
void model_init_env_vars(void);
|
||||
void model_init(void);
|
||||
void model_set_mmio_page(void *ptr);
|
||||
void model_set_event_page(void *ptr, unsigned event_limit);
|
||||
int model_kfd_ioctl(unsigned long request, void *arg);
|
||||
#endif /* _HSAKMTMODEL_H_ */
|
||||
@@ -0,0 +1,109 @@
|
||||
/*
|
||||
* Copyright © 2025 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#ifndef _HSAKMTMODELIFACE_H_
|
||||
#define _HSAKMTMODELIFACE_H_
|
||||
|
||||
#include <inttypes.h>
|
||||
|
||||
// Changelog:
|
||||
// 0.2: Add set_set_event function to hsakmt_model_functions
|
||||
#define HSAKMT_MODEL_INTERFACE_VERSION_MAJOR 0
|
||||
#define HSAKMT_MODEL_INTERFACE_VERSION_MINOR 4
|
||||
|
||||
typedef struct hsakmt_model hsakmt_model_t;
|
||||
typedef struct hsakmt_model_queue hsakmt_model_queue_t;
|
||||
|
||||
// Description of a queue to be registered with the model.
|
||||
//
|
||||
// Addresses are relative to the global aperture.
|
||||
struct hsakmt_model_queue_info {
|
||||
uint64_t ring_base_address;
|
||||
uint64_t write_pointer_address;
|
||||
uint64_t read_pointer_address;
|
||||
|
||||
uint64_t *doorbell;
|
||||
|
||||
uint32_t ring_size; // in bytes
|
||||
uint32_t queue_type;
|
||||
};
|
||||
|
||||
// Pointer to a "set event" function.
|
||||
//
|
||||
// data is a user-provided opaque pointer.
|
||||
// event_id is the ID of the event to set (as in amd_signal_s::event_id).
|
||||
typedef void (*hsakmt_model_set_event_fn)(void *data, unsigned event_id);
|
||||
|
||||
// Interface provided by the software model implementation.
|
||||
//
|
||||
// Queried from a shared library by calling an export called
|
||||
// `get_hsakmt_model_functions`
|
||||
//
|
||||
// Interface versioning follows the semantic versioning model: clients that
|
||||
// know about interface version X.Y can use any implementation that provides
|
||||
// version X.Z with Z >= Y.
|
||||
//
|
||||
// The model is designed to support only one VMID space.
|
||||
struct hsakmt_model_functions {
|
||||
uint32_t version_major; // HSAKMT_MODEL_INTERFACE_VERSION_MAJOR
|
||||
uint32_t version_minor; // HSAKMT_MODEL_INTERFACE_VERSION_MINOR
|
||||
|
||||
// Create a GPU device model.
|
||||
hsakmt_model_t *(*create)(void);
|
||||
|
||||
// Destroy a GPU device model.
|
||||
void (*destroy)(hsakmt_model_t *model);
|
||||
|
||||
// Set the global aperture. GPU virtual address 0 is at CPU address `base`.
|
||||
void (*set_global_aperture)(hsakmt_model_t *model, void *base, uint64_t size);
|
||||
void (*alloced_memory)(hsakmt_model_t *model, void *base, uint64_t size, uint32_t flags);
|
||||
void (*freed_memory)(hsakmt_model_t *model, void *base, uint64_t size);
|
||||
// Register a callback that the model should call when an event is signaled.
|
||||
// `data` is client data that is opaque to the model.
|
||||
//
|
||||
// TODO: Deprecated -- remove this!
|
||||
void (*set_notify_event)(hsakmt_model_t *model, void (*callback)(void *data), void *data);
|
||||
|
||||
// Register a callback that the model should call in order to wait for an
|
||||
// event to be signaled.
|
||||
// `data` is client data that is opaque to the model.
|
||||
void (*set_wait_event)(hsakmt_model_t *model, void (*callback)(void *data, uint64_t address, uint64_t age), void *data);
|
||||
|
||||
// Register a queue with the model. The model will immediately begin
|
||||
// asynchronous processing of the queue (but by default, the model need not
|
||||
// provide forward progress guarantees between multiple queues).
|
||||
hsakmt_model_queue_t *(*register_queue)(hsakmt_model_t *model, struct hsakmt_model_queue_info *info);
|
||||
|
||||
// Register a callback that allows the model to set an event.
|
||||
void (*set_set_event)(hsakmt_model_t *model, hsakmt_model_set_event_fn fn, void *data);
|
||||
|
||||
// Destroy a queue that was returned by register_queue.
|
||||
void (*destroy_queue)(hsakmt_model_t *model, hsakmt_model_queue_t *queue);
|
||||
};
|
||||
|
||||
// Type of a shared library export called `get_hsakmt_model_functions`.
|
||||
typedef const struct hsakmt_model_functions *(*get_hsakmt_model_functions_t)(void);
|
||||
|
||||
#endif // _HSAKMTMODELIFACE_H_
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,11 @@
|
||||
prefix=${pcfiledir}/../..
|
||||
exec_prefix=${prefix}
|
||||
libdir=${prefix}/@CMAKE_INSTALL_LIBDIR@
|
||||
includedir=${prefix}/@CMAKE_INSTALL_INCLUDEDIR@
|
||||
|
||||
Name: libhsakmt
|
||||
Description: HSA Kernel Mode Thunk library for AMD KFD support
|
||||
Version: @LIB_VERSION_STRING@
|
||||
|
||||
Libs: -L${libdir} -lhsakmt
|
||||
Cflags: -I${includedir}
|
||||
@@ -0,0 +1,559 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
#include <errno.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
static bool *is_device_debugged;
|
||||
static uint32_t runtime_capabilities_mask = 0;
|
||||
|
||||
HSAKMT_STATUS hsakmt_init_device_debugging_memory(unsigned int NumNodes)
|
||||
{
|
||||
unsigned int i;
|
||||
|
||||
is_device_debugged = malloc(NumNodes * sizeof(bool));
|
||||
if (!is_device_debugged)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
for (i = 0; i < NumNodes; i++)
|
||||
is_device_debugged[i] = false;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
void hsakmt_destroy_device_debugging_memory(void)
|
||||
{
|
||||
if (is_device_debugged) {
|
||||
free(is_device_debugged);
|
||||
is_device_debugged = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
bool hsakmt_debug_get_reg_status(uint32_t node_id)
|
||||
{
|
||||
return is_device_debugged[node_id];
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDbgRegister(HSAuint32 NodeId)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpu_id;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (!is_device_debugged)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
result = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
struct kfd_ioctl_dbg_register_args args = {0};
|
||||
|
||||
args.gpu_id = gpu_id;
|
||||
|
||||
long err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_REGISTER_DEPRECATED, &args);
|
||||
|
||||
if (err == 0)
|
||||
result = HSAKMT_STATUS_SUCCESS;
|
||||
else
|
||||
result = HSAKMT_STATUS_ERROR;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDbgUnregister(HSAuint32 NodeId)
|
||||
{
|
||||
uint32_t gpu_id;
|
||||
HSAKMT_STATUS result;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (!is_device_debugged)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
result = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
struct kfd_ioctl_dbg_unregister_args args = {0};
|
||||
|
||||
args.gpu_id = gpu_id;
|
||||
long err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_UNREGISTER_DEPRECATED, &args);
|
||||
|
||||
if (err)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDbgWavefrontControl(HSAuint32 NodeId,
|
||||
HSA_DBG_WAVEOP Operand,
|
||||
HSA_DBG_WAVEMODE Mode,
|
||||
HSAuint32 TrapId,
|
||||
HsaDbgWaveMessage *DbgWaveMsgRing)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpu_id;
|
||||
|
||||
struct kfd_ioctl_dbg_wave_control_args *args;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
result = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
|
||||
/* Determine Size of the ioctl buffer */
|
||||
uint32_t buff_size = sizeof(Operand) + sizeof(Mode) + sizeof(TrapId) +
|
||||
sizeof(DbgWaveMsgRing->DbgWaveMsg) +
|
||||
sizeof(DbgWaveMsgRing->MemoryVA) + sizeof(*args);
|
||||
|
||||
args = (struct kfd_ioctl_dbg_wave_control_args *)malloc(buff_size);
|
||||
if (!args)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
memset(args, 0, buff_size);
|
||||
|
||||
args->gpu_id = gpu_id;
|
||||
args->buf_size_in_bytes = buff_size;
|
||||
|
||||
/* increment pointer to the start of the non fixed part */
|
||||
unsigned char *run_ptr = (unsigned char *)args + sizeof(*args);
|
||||
|
||||
/* save variable content pointer for kfd */
|
||||
args->content_ptr = (uint64_t)run_ptr;
|
||||
|
||||
/* insert items, and increment pointer accordingly */
|
||||
*((HSA_DBG_WAVEOP *)run_ptr) = Operand;
|
||||
run_ptr += sizeof(Operand);
|
||||
|
||||
*((HSA_DBG_WAVEMODE *)run_ptr) = Mode;
|
||||
run_ptr += sizeof(Mode);
|
||||
|
||||
*((HSAuint32 *)run_ptr) = TrapId;
|
||||
run_ptr += sizeof(TrapId);
|
||||
|
||||
*((HsaDbgWaveMessageAMD *)run_ptr) = DbgWaveMsgRing->DbgWaveMsg;
|
||||
run_ptr += sizeof(DbgWaveMsgRing->DbgWaveMsg);
|
||||
|
||||
*((void **)run_ptr) = DbgWaveMsgRing->MemoryVA;
|
||||
run_ptr += sizeof(DbgWaveMsgRing->MemoryVA);
|
||||
|
||||
/* send to kernel */
|
||||
long err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_WAVE_CONTROL_DEPRECATED, args);
|
||||
|
||||
free(args);
|
||||
|
||||
if (err)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDbgAddressWatch(HSAuint32 NodeId,
|
||||
HSAuint32 NumWatchPoints,
|
||||
HSA_DBG_WATCH_MODE WatchMode[],
|
||||
void *WatchAddress[],
|
||||
HSAuint64 WatchMask[],
|
||||
HsaEvent *WatchEvent[])
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpu_id;
|
||||
|
||||
/* determine the size of the watch mask and event buffers
|
||||
* the value is NULL if and only if no vector data should be attached
|
||||
*/
|
||||
uint32_t watch_mask_items = WatchMask[0] > 0 ? NumWatchPoints:1;
|
||||
uint32_t watch_event_items = WatchEvent != NULL ? NumWatchPoints:0;
|
||||
|
||||
struct kfd_ioctl_dbg_address_watch_args *args;
|
||||
HSAuint32 i = 0;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
result = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
if (NumWatchPoints > MAX_ALLOWED_NUM_POINTS)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
/* Size and structure of the ioctl buffer is dynamic in this case
|
||||
* Here we calculate the buff size.
|
||||
*/
|
||||
uint32_t buff_size = sizeof(NumWatchPoints) +
|
||||
(sizeof(WatchMode[0]) + sizeof(WatchAddress[0])) *
|
||||
NumWatchPoints +
|
||||
watch_mask_items * sizeof(HSAuint64) +
|
||||
watch_event_items * sizeof(HsaEvent *) + sizeof(*args);
|
||||
|
||||
args = (struct kfd_ioctl_dbg_address_watch_args *) malloc(buff_size);
|
||||
if (!args)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
memset(args, 0, buff_size);
|
||||
|
||||
args->gpu_id = gpu_id;
|
||||
args->buf_size_in_bytes = buff_size;
|
||||
|
||||
|
||||
/* increment pointer to the start of the non fixed part */
|
||||
unsigned char *run_ptr = (unsigned char *)args + sizeof(*args);
|
||||
|
||||
/* save variable content pointer for kfd */
|
||||
args->content_ptr = (uint64_t)run_ptr;
|
||||
/* insert items, and increment pointer accordingly */
|
||||
|
||||
*((HSAuint32 *)run_ptr) = NumWatchPoints;
|
||||
run_ptr += sizeof(NumWatchPoints);
|
||||
|
||||
for (i = 0; i < NumWatchPoints; i++) {
|
||||
*((HSA_DBG_WATCH_MODE *)run_ptr) = WatchMode[i];
|
||||
run_ptr += sizeof(WatchMode[i]);
|
||||
}
|
||||
|
||||
for (i = 0; i < NumWatchPoints; i++) {
|
||||
*((void **)run_ptr) = WatchAddress[i];
|
||||
run_ptr += sizeof(WatchAddress[i]);
|
||||
}
|
||||
|
||||
for (i = 0; i < watch_mask_items; i++) {
|
||||
*((HSAuint64 *)run_ptr) = WatchMask[i];
|
||||
run_ptr += sizeof(WatchMask[i]);
|
||||
}
|
||||
|
||||
for (i = 0; i < watch_event_items; i++) {
|
||||
*((HsaEvent **)run_ptr) = WatchEvent[i];
|
||||
run_ptr += sizeof(WatchEvent[i]);
|
||||
}
|
||||
|
||||
/* send to kernel */
|
||||
long err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_ADDRESS_WATCH_DEPRECATED, args);
|
||||
|
||||
free(args);
|
||||
|
||||
if (err)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
#define HSA_RUNTIME_ENABLE_MAX_MAJOR 1
|
||||
#define HSA_RUNTIME_ENABLE_MIN_MINOR 13
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtCheckRuntimeDebugSupport(void) {
|
||||
HsaNodeProperties node = {0};
|
||||
HsaSystemProperties props = {0};
|
||||
HsaVersionInfo versionInfo = {0};
|
||||
|
||||
memset(&node, 0x00, sizeof(node));
|
||||
memset(&props, 0x00, sizeof(props));
|
||||
if (hsaKmtAcquireSystemProperties(&props))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
//the firmware of gpu node doesn't support the debugger, disable it.
|
||||
for (uint32_t i = 0; i < props.NumNodes; i++) {
|
||||
if (hsaKmtGetNodeProperties(i, &node))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
//ignore cpu node
|
||||
if (node.NumCPUCores && !node.NumFComputeCores)
|
||||
continue;
|
||||
if (!node.Capability.ui32.DebugSupportedFirmware)
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
}
|
||||
|
||||
if (hsaKmtGetVersion(&versionInfo))
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
|
||||
if (versionInfo.KernelInterfaceMajorVersion < HSA_RUNTIME_ENABLE_MAX_MAJOR ||
|
||||
(versionInfo.KernelInterfaceMajorVersion ==
|
||||
HSA_RUNTIME_ENABLE_MAX_MAJOR &&
|
||||
(int)versionInfo.KernelInterfaceMinorVersion < HSA_RUNTIME_ENABLE_MIN_MINOR))
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRuntimeEnable(void *rDebug,
|
||||
bool setupTtmp)
|
||||
{
|
||||
struct kfd_ioctl_runtime_enable_args args = {0};
|
||||
HSAKMT_STATUS result = hsaKmtCheckRuntimeDebugSupport();
|
||||
|
||||
if (result)
|
||||
return result;
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
args.mode_mask = KFD_RUNTIME_ENABLE_MODE_ENABLE_MASK |
|
||||
((setupTtmp) ? KFD_RUNTIME_ENABLE_MODE_TTMP_SAVE_MASK : 0);
|
||||
args.r_debug = (HSAuint64)rDebug;
|
||||
|
||||
long err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_RUNTIME_ENABLE, &args);
|
||||
|
||||
if (err) {
|
||||
if (errno == EBUSY)
|
||||
return HSAKMT_STATUS_UNAVAILABLE;
|
||||
else
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
runtime_capabilities_mask= args.capabilities_mask;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRuntimeDisable(void)
|
||||
{
|
||||
struct kfd_ioctl_runtime_enable_args args = {0};
|
||||
HSAKMT_STATUS result = hsaKmtCheckRuntimeDebugSupport();
|
||||
|
||||
if (result)
|
||||
return result;
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
args.mode_mask = 0; //Disable
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_RUNTIME_ENABLE, &args))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtGetRuntimeCapabilities(HSAuint32 *caps_mask)
|
||||
{
|
||||
*caps_mask = runtime_capabilities_mask;
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS dbg_trap_get_device_data(void *data,
|
||||
uint32_t *n_entries,
|
||||
uint32_t entry_size)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
args.device_snapshot.snapshot_buf_ptr = (uint64_t) data;
|
||||
args.device_snapshot.num_devices = *n_entries;
|
||||
args.device_snapshot.entry_size = entry_size;
|
||||
args.op = KFD_IOC_DBG_TRAP_GET_DEVICE_SNAPSHOT;
|
||||
args.pid = getpid();
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_TRAP, &args))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
*n_entries = args.device_snapshot.num_devices;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS dbg_trap_get_queue_data(void *data,
|
||||
uint32_t *n_entries,
|
||||
uint32_t entry_size,
|
||||
uint32_t *queue_ids)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
args.queue_snapshot.num_queues = *n_entries;
|
||||
args.queue_snapshot.entry_size = entry_size;
|
||||
args.queue_snapshot.exception_mask = KFD_EC_MASK(EC_QUEUE_NEW);
|
||||
args.op = KFD_IOC_DBG_TRAP_GET_QUEUE_SNAPSHOT;
|
||||
args.queue_snapshot.snapshot_buf_ptr = (uint64_t) data;
|
||||
args.pid = getpid();
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_TRAP, &args))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
*n_entries = args.queue_snapshot.num_queues;
|
||||
if (queue_ids && *n_entries) {
|
||||
struct kfd_queue_snapshot_entry *queue_entry =
|
||||
(struct kfd_queue_snapshot_entry *) data;
|
||||
for (uint32_t i = 0; i < *n_entries; i++)
|
||||
queue_ids[i] = queue_entry[i].queue_id;
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS dbg_trap_suspend_queues(uint32_t *queue_ids,
|
||||
uint32_t num_queues)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
int r;
|
||||
|
||||
args.suspend_queues.queue_array_ptr = (uint64_t) queue_ids;
|
||||
args.suspend_queues.num_queues = num_queues;
|
||||
args.suspend_queues.exception_mask = KFD_EC_MASK(EC_QUEUE_NEW);
|
||||
args.op = KFD_IOC_DBG_TRAP_SUSPEND_QUEUES;
|
||||
args.pid = getpid();
|
||||
|
||||
r = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_TRAP, &args);
|
||||
if (r < 0)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
/* Debugger support has been in KFD ABI 1.13. */
|
||||
#define KFD_MINOR_MIN_DEBUG 13
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDbgEnable(void **runtime_info,
|
||||
HSAuint32 *data_size)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(KFD_MINOR_MIN_DEBUG);
|
||||
*data_size = sizeof(struct kfd_runtime_info);
|
||||
args.enable.rinfo_size = *data_size;
|
||||
args.enable.dbg_fd = hsakmt_kfd_fd;
|
||||
*runtime_info = malloc(args.enable.rinfo_size);
|
||||
if (!*runtime_info)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
args.enable.rinfo_ptr = (uint64_t) *runtime_info;
|
||||
args.op = KFD_IOC_DBG_TRAP_ENABLE;
|
||||
args.pid = getpid();
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_TRAP, &args)) {
|
||||
free(*runtime_info);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDbgDisable(void)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(KFD_MINOR_MIN_DEBUG);
|
||||
args.enable.dbg_fd = hsakmt_kfd_fd;
|
||||
args.op = KFD_IOC_DBG_TRAP_DISABLE;
|
||||
args.pid = getpid();
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_TRAP, &args))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDbgGetDeviceData(void **data,
|
||||
HSAuint32 *n_entries,
|
||||
HSAuint32 *entry_size)
|
||||
{
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(KFD_MINOR_MIN_DEBUG);
|
||||
*n_entries = UINT32_MAX;
|
||||
*entry_size = sizeof(struct kfd_dbg_device_info_entry);
|
||||
*data = malloc(*entry_size * *n_entries);
|
||||
if (!*data)
|
||||
return ret;
|
||||
ret = dbg_trap_get_device_data(*data, n_entries, *entry_size);
|
||||
if (ret)
|
||||
free(*data);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDbgGetQueueData(void **data,
|
||||
HSAuint32 *n_entries,
|
||||
HSAuint32 *entry_size,
|
||||
bool suspend_queues)
|
||||
{
|
||||
uint32_t *queue_ids = NULL;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(KFD_MINOR_MIN_DEBUG);
|
||||
*entry_size = sizeof(struct kfd_queue_snapshot_entry);
|
||||
*n_entries = 0;
|
||||
if (dbg_trap_get_queue_data(NULL, n_entries, *entry_size, NULL))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
*data = malloc(*n_entries * *entry_size);
|
||||
if (!*data)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
if (suspend_queues && *n_entries)
|
||||
queue_ids = (uint32_t *)malloc(sizeof(uint32_t) * *n_entries);
|
||||
if (!queue_ids ||
|
||||
dbg_trap_get_queue_data(*data, n_entries, *entry_size, queue_ids))
|
||||
goto free_data;
|
||||
if (queue_ids) {
|
||||
if (dbg_trap_suspend_queues(queue_ids, *n_entries) ||
|
||||
dbg_trap_get_queue_data(*data, n_entries, *entry_size, NULL))
|
||||
goto free_data;
|
||||
free(queue_ids);
|
||||
}
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
free_data:
|
||||
free(*data);
|
||||
if (queue_ids)
|
||||
free(queue_ids);
|
||||
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDebugTrapIoctl(struct kfd_ioctl_dbg_trap_args *args,
|
||||
HSA_QUEUEID *Queues,
|
||||
HSAuint64 *DebugReturn)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (Queues) {
|
||||
int num_queues = args->op == KFD_IOC_DBG_TRAP_SUSPEND_QUEUES ?
|
||||
args->suspend_queues.num_queues :
|
||||
args->resume_queues.num_queues;
|
||||
void *queue_ptr = args->op == KFD_IOC_DBG_TRAP_SUSPEND_QUEUES ?
|
||||
(void *)args->suspend_queues.queue_array_ptr :
|
||||
(void *)args->resume_queues.queue_array_ptr;
|
||||
|
||||
uint32_t *queue_ids = hsakmt_convert_queue_ids(num_queues, Queues);
|
||||
if (!queue_ids) {
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
}
|
||||
memcpy(queue_ptr, queue_ids, num_queues * sizeof(uint32_t));
|
||||
free(queue_ids);
|
||||
}
|
||||
|
||||
long err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DBG_TRAP, args);
|
||||
if (DebugReturn)
|
||||
*DebugReturn = err;
|
||||
|
||||
if (args->op == KFD_IOC_DBG_TRAP_SUSPEND_QUEUES &&
|
||||
err >= 0 && err <= args->suspend_queues.num_queues)
|
||||
result = HSAKMT_STATUS_SUCCESS;
|
||||
else if (args->op == KFD_IOC_DBG_TRAP_RESUME_QUEUES &&
|
||||
err >= 0 && err <= args->resume_queues.num_queues)
|
||||
result = HSAKMT_STATUS_SUCCESS;
|
||||
else if (err == 0)
|
||||
result = HSAKMT_STATUS_SUCCESS;
|
||||
else
|
||||
result = HSAKMT_STATUS_ERROR;
|
||||
|
||||
return result;
|
||||
}
|
||||
@@ -0,0 +1,492 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <time.h>
|
||||
#include <errno.h>
|
||||
#include <unistd.h>
|
||||
#include <sys/mman.h>
|
||||
#include <stdio.h>
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
#include "fmm.h"
|
||||
#include "hsakmt/hsakmtmodel.h"
|
||||
|
||||
static HSAuint64 *events_page = NULL;
|
||||
|
||||
void hsakmt_clear_events_page(void)
|
||||
{
|
||||
events_page = NULL;
|
||||
}
|
||||
|
||||
static bool IsSystemEventType(HSA_EVENTTYPE type)
|
||||
{
|
||||
// Debug events behave as signal events.
|
||||
return (type != HSA_EVENTTYPE_SIGNAL && type != HSA_EVENTTYPE_DEBUG_EVENT);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtCreateEvent(HsaEventDescriptor *EventDesc,
|
||||
bool ManualReset, bool IsSignaled,
|
||||
HsaEvent **Event)
|
||||
{
|
||||
unsigned int event_limit = KFD_SIGNAL_EVENT_LIMIT;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (EventDesc->EventType >= HSA_EVENTTYPE_MAXID)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
HsaEvent *e = malloc(sizeof(HsaEvent));
|
||||
|
||||
if (!e)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
memset(e, 0, sizeof(*e));
|
||||
|
||||
struct kfd_ioctl_create_event_args args = {0};
|
||||
|
||||
args.event_type = EventDesc->EventType;
|
||||
args.node_id = EventDesc->NodeId;
|
||||
args.auto_reset = !ManualReset;
|
||||
|
||||
/* dGPU code */
|
||||
pthread_mutex_lock(&hsakmt_mutex);
|
||||
|
||||
if (hsakmt_is_dgpu && !events_page) {
|
||||
events_page = hsakmt_allocate_exec_aligned_memory_gpu(
|
||||
KFD_SIGNAL_EVENT_LIMIT * 8, PAGE_SIZE, 0, 0, true, false, true);
|
||||
if (!events_page) {
|
||||
free(e);
|
||||
pthread_mutex_unlock(&hsakmt_mutex);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
if (hsakmt_use_model)
|
||||
model_set_event_page(events_page, KFD_SIGNAL_EVENT_LIMIT);
|
||||
else
|
||||
hsakmt_fmm_get_handle(events_page, (uint64_t *)&args.event_page_offset);
|
||||
}
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_CREATE_EVENT, &args) != 0) {
|
||||
free(e);
|
||||
*Event = NULL;
|
||||
pthread_mutex_unlock(&hsakmt_mutex);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
e->EventId = args.event_id;
|
||||
|
||||
if (!events_page && args.event_page_offset > 0) {
|
||||
events_page = mmap(NULL, event_limit * 8, PROT_WRITE | PROT_READ,
|
||||
MAP_SHARED, hsakmt_kfd_fd, args.event_page_offset);
|
||||
if (events_page == MAP_FAILED) {
|
||||
/* old kernels only support 256 events */
|
||||
event_limit = 256;
|
||||
events_page = mmap(NULL, PAGE_SIZE, PROT_WRITE | PROT_READ,
|
||||
MAP_SHARED, hsakmt_kfd_fd, args.event_page_offset);
|
||||
}
|
||||
if (events_page == MAP_FAILED) {
|
||||
events_page = NULL;
|
||||
pthread_mutex_unlock(&hsakmt_mutex);
|
||||
hsaKmtDestroyEvent(e);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
}
|
||||
|
||||
if (args.event_page_offset > 0 && args.event_slot_index < event_limit)
|
||||
e->EventData.HWData2 = (HSAuint64)&events_page[args.event_slot_index];
|
||||
|
||||
pthread_mutex_unlock(&hsakmt_mutex);
|
||||
|
||||
e->EventData.EventType = EventDesc->EventType;
|
||||
e->EventData.HWData1 = args.event_id;
|
||||
|
||||
e->EventData.HWData3 = args.event_trigger_data;
|
||||
e->EventData.EventData.SyncVar.SyncVar.UserData =
|
||||
EventDesc->SyncVar.SyncVar.UserData;
|
||||
e->EventData.EventData.SyncVar.SyncVarSize =
|
||||
EventDesc->SyncVar.SyncVarSize;
|
||||
|
||||
if (IsSignaled && !IsSystemEventType(e->EventData.EventType)) {
|
||||
struct kfd_ioctl_set_event_args set_args = {0};
|
||||
|
||||
set_args.event_id = args.event_id;
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SET_EVENT,
|
||||
&set_args) != 0) {
|
||||
hsaKmtDestroyEvent(e);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
}
|
||||
|
||||
*Event = e;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDestroyEvent(HsaEvent *Event)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (!Event)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
struct kfd_ioctl_destroy_event_args args = {0};
|
||||
|
||||
args.event_id = Event->EventId;
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DESTROY_EVENT, &args) != 0)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
free(Event);
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtSetEvent(HsaEvent *Event)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (!Event)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
/* Although the spec is doesn't say, don't allow system-defined events
|
||||
* to be signaled.
|
||||
*/
|
||||
if (IsSystemEventType(Event->EventData.EventType))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
struct kfd_ioctl_set_event_args args = {0};
|
||||
|
||||
args.event_id = Event->EventId;
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SET_EVENT, &args) == -1)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtResetEvent(HsaEvent *Event)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (!Event)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
/* Although the spec is doesn't say, don't allow system-defined events
|
||||
* to be signaled.
|
||||
*/
|
||||
if (IsSystemEventType(Event->EventData.EventType))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
struct kfd_ioctl_reset_event_args args = {0};
|
||||
|
||||
args.event_id = Event->EventId;
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_RESET_EVENT, &args) == -1)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtQueryEventState(HsaEvent *Event)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (!Event)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtWaitOnEvent(HsaEvent *Event,
|
||||
HSAuint32 Milliseconds)
|
||||
{
|
||||
return hsaKmtWaitOnEvent_Ext(Event, Milliseconds, NULL);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtWaitOnEvent_Ext(HsaEvent *Event,
|
||||
HSAuint32 Milliseconds, uint64_t *event_age)
|
||||
{
|
||||
if (!Event)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
return hsaKmtWaitOnMultipleEvents_Ext(&Event, 1, true, Milliseconds, event_age);
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS get_mem_info_svm_api(uint64_t address, uint32_t gpu_id)
|
||||
{
|
||||
struct kfd_ioctl_svm_args *args;
|
||||
uint32_t node_id = 0;
|
||||
HSAuint32 s_attr;
|
||||
HSAuint32 i;
|
||||
HSA_SVM_ATTRIBUTE attrs[] = {
|
||||
{HSA_SVM_ATTR_PREFERRED_LOC, 0},
|
||||
{HSA_SVM_ATTR_PREFETCH_LOC, 0},
|
||||
{HSA_SVM_ATTR_ACCESS, gpu_id},
|
||||
{HSA_SVM_ATTR_SET_FLAGS, 0},
|
||||
};
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(5);
|
||||
|
||||
s_attr = sizeof(attrs);
|
||||
args = alloca(sizeof(*args) + s_attr);
|
||||
args->start_addr = address;
|
||||
args->size = PAGE_SIZE;
|
||||
args->op = KFD_IOCTL_SVM_OP_GET_ATTR;
|
||||
args->nattr = s_attr / sizeof(*attrs);
|
||||
memcpy(args->attrs, attrs, s_attr);
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SVM + (s_attr << _IOC_SIZESHIFT), args)) {
|
||||
pr_debug("op get range attrs failed %s\n", strerror(errno));
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
pr_err("GPU address 0x%lx, is Unified memory\n", address);
|
||||
for (i = 0; i < args->nattr; i++) {
|
||||
if (args->attrs[i].value == KFD_IOCTL_SVM_LOCATION_SYSMEM ||
|
||||
args->attrs[i].value == KFD_IOCTL_SVM_LOCATION_UNDEFINED)
|
||||
node_id = args->attrs[i].value;
|
||||
else
|
||||
hsakmt_gpuid_to_nodeid(args->attrs[i].value, &node_id);
|
||||
switch (args->attrs[i].type) {
|
||||
case KFD_IOCTL_SVM_ATTR_PREFERRED_LOC:
|
||||
pr_err("Preferred location for address 0x%lx is Node id %d\n",
|
||||
address, node_id);
|
||||
break;
|
||||
case KFD_IOCTL_SVM_ATTR_PREFETCH_LOC:
|
||||
pr_err("Prefetch location for address 0x%lx is Node id %d\n",
|
||||
address, node_id);
|
||||
break;
|
||||
case KFD_IOCTL_SVM_ATTR_ACCESS:
|
||||
pr_err("Node id %d has access to address 0x%lx\n",
|
||||
node_id, address);
|
||||
break;
|
||||
case KFD_IOCTL_SVM_ATTR_ACCESS_IN_PLACE:
|
||||
pr_err("Node id %d has access in place to address 0x%lx\n",
|
||||
node_id, address);
|
||||
break;
|
||||
case KFD_IOCTL_SVM_ATTR_NO_ACCESS:
|
||||
pr_err("Node id %d has no access to address 0x%lx\n",
|
||||
node_id, address);
|
||||
break;
|
||||
case KFD_IOCTL_SVM_ATTR_SET_FLAGS:
|
||||
if (args->attrs[i].value & KFD_IOCTL_SVM_FLAG_COHERENT)
|
||||
pr_err("Fine grained coherency between devices\n");
|
||||
if (args->attrs[i].value & KFD_IOCTL_SVM_FLAG_GPU_RO)
|
||||
pr_err("Read only\n");
|
||||
if (args->attrs[i].value & KFD_IOCTL_SVM_FLAG_GPU_EXEC)
|
||||
pr_err("GPU exec allowed\n");
|
||||
if (args->attrs[i].value & KFD_IOCTL_SVM_FLAG_GPU_ALWAYS_MAPPED)
|
||||
pr_err("GPU always mapped\n");
|
||||
if (args->attrs[i].value & KFD_IOCTL_SVM_FLAG_EXT_COHERENT)
|
||||
pr_err("Extended-scope fine grained coherency between devices\n");
|
||||
break;
|
||||
default:
|
||||
pr_debug("get invalid attr type 0x%x\n", args->attrs[i].type);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
//Analysis memory exception data, print debug messages
|
||||
static void analysis_memory_exception(struct kfd_hsa_memory_exception_data *
|
||||
memory_exception_data)
|
||||
{
|
||||
HSAKMT_STATUS ret;
|
||||
HsaPointerInfo info;
|
||||
const uint64_t addr = memory_exception_data->va;
|
||||
uint32_t node_id = 0;
|
||||
unsigned int i;
|
||||
|
||||
hsakmt_gpuid_to_nodeid(memory_exception_data->gpu_id, &node_id);
|
||||
pr_err("Memory exception on virtual address 0x%lx, ", addr);
|
||||
pr_err("node id %d : ", node_id);
|
||||
if (memory_exception_data->failure.NotPresent)
|
||||
pr_err("Page not present\n");
|
||||
else if (memory_exception_data->failure.ReadOnly)
|
||||
pr_err("Writing to readonly page\n");
|
||||
else if (memory_exception_data->failure.NoExecute)
|
||||
pr_err("Execute to none-executable page\n");
|
||||
|
||||
ret = hsakmt_fmm_get_mem_info((const void *)addr, &info);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
ret = get_mem_info_svm_api(addr, memory_exception_data->gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
pr_err("Address does not belong to a known buffer\n");
|
||||
return;
|
||||
}
|
||||
|
||||
pr_err("GPU address 0x%lx, node id %d, size in byte 0x%lx\n",
|
||||
info.GPUAddress, info.Node, info.SizeInBytes);
|
||||
switch (info.Type) {
|
||||
case HSA_POINTER_REGISTERED_SHARED:
|
||||
pr_err("Memory is registered shared buffer (IPC)\n");
|
||||
break;
|
||||
case HSA_POINTER_REGISTERED_GRAPHICS:
|
||||
pr_err("Memory is registered graphics buffer\n");
|
||||
break;
|
||||
case HSA_POINTER_REGISTERED_USER:
|
||||
pr_err("Memory is registered user pointer\n");
|
||||
pr_err("CPU address of the memory is %p\n", info.CPUAddress);
|
||||
break;
|
||||
case HSA_POINTER_ALLOCATED:
|
||||
pr_err("Memory is allocated using hsaKmtAllocMemory\n");
|
||||
pr_err("CPU address of the memory is %p\n", info.CPUAddress);
|
||||
break;
|
||||
case HSA_POINTER_RESERVED_ADDR:
|
||||
pr_err("Memory is allocated by OnlyAddress mode\n");
|
||||
break;
|
||||
default:
|
||||
pr_err("Invalid memory type %d\n", info.Type);
|
||||
break;
|
||||
}
|
||||
|
||||
if (info.RegisteredNodes) {
|
||||
pr_err("Memory is registered to node id: ");
|
||||
for (i = 0; i < info.NRegisteredNodes; i++)
|
||||
pr_err("%d ", info.RegisteredNodes[i]);
|
||||
pr_err("\n");
|
||||
}
|
||||
if (info.MappedNodes) {
|
||||
pr_err("Memory is mapped to node id: ");
|
||||
for (i = 0; i < info.NMappedNodes; i++)
|
||||
pr_err("%d ", info.MappedNodes[i]);
|
||||
pr_err("\n");
|
||||
}
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtWaitOnMultipleEvents(HsaEvent *Events[],
|
||||
HSAuint32 NumEvents,
|
||||
bool WaitOnAll,
|
||||
HSAuint32 Milliseconds)
|
||||
{
|
||||
return hsaKmtWaitOnMultipleEvents_Ext(Events, NumEvents, WaitOnAll, Milliseconds, NULL);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtWaitOnMultipleEvents_Ext(HsaEvent *Events[],
|
||||
HSAuint32 NumEvents,
|
||||
bool WaitOnAll,
|
||||
HSAuint32 Milliseconds,
|
||||
uint64_t *event_age)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (!Events)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
struct kfd_event_data *event_data =
|
||||
calloc(NumEvents, sizeof(struct kfd_event_data));
|
||||
if (!event_data) {
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
}
|
||||
for (HSAuint32 i = 0; i < NumEvents; i++) {
|
||||
event_data[i].event_id = Events[i]->EventId;
|
||||
event_data[i].kfd_event_data_ext = (uint64_t)(uintptr_t)NULL;
|
||||
if (event_age && Events[i]->EventData.EventType == HSA_EVENTTYPE_SIGNAL)
|
||||
event_data[i].signal_event_data.last_event_age = event_age[i];
|
||||
}
|
||||
|
||||
struct kfd_ioctl_wait_events_args args = {0};
|
||||
|
||||
args.wait_for_all = WaitOnAll;
|
||||
args.timeout = Milliseconds;
|
||||
args.num_events = NumEvents;
|
||||
args.events_ptr = (uint64_t)(uintptr_t)event_data;
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_WAIT_EVENTS, &args) == -1)
|
||||
result = HSAKMT_STATUS_ERROR;
|
||||
else if (args.wait_result == KFD_IOC_WAIT_RESULT_TIMEOUT)
|
||||
result = HSAKMT_STATUS_WAIT_TIMEOUT;
|
||||
else {
|
||||
result = HSAKMT_STATUS_SUCCESS;
|
||||
for (HSAuint32 i = 0; i < NumEvents; i++) {
|
||||
if (Events[i]->EventData.EventType == HSA_EVENTTYPE_MEMORY &&
|
||||
event_data[i].memory_exception_data.gpu_id) {
|
||||
Events[i]->EventData.EventData.MemoryAccessFault.VirtualAddress = event_data[i].memory_exception_data.va;
|
||||
result = hsakmt_gpuid_to_nodeid(event_data[i].memory_exception_data.gpu_id, &Events[i]->EventData.EventData.MemoryAccessFault.NodeId);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
goto out;
|
||||
Events[i]->EventData.EventData.MemoryAccessFault.Failure.NotPresent = event_data[i].memory_exception_data.failure.NotPresent;
|
||||
Events[i]->EventData.EventData.MemoryAccessFault.Failure.ReadOnly = event_data[i].memory_exception_data.failure.ReadOnly;
|
||||
Events[i]->EventData.EventData.MemoryAccessFault.Failure.NoExecute = event_data[i].memory_exception_data.failure.NoExecute;
|
||||
Events[i]->EventData.EventData.MemoryAccessFault.Failure.Imprecise = event_data[i].memory_exception_data.failure.imprecise;
|
||||
Events[i]->EventData.EventData.MemoryAccessFault.Failure.ErrorType = event_data[i].memory_exception_data.ErrorType;
|
||||
Events[i]->EventData.EventData.MemoryAccessFault.Failure.ECC =
|
||||
((event_data[i].memory_exception_data.ErrorType == 1) || (event_data[i].memory_exception_data.ErrorType == 2)) ? 1 : 0;
|
||||
Events[i]->EventData.EventData.MemoryAccessFault.Flags = HSA_EVENTID_MEMORY_FATAL_PROCESS;
|
||||
analysis_memory_exception(&event_data[i].memory_exception_data);
|
||||
} else if (Events[i]->EventData.EventType == HSA_EVENTTYPE_HW_EXCEPTION &&
|
||||
event_data[i].hw_exception_data.gpu_id) {
|
||||
|
||||
result = hsakmt_gpuid_to_nodeid(event_data[i].hw_exception_data.gpu_id, &Events[i]->EventData.EventData.HwException.NodeId);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
goto out;
|
||||
|
||||
Events[i]->EventData.EventData.HwException.ResetType = event_data[i].hw_exception_data.reset_type;
|
||||
Events[i]->EventData.EventData.HwException.ResetCause = event_data[i].hw_exception_data.reset_cause;
|
||||
Events[i]->EventData.EventData.HwException.MemoryLost = event_data[i].hw_exception_data.memory_lost;
|
||||
}
|
||||
}
|
||||
}
|
||||
out:
|
||||
|
||||
for (HSAuint32 i = 0; i < NumEvents; i++) {
|
||||
if (event_age && Events[i]->EventData.EventType == HSA_EVENTTYPE_SIGNAL)
|
||||
event_age[i] = event_data[i].signal_event_data.last_event_age;
|
||||
}
|
||||
|
||||
free(event_data);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtOpenSMI(HSAuint32 NodeId, int *fd)
|
||||
{
|
||||
struct kfd_ioctl_smi_events_args args;
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpuid;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] node %d\n", __func__, NodeId);
|
||||
|
||||
result = hsakmt_validate_nodeid(NodeId, &gpuid);
|
||||
if (result != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, NodeId);
|
||||
return result;
|
||||
}
|
||||
|
||||
args.gpuid = gpuid;
|
||||
result = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SMI_EVENTS, &args);
|
||||
if (result) {
|
||||
pr_debug("open SMI event fd failed %s\n", strerror(errno));
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
*fd = args.anon_fd;
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,106 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#ifndef FMM_H_
|
||||
#define FMM_H_
|
||||
|
||||
#include "hsakmt/hsakmttypes.h"
|
||||
#include <stddef.h>
|
||||
|
||||
typedef enum {
|
||||
FMM_FIRST_APERTURE_TYPE = 0,
|
||||
FMM_GPUVM = FMM_FIRST_APERTURE_TYPE,
|
||||
FMM_LDS,
|
||||
FMM_SCRATCH,
|
||||
FMM_SVM,
|
||||
FMM_MMIO,
|
||||
FMM_LAST_APERTURE_TYPE
|
||||
} aperture_type_e;
|
||||
|
||||
typedef struct {
|
||||
aperture_type_e app_type;
|
||||
uint64_t size;
|
||||
void *start_address;
|
||||
} aperture_properties_t;
|
||||
|
||||
HSAKMT_STATUS hsakmt_fmm_get_amdgpu_device_handle(uint32_t node_id, HsaAMDGPUDeviceHandle *DeviceHandle);
|
||||
HSAKMT_STATUS hsakmt_fmm_init_process_apertures(unsigned int NumNodes);
|
||||
void hsakmt_fmm_destroy_process_apertures(void);
|
||||
|
||||
/* Memory interface */
|
||||
void *hsakmt_fmm_allocate_scratch(uint32_t gpu_id, void *address, uint64_t MemorySizeInBytes);
|
||||
void *hsakmt_fmm_allocate_device(uint32_t gpu_id, uint32_t node_id, void *address,
|
||||
uint64_t MemorySizeInBytes, uint64_t alignment, HsaMemFlags flags);
|
||||
void *hsakmt_fmm_allocate_doorbell(uint32_t gpu_id, uint64_t MemorySizeInBytes, uint64_t doorbell_offset);
|
||||
void *hsakmt_fmm_allocate_host(uint32_t gpu_id, uint32_t node_id, void *address, uint64_t MemorySizeInBytes,
|
||||
uint64_t alignment, HsaMemFlags flags);
|
||||
void hsakmt_fmm_print(uint32_t node);
|
||||
HSAKMT_STATUS hsakmt_fmm_release(void *address);
|
||||
HSAKMT_STATUS hsakmt_fmm_map_to_gpu(void *address, uint64_t size, uint64_t *gpuvm_address);
|
||||
int hsakmt_fmm_unmap_from_gpu(void *address);
|
||||
bool hsakmt_fmm_get_handle(void *address, uint64_t *handle);
|
||||
HSAKMT_STATUS hsakmt_fmm_get_mem_info(const void *address, HsaPointerInfo *info);
|
||||
HSAKMT_STATUS hsakmt_fmm_set_mem_user_data(const void *mem, void *usr_data);
|
||||
#ifdef SANITIZER_AMDGPU
|
||||
HSAKMT_STATUS hsakmt_fmm_replace_asan_header_page(void* address);
|
||||
HSAKMT_STATUS hsakmt_fmm_return_asan_header_page(void* address);
|
||||
#endif
|
||||
|
||||
/* Topology interface*/
|
||||
HSAKMT_STATUS hsakmt_fmm_get_aperture_base_and_limit(aperture_type_e aperture_type, HSAuint32 gpu_id,
|
||||
HSAuint64 *aperture_base, HSAuint64 *aperture_limit);
|
||||
|
||||
HSAKMT_STATUS hsakmt_fmm_register_memory(void *address, uint64_t size_in_bytes,
|
||||
uint32_t *gpu_id_array,
|
||||
uint32_t gpu_id_array_size,
|
||||
bool coarse_grain,
|
||||
bool ext_coherent);
|
||||
HSAKMT_STATUS hsakmt_fmm_register_graphics_handle(HSAuint64 GraphicsResourceHandle,
|
||||
HsaGraphicsResourceInfo *GraphicsResourceInfo,
|
||||
uint32_t *gpu_id_array,
|
||||
uint32_t gpu_id_array_size,
|
||||
HSA_REGISTER_MEM_FLAGS RegisterFlags);
|
||||
HSAKMT_STATUS hsakmt_fmm_deregister_memory(void *address);
|
||||
HSAKMT_STATUS hsakmt_fmm_export_dma_buf_fd(void *MemoryAddress,
|
||||
HSAuint64 MemorySizeInBytes,
|
||||
int *DMABufFd,
|
||||
HSAuint64 *Offset);
|
||||
HSAKMT_STATUS hsakmt_fmm_share_memory(void *MemoryAddress,
|
||||
HSAuint64 SizeInBytes,
|
||||
HsaSharedMemoryHandle *SharedMemoryHandle);
|
||||
HSAKMT_STATUS hsakmt_fmm_register_shared_memory(const HsaSharedMemoryHandle *SharedMemoryHandle,
|
||||
HSAuint64 *SizeInBytes,
|
||||
void **MemoryAddress,
|
||||
uint32_t *gpu_id_array,
|
||||
uint32_t gpu_id_array_size);
|
||||
HSAKMT_STATUS hsakmt_fmm_map_to_gpu_nodes(void *address, uint64_t size,
|
||||
uint32_t *nodes_to_map, uint64_t num_of_nodes, uint64_t *gpuvm_address);
|
||||
|
||||
int hsakmt_open_drm_render_device(int minor);
|
||||
void *hsakmt_mmap_allocate_aligned(int prot, int flags, uint64_t size, uint64_t align,
|
||||
uint64_t guard_size, void *aper_base, void *aper_limit);
|
||||
|
||||
extern int (*hsakmt_fn_amdgpu_device_get_fd)(HsaAMDGPUDeviceHandle device_handle);
|
||||
#endif /* FMM_H_ */
|
||||
@@ -0,0 +1,42 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
|
||||
// HSAKMT global data
|
||||
|
||||
int hsakmt_kfd_fd = -1;
|
||||
unsigned long hsakmt_kfd_open_count;
|
||||
unsigned long hsakmt_system_properties_count;
|
||||
pthread_mutex_t hsakmt_mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
bool hsakmt_is_dgpu;
|
||||
|
||||
int hsakmt_page_size;
|
||||
int hsakmt_page_shift;
|
||||
|
||||
/* whether to check all dGPUs in the topology support SVM API */
|
||||
bool hsakmt_is_svm_api_supported;
|
||||
/* zfb is mainly used during emulation */
|
||||
int hsakmt_zfb_support;
|
||||
@@ -0,0 +1,823 @@
|
||||
/*
|
||||
* Copyright © 2025 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "hsakmt/hsakmtmodel.h"
|
||||
#include "libhsakmt.h"
|
||||
#include "hsakmt/hsakmttypes.h"
|
||||
#include "hsakmt/hsakmtmodeliface.h"
|
||||
#define _GNU_SOURCE
|
||||
#define __USE_GNU
|
||||
#include <assert.h>
|
||||
#include <errno.h>
|
||||
#include <inttypes.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/types.h>
|
||||
#include <unistd.h>
|
||||
#include <dlfcn.h>
|
||||
#include <sys/mman.h>
|
||||
#include <fcntl.h>
|
||||
|
||||
bool hsakmt_use_model;
|
||||
char *hsakmt_model_topology;
|
||||
|
||||
struct model_node
|
||||
{
|
||||
bool is_gpu;
|
||||
void *aperture;
|
||||
hsakmt_model_t *model;
|
||||
uint64_t doorbell_offset;
|
||||
uint64_t total_memory_size;
|
||||
uint64_t allocated_memory_size;
|
||||
};
|
||||
|
||||
struct model_event
|
||||
{
|
||||
uint32_t event_type;
|
||||
uint32_t auto_reset;
|
||||
uint64_t value;
|
||||
};
|
||||
|
||||
struct model_mem_data
|
||||
{
|
||||
uint64_t va_addr;
|
||||
uint64_t file_offset;
|
||||
uint64_t size;
|
||||
uint64_t mapped_nodes_bitmask;
|
||||
uint32_t flags;
|
||||
uint32_t node_id;
|
||||
};
|
||||
|
||||
struct model_queue
|
||||
{
|
||||
hsakmt_model_queue_t *queue;
|
||||
uint32_t node_id;
|
||||
};
|
||||
|
||||
#define MAX_MODEL_QUEUES 128
|
||||
// Use a 256GB aperture for the model.
|
||||
#define MODEL_APERTURE_SIZE (1llu << 38)
|
||||
static void *model_mmio_page;
|
||||
static pthread_mutex_t model_ioctl_mutex = PTHREAD_MUTEX_INITIALIZER;
|
||||
static unsigned model_event_limit;
|
||||
static uint64_t *model_event_bitmap;
|
||||
static struct model_event *model_events;
|
||||
static pthread_cond_t model_event_condvar;
|
||||
static void *model_library;
|
||||
static const struct hsakmt_model_functions *model_functions;
|
||||
static uint64_t model_memfd_size;
|
||||
static uint64_t model_num_nodes;
|
||||
static struct model_node *model_nodes;
|
||||
static struct model_queue model_queues[MAX_MODEL_QUEUES];
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtModelEnabled(bool* enable)
|
||||
{
|
||||
*enable = hsakmt_use_model;
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
void model_init_env_vars(void)
|
||||
{
|
||||
/* Check whether to use a model instead of real hardware */
|
||||
hsakmt_model_topology = getenv("HSA_MODEL_TOPOLOGY");
|
||||
if (hsakmt_model_topology)
|
||||
hsakmt_use_model = true;
|
||||
if (hsakmt_use_model)
|
||||
{
|
||||
/* Backing memory file is used to stand in for the kfd_fd,
|
||||
* which is needed early, so create it already.
|
||||
*
|
||||
* For old systems without memfd_create, or if the user prefers,
|
||||
* we create a regular backing file. Prefer to use memfd_create
|
||||
* by default where possible.
|
||||
*/
|
||||
int fd = -1;
|
||||
const char *fname = getenv("HSA_MODEL_MEMFILE");
|
||||
if (fname)
|
||||
{
|
||||
fprintf(stderr, "model: use memory backing file given in HSA_MODEL_MEMFILE: %s\n", fname);
|
||||
|
||||
fd = open(fname, O_CREAT | O_EXCL | O_CLOEXEC | O_RDWR, S_IRUSR | S_IWUSR);
|
||||
if (fd < 0)
|
||||
{
|
||||
perror("model: failed to create backing file");
|
||||
abort();
|
||||
}
|
||||
|
||||
unlink(fname);
|
||||
}
|
||||
|
||||
if (fd < 0)
|
||||
{
|
||||
#ifdef HAVE_MEMFD_CREATE
|
||||
fd = memfd_create("hsakmt_model", MFD_CLOEXEC);
|
||||
if (fd < 0)
|
||||
{
|
||||
fprintf(stderr, "model: Failed to create memfd\n");
|
||||
abort();
|
||||
}
|
||||
#else
|
||||
fprintf(stderr, "model: built without memfd support\n"
|
||||
"model: set HSA_MODEL_MEMFILE to path of a backing file\n");
|
||||
abort();
|
||||
#endif
|
||||
}
|
||||
assert(hsakmt_kfd_fd < 0);
|
||||
hsakmt_kfd_fd = fd;
|
||||
pthread_condattr_t condattr;
|
||||
pthread_condattr_init(&condattr);
|
||||
pthread_condattr_setclock(&condattr, CLOCK_MONOTONIC);
|
||||
pthread_cond_init(&model_event_condvar, &condattr);
|
||||
pthread_condattr_destroy(&condattr);
|
||||
const char *libname = getenv("HSA_MODEL_LIB");
|
||||
if (!libname)
|
||||
{
|
||||
fprintf(stderr, "model: HSA_MODEL_LIB environment variable must be set to FFM .so\n");
|
||||
abort();
|
||||
}
|
||||
// model_library = dlmopen(LM_ID_NEWLM, libname, RTLD_NOW);
|
||||
model_library = dlopen(libname, RTLD_NOW | RTLD_LOCAL);
|
||||
if (!model_library)
|
||||
{
|
||||
fprintf(stderr, "model: failed to load %s: %s\n", libname, dlerror());
|
||||
abort();
|
||||
}
|
||||
get_hsakmt_model_functions_t getter = dlsym(model_library, "get_hsakmt_model_functions");
|
||||
if (!getter)
|
||||
{
|
||||
fprintf(stderr, "model: Failed to get hsakmt_model_functions\n");
|
||||
abort();
|
||||
}
|
||||
model_functions = getter();
|
||||
if (model_functions->version_major != HSAKMT_MODEL_INTERFACE_VERSION_MAJOR ||
|
||||
model_functions->version_minor < HSAKMT_MODEL_INTERFACE_VERSION_MINOR)
|
||||
{
|
||||
fprintf(stderr, "model: Model has interface version %u.%u, need version %u.%u\n",
|
||||
model_functions->version_major, model_functions->version_minor,
|
||||
HSAKMT_MODEL_INTERFACE_VERSION_MAJOR, HSAKMT_MODEL_INTERFACE_VERSION_MINOR);
|
||||
abort();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static uint64_t allocate_from_memfd(uint64_t size, uint64_t align)
|
||||
{
|
||||
if (!align)
|
||||
align = 4096;
|
||||
assert(POWER_OF_2(align)); /* must be power of two */
|
||||
assert(align >= 4096);
|
||||
size = (size + 4095) & ~4095;
|
||||
model_memfd_size = (model_memfd_size + align - 1) & ~(align - 1);
|
||||
uint64_t offset = model_memfd_size;
|
||||
model_memfd_size += size;
|
||||
int ret = ftruncate(hsakmt_kfd_fd, model_memfd_size);
|
||||
if (ret < 0)
|
||||
{
|
||||
fprintf(stderr, "model: ftruncate on memfd failed\n");
|
||||
abort();
|
||||
}
|
||||
return offset;
|
||||
}
|
||||
static uint64_t get_sysfs_mem_bank_size(unsigned node_id, unsigned mem_id)
|
||||
{
|
||||
char prop_name[256];
|
||||
char path[256];
|
||||
snprintf(path, sizeof(path), "%s/nodes/%u/mem_banks/%u/properties",
|
||||
hsakmt_model_topology, node_id, mem_id);
|
||||
FILE *f = fopen(path, "r");
|
||||
if (!f)
|
||||
{
|
||||
fprintf(stderr, "model: Failed to open %s\n", path);
|
||||
abort();
|
||||
}
|
||||
uint64_t prop_val;
|
||||
while (fscanf(f, "%s %" PRIu64 "\n", prop_name, &prop_val) == 2)
|
||||
{
|
||||
if (!strcmp(prop_name, "size_in_bytes"))
|
||||
{
|
||||
fclose(f);
|
||||
return prop_val;
|
||||
}
|
||||
}
|
||||
fprintf(stderr, "model: Missing size_in_bytes in %s\n", path);
|
||||
abort();
|
||||
}
|
||||
|
||||
static void model_set_event(void *data, unsigned event_id)
|
||||
{
|
||||
if (!event_id)
|
||||
return;
|
||||
|
||||
if (event_id > model_event_limit)
|
||||
{
|
||||
fprintf(stderr, "model_set_event: event_id = %u out of bounds\n",
|
||||
event_id);
|
||||
abort();
|
||||
}
|
||||
|
||||
unsigned slot = event_id - 1;
|
||||
|
||||
if (!((model_event_bitmap[slot / 64] >> (slot % 64)) & 1))
|
||||
{
|
||||
fprintf(stderr, "model_set_event: event_id = %u is not allocated\n",
|
||||
event_id);
|
||||
abort();
|
||||
}
|
||||
|
||||
struct model_event *event = &model_events[slot];
|
||||
if (event->event_type == HSA_EVENTTYPE_SIGNAL)
|
||||
{
|
||||
assert(model_events[slot].value <= 1);
|
||||
model_events[slot].value = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "model: Unimplemented event type\n");
|
||||
abort();
|
||||
}
|
||||
|
||||
pthread_cond_broadcast(&model_event_condvar);
|
||||
}
|
||||
|
||||
void model_init(void)
|
||||
{
|
||||
if (!hsakmt_use_model)
|
||||
return;
|
||||
HSAKMT_STATUS result;
|
||||
HsaSystemProperties props;
|
||||
/* Read the topology to determine nodes. */
|
||||
result = hsakmt_topology_sysfs_get_system_props(&props);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
{
|
||||
fprintf(stderr, "model: Failed to parse topology\n");
|
||||
abort();
|
||||
}
|
||||
model_nodes = calloc(props.NumNodes, sizeof(*model_nodes));
|
||||
if (!model_nodes)
|
||||
abort();
|
||||
model_num_nodes = props.NumNodes;
|
||||
for (unsigned node_id = 0; node_id < props.NumNodes; node_id++)
|
||||
{
|
||||
HsaNodeProperties node_props;
|
||||
result = hsakmt_topology_get_node_props(node_id, &node_props);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
{
|
||||
fprintf(stderr, "model: Failed to get node %u properties\n", node_id);
|
||||
abort();
|
||||
}
|
||||
if (node_props.KFDGpuID == 0)
|
||||
continue;
|
||||
if (node_props.KFDGpuID != node_id + 1)
|
||||
{
|
||||
fprintf(stderr,
|
||||
"model: Node %u has KFD GPU ID %u, but should be %u."
|
||||
" Please change the gpu_id file.\n",
|
||||
node_id, node_props.KFDGpuID, node_id + 1);
|
||||
abort();
|
||||
}
|
||||
model_nodes[node_id].is_gpu = true;
|
||||
/* Reserve the VA space for the aperture, but don't fill it with pages. */
|
||||
model_nodes[node_id].aperture =
|
||||
mmap(NULL, MODEL_APERTURE_SIZE, PROT_NONE,
|
||||
MAP_PRIVATE | MAP_NORESERVE | MAP_ANONYMOUS, -1, 0);
|
||||
pr_debug("Modeling Creating Memory Aperture: %p\n", model_nodes[node_id].aperture);
|
||||
if (model_nodes[node_id].aperture == MAP_FAILED)
|
||||
{
|
||||
fprintf(stderr, "model: Failed to reserve aperture via mmap\n");
|
||||
abort();
|
||||
}
|
||||
/* Create the doorbell region */
|
||||
model_nodes[node_id].doorbell_offset = allocate_from_memfd(8192, 8192);
|
||||
for (unsigned mem_id = 0; mem_id < node_props.NumMemoryBanks; ++mem_id)
|
||||
{
|
||||
model_nodes[node_id].total_memory_size += get_sysfs_mem_bank_size(node_id, mem_id);
|
||||
}
|
||||
/* Create the model */
|
||||
// TODO: Move this into a separate thread
|
||||
model_nodes[node_id].model = model_functions->create();
|
||||
if (!model_nodes[node_id].model)
|
||||
{
|
||||
fprintf(stderr, "model: Failed to create model\n");
|
||||
abort();
|
||||
}
|
||||
model_functions->set_global_aperture(model_nodes[node_id].model,
|
||||
model_nodes[node_id].aperture,
|
||||
MODEL_APERTURE_SIZE);
|
||||
|
||||
model_functions->set_set_event(model_nodes[node_id].model, model_set_event, NULL);
|
||||
}
|
||||
}
|
||||
void model_set_mmio_page(void *ptr)
|
||||
{
|
||||
assert(!model_mmio_page);
|
||||
model_mmio_page = ptr;
|
||||
}
|
||||
void model_set_event_page(void *ptr, unsigned event_limit)
|
||||
{
|
||||
// TODO: Fully understand what's happening with this page and the event limit.
|
||||
// ROCR-Runtime allocates a pool of 4096 events, but also a handful or so
|
||||
// of additional events, which blows through the event_limit of 4096
|
||||
// that is passed here. And it seems that not using the page at all
|
||||
// is supported?
|
||||
assert(!model_event_limit);
|
||||
assert(event_limit % 64 == 0);
|
||||
event_limit *= 2;
|
||||
model_event_limit = event_limit;
|
||||
model_event_bitmap = calloc(event_limit / 64, 8);
|
||||
model_events = calloc(event_limit, sizeof(*model_events));
|
||||
}
|
||||
/* Model implementation of KFD ioctl. */
|
||||
|
||||
static int model_kfd_ioctl_locked(unsigned long request, void *arg)
|
||||
{
|
||||
assert(_IOC_TYPE(request) == AMDKFD_IOCTL_BASE);
|
||||
if (_IOC_NR(request) == 0x20)
|
||||
{
|
||||
// This is AMDKFD_IOC_SVM. It is defined / used in an unusual way.
|
||||
struct kfd_ioctl_svm_args *args = arg;
|
||||
if (args->op == KFD_IOCTL_SVM_OP_SET_ATTR)
|
||||
{
|
||||
// todo?
|
||||
return 0;
|
||||
}
|
||||
fprintf(stderr, "model: Unimplemented SVM op\n");
|
||||
abort();
|
||||
}
|
||||
switch (request)
|
||||
{
|
||||
case AMDKFD_IOC_GET_VERSION:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_GET_VERSION\n");
|
||||
struct kfd_ioctl_get_version_args *args = arg;
|
||||
args->major_version = 1;
|
||||
args->minor_version = 14;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_GET_PROCESS_APERTURES_NEW:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_GET_PROCESS_APERTURES_NEW\n");
|
||||
struct kfd_ioctl_get_process_apertures_new_args *args = arg;
|
||||
struct kfd_process_device_apertures *apertures =
|
||||
(void *)args->kfd_process_device_apertures_ptr;
|
||||
assert(args->num_of_nodes == model_num_nodes);
|
||||
for (unsigned node_id = 0; node_id < args->num_of_nodes; ++node_id)
|
||||
{
|
||||
memset(&apertures[node_id], 0, sizeof(apertures[node_id]));
|
||||
if (!model_nodes[node_id].is_gpu)
|
||||
continue;
|
||||
apertures[node_id].gpu_id = 1 + node_id;
|
||||
apertures[node_id].gpuvm_base = 0x4000llu;
|
||||
apertures[node_id].gpuvm_limit = MODEL_APERTURE_SIZE;
|
||||
apertures[node_id].lds_base = 0x4000000000000000llu; // 0x1000000000000?
|
||||
apertures[node_id].lds_limit = 0x40000000ffffffffllu;
|
||||
apertures[node_id].scratch_base = 0x5000000000000000llu; // 0x2000000000000?
|
||||
apertures[node_id].scratch_limit = 0x50000000ffffffffllu;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_SET_XNACK_MODE:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_SET_XNACK_MODE\n");
|
||||
// Don't support XNACK
|
||||
struct kfd_ioctl_set_xnack_mode_args *args = arg;
|
||||
if (args->xnack_enabled < 0)
|
||||
{
|
||||
args->xnack_enabled = 0;
|
||||
return 0;
|
||||
}
|
||||
errno = EPERM;
|
||||
return -1;
|
||||
}
|
||||
case AMDKFD_IOC_GET_CLOCK_COUNTERS:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_GET_CLOCK_COUNTERS\n");
|
||||
struct kfd_ioctl_get_clock_counters_args *args = arg;
|
||||
args->gpu_clock_counter = 0; // TODO
|
||||
args->cpu_clock_counter = 0;
|
||||
args->system_clock_counter = 0;
|
||||
args->system_clock_freq = 0;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_ACQUIRE_VM:
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_ACQUIRE_VM\n");
|
||||
return 0;
|
||||
case AMDKFD_IOC_SET_MEMORY_POLICY:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_SET_MEMORY_POLICY\n");
|
||||
// todo?
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_AVAILABLE_MEMORY:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_AVAILABLE_MEMORY\n");
|
||||
static const uint64_t minimum_reported = 128 * 1024 * 1024;
|
||||
struct kfd_ioctl_get_available_memory_args *args = arg;
|
||||
unsigned node_id = args->gpu_id - 1;
|
||||
struct model_node *node = &model_nodes[node_id];
|
||||
assert(node_id < model_num_nodes);
|
||||
if (node->allocated_memory_size + minimum_reported >= node->total_memory_size)
|
||||
args->available = minimum_reported;
|
||||
else
|
||||
args->available = node->total_memory_size - node->allocated_memory_size;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_ALLOC_MEMORY_OF_GPU:
|
||||
{
|
||||
// Expect an SVM style allocation: The memory is allocated on the host
|
||||
// side e.g. via mmap(), and this IOCTL "only" registers the memory
|
||||
// with the GPU. This is a no-op for us because we aren't a GPU.
|
||||
struct kfd_ioctl_alloc_memory_of_gpu_args *args = arg;
|
||||
unsigned node_id = args->gpu_id - 1;
|
||||
assert(node_id < model_num_nodes);
|
||||
assert(model_nodes[node_id].is_gpu);
|
||||
if (args->va_addr == 0)
|
||||
{
|
||||
fprintf(stderr, "model: Expect only SVM allocations?\n");
|
||||
abort();
|
||||
}
|
||||
if (args->size % PAGE_SIZE != 0)
|
||||
{
|
||||
fprintf(stderr, "model: Allocation size not a multiple of page size\n");
|
||||
abort();
|
||||
}
|
||||
if (args->flags & KFD_IOC_ALLOC_MEM_FLAGS_USERPTR)
|
||||
{
|
||||
fprintf(stderr, "model: userptr not supported\n");
|
||||
abort();
|
||||
}
|
||||
struct model_mem_data *mem_data = calloc(1, sizeof(*mem_data));
|
||||
if (!mem_data)
|
||||
abort();
|
||||
mem_data->va_addr = args->va_addr;
|
||||
mem_data->size = args->size;
|
||||
mem_data->flags = args->flags;
|
||||
mem_data->node_id = node_id;
|
||||
if (args->flags & KFD_IOC_ALLOC_MEM_FLAGS_DOORBELL)
|
||||
{
|
||||
assert(args->size == 8192);
|
||||
mem_data->file_offset = model_nodes[node_id].doorbell_offset;
|
||||
}
|
||||
else
|
||||
{
|
||||
mem_data->file_offset = allocate_from_memfd(args->size, 0);
|
||||
}
|
||||
args->handle = (__u64)mem_data;
|
||||
args->mmap_offset = mem_data->file_offset;
|
||||
model_nodes[node_id].allocated_memory_size += args->size;
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_ALLOC_MEMORY_OF_GPU: VA: %lx : Size: %lu, Flags: %x\n", mem_data->va_addr, mem_data->size, mem_data->flags);
|
||||
model_functions->alloced_memory(model_nodes[node_id].model, (uint64_t *)mem_data->va_addr, mem_data->size, mem_data->flags);
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_FREE_MEMORY_OF_GPU:
|
||||
{
|
||||
struct kfd_ioctl_free_memory_of_gpu_args *args = arg;
|
||||
struct model_mem_data *mem_data = (void *)args->handle;
|
||||
assert(!mem_data->mapped_nodes_bitmask);
|
||||
// Free the memory by punching a hole into the underlying memfd.
|
||||
//
|
||||
// Ideally, we'd also remember holes in the file and re-use them for
|
||||
// allocations to avoid the file size from growing indefinitely. It's
|
||||
// unclear whether the current implementation causes kernel data
|
||||
// structures to grow. But in practice, it almost certainly never
|
||||
// matters.
|
||||
int ret = fallocate(hsakmt_kfd_fd, FALLOC_FL_PUNCH_HOLE | FALLOC_FL_KEEP_SIZE,
|
||||
mem_data->file_offset, mem_data->size);
|
||||
if (ret != 0)
|
||||
{
|
||||
perror("model: failed to punch hole in memfd");
|
||||
abort();
|
||||
}
|
||||
model_nodes[mem_data->node_id].allocated_memory_size -= mem_data->size;
|
||||
model_functions->freed_memory(model_nodes[mem_data->node_id].model, (uint64_t *)mem_data->va_addr, mem_data->size);
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_FREE_MEMORY_OF_GPU: VA: %lx : Size: %lu, Flags: %x\n", mem_data->va_addr, mem_data->size, mem_data->flags);
|
||||
free(mem_data);
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_MAP_MEMORY_TO_GPU:
|
||||
{
|
||||
struct kfd_ioctl_map_memory_to_gpu_args *args = arg;
|
||||
struct model_mem_data *mem_data = (void *)args->handle;
|
||||
while (args->n_success < args->n_devices)
|
||||
{
|
||||
uint32_t gpu_id = ((uint32_t *)args->device_ids_array_ptr)[args->n_success];
|
||||
uint32_t node_id = gpu_id - 1;
|
||||
assert(node_id < model_num_nodes);
|
||||
if (mem_data->mapped_nodes_bitmask & (1llu << node_id))
|
||||
{
|
||||
fprintf(stderr, "model: Already mapped\n");
|
||||
abort();
|
||||
}
|
||||
assert(model_nodes[node_id].aperture);
|
||||
unsigned prot = PROT_READ;
|
||||
if (mem_data->flags & KFD_IOC_ALLOC_MEM_FLAGS_WRITABLE)
|
||||
prot |= PROT_WRITE;
|
||||
// TODO: Mark *shader*-executable memory?
|
||||
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_MAP_MEMORY_TO_GPU: VA: %lx : Size: %lu, Flags: %x\n", mem_data->va_addr, mem_data->size, mem_data->flags);
|
||||
void *ret = mmap(VOID_PTR_ADD(model_nodes[node_id].aperture, mem_data->va_addr),
|
||||
mem_data->size, prot,
|
||||
MAP_SHARED | MAP_FIXED, hsakmt_kfd_fd, mem_data->file_offset);
|
||||
if (ret == MAP_FAILED)
|
||||
{
|
||||
fprintf(stderr, "model: mmap failed\n");
|
||||
abort();
|
||||
}
|
||||
mem_data->mapped_nodes_bitmask |= (1llu << node_id);
|
||||
args->n_success++;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_UNMAP_MEMORY_FROM_GPU:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_UNMAP_MEMORY_FROM_GPU\n");
|
||||
struct kfd_ioctl_unmap_memory_from_gpu_args *args = arg;
|
||||
struct model_mem_data *mem_data = (void *)args->handle;
|
||||
while (args->n_success < args->n_devices)
|
||||
{
|
||||
uint32_t gpu_id = ((uint32_t *)args->device_ids_array_ptr)[args->n_success];
|
||||
uint32_t node_id = gpu_id - 1;
|
||||
assert(node_id < model_num_nodes);
|
||||
if (!(mem_data->mapped_nodes_bitmask & (1llu << node_id)))
|
||||
{
|
||||
fprintf(stderr, "model: Not mapped\n");
|
||||
abort();
|
||||
}
|
||||
assert(model_nodes[node_id].aperture);
|
||||
/* Overwrite the mapping with an empty mapping to keep
|
||||
* it reserved. */
|
||||
void *ret = mmap(VOID_PTR_ADD(model_nodes[node_id].aperture, mem_data->va_addr),
|
||||
mem_data->size, PROT_NONE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED | MAP_NORESERVE, -1, 0);
|
||||
if (ret == MAP_FAILED)
|
||||
{
|
||||
perror("model: unmap failed");
|
||||
abort();
|
||||
}
|
||||
mem_data->mapped_nodes_bitmask &= ~(1llu << node_id);
|
||||
args->n_success++;
|
||||
}
|
||||
args->n_success = args->n_devices;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_CREATE_EVENT:
|
||||
{
|
||||
struct kfd_ioctl_create_event_args *args = arg;
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_CREATE_EVENT: %u\n", args->event_type);
|
||||
// Find a free slot
|
||||
unsigned i;
|
||||
for (i = 0; i < model_event_limit; i += 64)
|
||||
{
|
||||
uint64_t bitmap = model_event_bitmap[i / 64];
|
||||
if (bitmap == ~(uint64_t)0)
|
||||
continue;
|
||||
i += ffsll(~bitmap) - 1;
|
||||
break;
|
||||
}
|
||||
if (i >= model_event_limit)
|
||||
{
|
||||
fprintf(stderr, "model: Ran out of event slots. Should be an application error.\n");
|
||||
abort();
|
||||
}
|
||||
// Allocate the signal
|
||||
model_event_bitmap[i / 64] |= (uint64_t)1 << (i % 64);
|
||||
model_events[i].event_type = args->event_type;
|
||||
model_events[i].auto_reset = args->auto_reset;
|
||||
model_events[i].value = 0;
|
||||
args->event_trigger_data = 0xbadf001; // ???
|
||||
args->event_id = 1 + i;
|
||||
args->event_slot_index = ~0;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_WAIT_EVENTS:
|
||||
{
|
||||
struct kfd_ioctl_wait_events_args *args = arg;
|
||||
struct kfd_event_data *events = (void *)args->events_ptr;
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_WAIT_EVENTS: %u\n", args->num_events);
|
||||
bool have_timeout = args->timeout != 0xffffffffu;
|
||||
bool hit_timeout = false;
|
||||
struct timespec timeout;
|
||||
if (have_timeout)
|
||||
{
|
||||
clock_gettime(CLOCK_MONOTONIC, &timeout);
|
||||
timeout.tv_sec += args->timeout / 1000;
|
||||
timeout.tv_nsec += (args->timeout % 1000) * 1000000;
|
||||
if (timeout.tv_nsec > 1000000000)
|
||||
{
|
||||
timeout.tv_nsec -= 1000000000;
|
||||
timeout.tv_sec++;
|
||||
}
|
||||
}
|
||||
for (;;)
|
||||
{
|
||||
bool final_ready = args->wait_for_all;
|
||||
for (unsigned i = 0; i < args->num_events; ++i)
|
||||
{
|
||||
unsigned slot = events[i].event_id - 1;
|
||||
struct model_event *event = &model_events[slot];
|
||||
bool this_ready = false;
|
||||
if (event->event_type == HSA_EVENTTYPE_SIGNAL)
|
||||
{
|
||||
uint64_t current_age = event->value;
|
||||
uint64_t target_age = events[i].signal_event_data.last_event_age;
|
||||
this_ready = current_age >= target_age;
|
||||
}
|
||||
else if (event->event_type == HSA_EVENTTYPE_HW_EXCEPTION ||
|
||||
event->event_type == HSA_EVENTTYPE_NODECHANGE ||
|
||||
event->event_type == HSA_EVENTTYPE_DEVICESTATECHANGE ||
|
||||
event->event_type == HSA_EVENTTYPE_HW_EXCEPTION ||
|
||||
event->event_type == HSA_EVENTTYPE_DEBUG_EVENT ||
|
||||
event->event_type == HSA_EVENTTYPE_PROFILE_EVENT ||
|
||||
event->event_type == HSA_EVENTTYPE_MEMORY)
|
||||
{
|
||||
// These never happen in the model
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "model: Unimplemented event type\n");
|
||||
abort();
|
||||
}
|
||||
if (final_ready != this_ready)
|
||||
{
|
||||
final_ready = this_ready;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (final_ready)
|
||||
break;
|
||||
if (have_timeout)
|
||||
{
|
||||
int ret = pthread_cond_timedwait(
|
||||
&model_event_condvar, &model_ioctl_mutex, &timeout);
|
||||
if (ret == ETIMEDOUT)
|
||||
{
|
||||
hit_timeout = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
pthread_cond_wait(&model_event_condvar, &model_ioctl_mutex);
|
||||
}
|
||||
}
|
||||
/* Record most recent event ages and perform auto reset. */
|
||||
for (unsigned i = 0; i < args->num_events; ++i)
|
||||
{
|
||||
unsigned slot = events[i].event_id - 1;
|
||||
struct model_event *event = &model_events[slot];
|
||||
if (event->event_type == HSA_EVENTTYPE_SIGNAL)
|
||||
{
|
||||
uint64_t last_age = event->value;
|
||||
if (event->auto_reset && last_age >= events[i].signal_event_data.last_event_age)
|
||||
event->value = 0;
|
||||
events[i].signal_event_data.last_event_age = last_age;
|
||||
}
|
||||
}
|
||||
args->wait_result = hit_timeout ? KFD_IOC_WAIT_RESULT_TIMEOUT
|
||||
: KFD_IOC_WAIT_RESULT_COMPLETE;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_SET_EVENT:
|
||||
{
|
||||
struct kfd_ioctl_set_event_args *args = arg;
|
||||
model_set_event(NULL, args->event_id);
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_RESET_EVENT:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_RESET_EVENT\n");
|
||||
struct kfd_ioctl_reset_event_args *args = arg;
|
||||
unsigned slot = args->event_id - 1;
|
||||
struct model_event *event = &model_events[slot];
|
||||
if (event->event_type == HSA_EVENTTYPE_SIGNAL)
|
||||
{
|
||||
model_events[slot].value = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(stderr, "model: Unimplemented event type\n");
|
||||
abort();
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_DESTROY_EVENT:
|
||||
{
|
||||
struct kfd_ioctl_destroy_event_args *args = arg;
|
||||
unsigned i = args->event_id - 1;
|
||||
if (i >= model_event_limit || !(model_event_bitmap[i / 64] & ((uint64_t)1 << (i % 64))))
|
||||
{
|
||||
fprintf(stderr, "model: trying to destroy an event that doesn't exist.\n");
|
||||
abort();
|
||||
}
|
||||
memset(&model_events[i], 0, sizeof(model_events[i]));
|
||||
model_event_bitmap[i / 64] &= ~((uint64_t)1 << (i % 64));
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_CREATE_QUEUE:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_CREATE_QUEUE\n");
|
||||
struct kfd_ioctl_create_queue_args *args = arg;
|
||||
unsigned node_id = args->gpu_id - 1;
|
||||
assert(node_id < model_num_nodes);
|
||||
assert(model_nodes[node_id].model);
|
||||
const bool supported_queue_type = args->queue_type == KFD_IOC_QUEUE_TYPE_COMPUTE_AQL ||
|
||||
args->queue_type == KFD_IOC_QUEUE_TYPE_SDMA;
|
||||
if (!supported_queue_type)
|
||||
{
|
||||
fprintf(stderr, "model: Unsupported queue type\n");
|
||||
abort();
|
||||
}
|
||||
unsigned queue_id = 0;
|
||||
while (queue_id < MAX_MODEL_QUEUES && model_queues[queue_id].queue)
|
||||
queue_id++;
|
||||
if (queue_id >= MAX_MODEL_QUEUES)
|
||||
{
|
||||
fprintf(stderr, "model: too many queues\n");
|
||||
abort();
|
||||
}
|
||||
struct hsakmt_model_queue_info info = {0};
|
||||
info.ring_base_address = args->ring_base_address;
|
||||
info.ring_size = args->ring_size;
|
||||
info.write_pointer_address = args->write_pointer_address;
|
||||
info.read_pointer_address = args->read_pointer_address;
|
||||
info.queue_type = args->queue_type;
|
||||
model_queues[queue_id].queue =
|
||||
model_functions->register_queue(model_nodes[node_id].model, &info);
|
||||
model_queues[queue_id].node_id = node_id;
|
||||
args->queue_id = queue_id;
|
||||
// Note that strictly speaking, this is the offset into the hsakmt_kfd_fd
|
||||
// file, not the DRM fd (but they are the same in our case).
|
||||
args->doorbell_offset = model_nodes[node_id].doorbell_offset + 8 * queue_id;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_DESTROY_QUEUE:
|
||||
{
|
||||
struct kfd_ioctl_destroy_queue_args *args = arg;
|
||||
if (args->queue_id >= MAX_MODEL_QUEUES || !model_queues[args->queue_id].queue)
|
||||
{
|
||||
fprintf(stderr, "model: trying to destroy a queue that doesn't exist\n");
|
||||
abort();
|
||||
}
|
||||
struct model_queue *queue = &model_queues[args->queue_id];
|
||||
// Older model versions simply leak the queue.
|
||||
if (model_functions->version_minor >= 3)
|
||||
model_functions->destroy_queue(model_nodes[queue->node_id].model, queue->queue);
|
||||
queue->queue = NULL;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_GET_TILE_CONFIG:
|
||||
{
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_GET_TILE_CONFIG\n");
|
||||
struct kfd_ioctl_get_tile_config_args *args = arg;
|
||||
args->gb_addr_config = 0x10000444;
|
||||
return 0;
|
||||
}
|
||||
case AMDKFD_IOC_SET_SCRATCH_BACKING_VA:
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_SET_SCRATCH_BACKING_VA\n");
|
||||
// no-op -- scratch allocations are communicated via amd_queue_s
|
||||
return 0;
|
||||
case AMDKFD_IOC_RUNTIME_ENABLE:
|
||||
pr_debug("MODEL IOCTL: AMDKFD_IOC_RUNTIME_ENABLE\n");
|
||||
fprintf(stderr, "model: Debugger runtime not implemented\n");
|
||||
fprintf(stderr, "Fix this by clearing bit 30 of the 'capability' field in $HSA_MODEL_TOPOLOGY/%%d/properties\n");
|
||||
abort();
|
||||
default:
|
||||
fprintf(stderr, "model: Unimplemented KFD ioctl\n");
|
||||
abort();
|
||||
}
|
||||
}
|
||||
int model_kfd_ioctl(unsigned long request, void *arg)
|
||||
{
|
||||
/* Use a very simle locking strategy for correctness. IOCTLs should
|
||||
* be rare anyway and not contended considering the cost of running
|
||||
* the model itself.
|
||||
*
|
||||
* The bulk of model execution happens in a separate thread *without*
|
||||
* holding the IOCTL mutex. */
|
||||
pthread_mutex_lock(&model_ioctl_mutex);
|
||||
int ret = model_kfd_ioctl_locked(request, arg);
|
||||
pthread_mutex_unlock(&model_ioctl_mutex);
|
||||
return ret;
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
#include <stdio.h>
|
||||
#include <errno.h>
|
||||
#include <sys/ioctl.h>
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include "hsakmt/hsakmtmodel.h"
|
||||
|
||||
/* Call ioctl, restarting if it is interrupted */
|
||||
int hsakmt_ioctl(int fd, unsigned long request, void *arg)
|
||||
{
|
||||
if (hsakmt_use_model)
|
||||
return model_kfd_ioctl(request, arg);
|
||||
|
||||
int ret;
|
||||
|
||||
do {
|
||||
ret = ioctl(fd, request, arg);
|
||||
} while (ret == -1 && (errno == EINTR || errno == EAGAIN));
|
||||
|
||||
if (ret == -1 && errno == EBADF) {
|
||||
/* In case pthread_atfork didn't catch it, this will
|
||||
* make any subsequent hsaKmt calls fail in CHECK_KFD_OPEN.
|
||||
*/
|
||||
pr_err("KFD file descriptor not valid in this process\n");
|
||||
hsakmt_is_forked_child();
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
@@ -0,0 +1,251 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#ifndef LIBHSAKMT_H_INCLUDED
|
||||
#define LIBHSAKMT_H_INCLUDED
|
||||
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
#include "hsakmt/hsakmt.h"
|
||||
#include <pthread.h>
|
||||
#include <stdint.h>
|
||||
#include <limits.h>
|
||||
|
||||
extern int hsakmt_kfd_fd;
|
||||
extern unsigned long hsakmt_kfd_open_count;
|
||||
extern bool hsakmt_forked;
|
||||
extern pthread_mutex_t hsakmt_mutex;
|
||||
extern bool hsakmt_is_dgpu;
|
||||
extern bool hsakmt_is_svm_api_supported;
|
||||
extern int hsakmt_zfb_support;
|
||||
|
||||
extern HsaVersionInfo hsakmt_kfd_version_info;
|
||||
|
||||
#undef HSAKMTAPI
|
||||
#define HSAKMTAPI __attribute__((visibility ("default")))
|
||||
|
||||
#if defined(__clang__)
|
||||
#if __has_feature(address_sanitizer)
|
||||
#define SANITIZER_AMDGPU 1
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/*Avoid pointer-to-int-cast warning*/
|
||||
#define PORT_VPTR_TO_UINT64(vptr) ((uint64_t)(unsigned long)(vptr))
|
||||
|
||||
/*Avoid int-to-pointer-cast warning*/
|
||||
#define PORT_UINT64_TO_VPTR(v) ((void*)(unsigned long)(v))
|
||||
|
||||
#define CHECK_KFD_OPEN() \
|
||||
do { if (hsakmt_kfd_open_count == 0 || hsakmt_forked) return HSAKMT_STATUS_KERNEL_IO_CHANNEL_NOT_OPENED; } while (0)
|
||||
|
||||
#define CHECK_KFD_MINOR_VERSION(minor) \
|
||||
do { if ((minor) > hsakmt_kfd_version_info.KernelInterfaceMinorVersion)\
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED; } while (0)
|
||||
|
||||
extern int hsakmt_page_size;
|
||||
extern int hsakmt_page_shift;
|
||||
|
||||
/* Might be defined in limits.h on platforms where it is constant (used by musl) */
|
||||
/* See also: https://pubs.opengroup.org/onlinepubs/7908799/xsh/limits.h.html */
|
||||
#ifndef PAGE_SIZE
|
||||
#define PAGE_SIZE hsakmt_page_size
|
||||
#endif
|
||||
#ifndef PAGE_SHIFT
|
||||
#define PAGE_SHIFT hsakmt_page_shift
|
||||
#endif
|
||||
|
||||
/* VI HW bug requires this virtual address alignment */
|
||||
#define TONGA_PAGE_SIZE 0x8000
|
||||
|
||||
/* 64KB BigK fragment size for TLB efficiency */
|
||||
#define GPU_BIGK_PAGE_SIZE (1 << 16)
|
||||
|
||||
/* 2MB huge page size for 4-level page tables on Vega10 and later GPUs */
|
||||
#define GPU_HUGE_PAGE_SIZE (2 << 20)
|
||||
|
||||
#define CHECK_PAGE_MULTIPLE(x) \
|
||||
do { if ((uint64_t)PORT_VPTR_TO_UINT64(x) % PAGE_SIZE) return HSAKMT_STATUS_INVALID_PARAMETER; } while(0)
|
||||
|
||||
#define ALIGN_UP(x,align) (((uint64_t)(x) + (align) - 1) & ~(uint64_t)((align)-1))
|
||||
#define ALIGN_UP_32(x,align) (((uint32_t)(x) + (align) - 1) & ~(uint32_t)((align)-1))
|
||||
#define PAGE_ALIGN_UP(x) ALIGN_UP(x,PAGE_SIZE)
|
||||
#define BITMASK(n) ((n) ? (UINT64_MAX >> (sizeof(UINT64_MAX) * CHAR_BIT - (n))) : 0)
|
||||
#define ARRAY_LEN(array) (sizeof(array) / sizeof(array[0]))
|
||||
|
||||
/* HSA Thunk logging usage */
|
||||
extern int hsakmt_debug_level;
|
||||
#define hsakmt_print(level, fmt, ...) \
|
||||
do { if (level <= hsakmt_debug_level) fprintf(stderr, fmt, ##__VA_ARGS__); } while (0)
|
||||
#define HSAKMT_DEBUG_LEVEL_DEFAULT -1
|
||||
#define HSAKMT_DEBUG_LEVEL_ERR 3
|
||||
#define HSAKMT_DEBUG_LEVEL_WARNING 4
|
||||
#define HSAKMT_DEBUG_LEVEL_INFO 6
|
||||
#define HSAKMT_DEBUG_LEVEL_DEBUG 7
|
||||
#define pr_err(fmt, ...) \
|
||||
hsakmt_print(HSAKMT_DEBUG_LEVEL_ERR, fmt, ##__VA_ARGS__)
|
||||
#define pr_warn(fmt, ...) \
|
||||
hsakmt_print(HSAKMT_DEBUG_LEVEL_WARNING, fmt, ##__VA_ARGS__)
|
||||
#define pr_info(fmt, ...) \
|
||||
hsakmt_print(HSAKMT_DEBUG_LEVEL_INFO, fmt, ##__VA_ARGS__)
|
||||
#define pr_debug(fmt, ...) \
|
||||
hsakmt_print(HSAKMT_DEBUG_LEVEL_DEBUG, fmt, ##__VA_ARGS__)
|
||||
#define pr_err_once(fmt, ...) \
|
||||
({ \
|
||||
static bool __print_once; \
|
||||
if (!__print_once) { \
|
||||
__print_once = true; \
|
||||
pr_err(fmt, ##__VA_ARGS__); \
|
||||
} \
|
||||
})
|
||||
#define pr_warn_once(fmt, ...) \
|
||||
({ \
|
||||
static bool __print_once; \
|
||||
if (!__print_once) { \
|
||||
__print_once = true; \
|
||||
pr_warn(fmt, ##__VA_ARGS__); \
|
||||
} \
|
||||
})
|
||||
|
||||
/* Expects gfxv (full) in decimal */
|
||||
#define HSA_GET_GFX_VERSION_MAJOR(gfxv) (((gfxv) / 10000) % 100)
|
||||
#define HSA_GET_GFX_VERSION_MINOR(gfxv) (((gfxv) / 100) % 100)
|
||||
#define HSA_GET_GFX_VERSION_STEP(gfxv) ((gfxv) % 100)
|
||||
|
||||
/* Expects HSA_ENGINE_ID.ui32, returns gfxv (full) in hex */
|
||||
#define HSA_GET_GFX_VERSION_FULL(ui32) \
|
||||
(((ui32.Major) << 16) | ((ui32.Minor) << 8) | (ui32.Stepping))
|
||||
|
||||
enum full_gfx_versions {
|
||||
GFX_VERSION_KAVERI = 0x070000,
|
||||
GFX_VERSION_HAWAII = 0x070001,
|
||||
GFX_VERSION_CARRIZO = 0x080001,
|
||||
GFX_VERSION_TONGA = 0x080002,
|
||||
GFX_VERSION_FIJI = 0x080003,
|
||||
GFX_VERSION_POLARIS10 = 0x080003,
|
||||
GFX_VERSION_POLARIS11 = 0x080003,
|
||||
GFX_VERSION_POLARIS12 = 0x080003,
|
||||
GFX_VERSION_VEGAM = 0x080003,
|
||||
GFX_VERSION_VEGA10 = 0x090000,
|
||||
GFX_VERSION_RAVEN = 0x090002,
|
||||
GFX_VERSION_VEGA12 = 0x090004,
|
||||
GFX_VERSION_VEGA20 = 0x090006,
|
||||
GFX_VERSION_ARCTURUS = 0x090008,
|
||||
GFX_VERSION_ALDEBARAN = 0x09000A,
|
||||
GFX_VERSION_AQUA_VANJARAM = 0x090400,
|
||||
GFX_VERSION_GFX950 = 0x090500,
|
||||
GFX_VERSION_RENOIR = 0x09000C,
|
||||
GFX_VERSION_NAVI10 = 0x0A0100,
|
||||
GFX_VERSION_NAVI12 = 0x0A0101,
|
||||
GFX_VERSION_NAVI14 = 0x0A0102,
|
||||
GFX_VERSION_CYAN_SKILLFISH = 0x0A0103,
|
||||
GFX_VERSION_SIENNA_CICHLID = 0x0A0300,
|
||||
GFX_VERSION_NAVY_FLOUNDER = 0x0A0301,
|
||||
GFX_VERSION_DIMGREY_CAVEFISH = 0x0A0302,
|
||||
GFX_VERSION_VANGOGH = 0x0A0303,
|
||||
GFX_VERSION_BEIGE_GOBY = 0x0A0304,
|
||||
GFX_VERSION_YELLOW_CARP = 0x0A0305,
|
||||
GFX_VERSION_PLUM_BONITO = 0x0B0000,
|
||||
GFX_VERSION_WHEAT_NAS = 0x0B0001,
|
||||
GFX_VERSION_GFX1200 = 0x0C0000,
|
||||
GFX_VERSION_GFX1201 = 0x0C0001,
|
||||
};
|
||||
|
||||
struct hsa_gfxip_table {
|
||||
uint16_t device_id; // Device ID
|
||||
unsigned char major; // GFXIP Major engine version
|
||||
unsigned char minor; // GFXIP Minor engine version
|
||||
unsigned char stepping; // GFXIP Stepping info
|
||||
const char *amd_name; // CALName of the device
|
||||
};
|
||||
|
||||
HSAKMT_STATUS hsakmt_init_kfd_version(void);
|
||||
|
||||
#define IS_SOC15(gfxv) ((gfxv) >= GFX_VERSION_VEGA10)
|
||||
|
||||
HSAKMT_STATUS hsakmt_validate_nodeid(uint32_t nodeid, uint32_t *gpu_id);
|
||||
HSAKMT_STATUS hsakmt_gpuid_to_nodeid(uint32_t gpu_id, uint32_t* node_id);
|
||||
uint32_t hsakmt_get_gfxv_by_node_id(HSAuint32 node_id);
|
||||
bool hsakmt_prefer_ats(HSAuint32 node_id);
|
||||
uint16_t hsakmt_get_device_id_by_node_id(HSAuint32 node_id);
|
||||
uint16_t hsakmt_get_device_id_by_gpu_id(HSAuint32 gpu_id);
|
||||
uint32_t hsakmt_get_direct_link_cpu(uint32_t gpu_node);
|
||||
int get_drm_render_fd_by_gpu_id(HSAuint32 gpu_id);
|
||||
HSAKMT_STATUS hsakmt_validate_nodeid_array(uint32_t **gpu_id_array,
|
||||
uint32_t NumberOfNodes, uint32_t *NodeArray);
|
||||
|
||||
HSAKMT_STATUS hsakmt_topology_sysfs_get_system_props(HsaSystemProperties *props);
|
||||
HSAKMT_STATUS hsakmt_topology_get_node_props(HSAuint32 NodeId,
|
||||
HsaNodeProperties *NodeProperties);
|
||||
HSAKMT_STATUS hsakmt_topology_get_iolink_props(HSAuint32 NodeId,
|
||||
HSAuint32 NumIoLinks,
|
||||
HsaIoLinkProperties *IoLinkProperties);
|
||||
void hsakmt_topology_setup_is_dgpu_param(HsaNodeProperties *props);
|
||||
bool hsakmt_topology_is_svm_needed(HSA_ENGINE_ID EngineId);
|
||||
|
||||
HSAuint32 hsakmt_PageSizeFromFlags(unsigned int pageSizeFlags);
|
||||
|
||||
void* hsakmt_allocate_exec_aligned_memory_gpu(uint32_t size, uint32_t align,
|
||||
uint32_t gpu_id,
|
||||
uint32_t NodeId, bool NonPaged,
|
||||
bool DeviceLocal, bool Uncached);
|
||||
void hsakmt_free_exec_aligned_memory_gpu(void *addr, uint32_t size, uint32_t align);
|
||||
HSAKMT_STATUS hsakmt_init_process_doorbells(unsigned int NumNodes);
|
||||
void hsakmt_destroy_process_doorbells(void);
|
||||
HSAKMT_STATUS hsakmt_init_device_debugging_memory(unsigned int NumNodes);
|
||||
void hsakmt_destroy_device_debugging_memory(void);
|
||||
bool hsakmt_debug_get_reg_status(uint32_t node_id);
|
||||
HSAKMT_STATUS hsakmt_init_counter_props(unsigned int NumNodes);
|
||||
void hsakmt_destroy_counter_props(void);
|
||||
uint32_t *hsakmt_convert_queue_ids(HSAuint32 NumQueues, HSA_QUEUEID *Queues);
|
||||
|
||||
extern int hsakmt_ioctl(int fd, unsigned long request, void *arg);
|
||||
|
||||
/* Void pointer arithmetic (or remove -Wpointer-arith to allow void pointers arithmetic) */
|
||||
#define VOID_PTR_ADD32(ptr,n) (void*)((uint32_t*)(ptr) + n)/*ptr + offset*/
|
||||
#define VOID_PTR_ADD(ptr,n) (void*)((uint8_t*)(ptr) + n)/*ptr + offset*/
|
||||
#define VOID_PTR_SUB(ptr,n) (void*)((uint8_t*)(ptr) - n)/*ptr - offset*/
|
||||
#define VOID_PTRS_SUB(ptr1,ptr2) (uint64_t)((uint8_t*)(ptr1) - (uint8_t*)(ptr2)) /*ptr1 - ptr2*/
|
||||
|
||||
#define MIN(a, b) ({ \
|
||||
typeof(a) tmp1 = (a), tmp2 = (b); \
|
||||
tmp1 < tmp2 ? tmp1 : tmp2; })
|
||||
|
||||
#define MAX(a, b) ({ \
|
||||
typeof(a) tmp1 = (a), tmp2 = (b); \
|
||||
tmp1 > tmp2 ? tmp1 : tmp2; })
|
||||
|
||||
#define POWER_OF_2(x) ((x && (!(x & (x - 1)))) ? 1 : 0)
|
||||
|
||||
void hsakmt_clear_events_page(void);
|
||||
void hsakmt_fmm_clear_all_mem(void);
|
||||
void hsakmt_clear_process_doorbells(void);
|
||||
uint32_t hsakmt_get_num_sysfs_nodes(void);
|
||||
|
||||
bool hsakmt_is_forked_child(void);
|
||||
|
||||
/* Calculate VGPR and SGPR register file size per CU */
|
||||
uint32_t hsakmt_get_vgpr_size_per_cu(uint32_t gfxv);
|
||||
#define SGPR_SIZE_PER_CU 0x4000
|
||||
#endif
|
||||
@@ -0,0 +1,94 @@
|
||||
HSAKMT_1
|
||||
{
|
||||
global:
|
||||
hsaKmtOpenKFD;
|
||||
hsaKmtCloseKFD;
|
||||
hsaKmtGetVersion;
|
||||
hsaKmtAcquireSystemProperties;
|
||||
hsaKmtReleaseSystemProperties;
|
||||
hsaKmtGetNodeProperties;
|
||||
hsaKmtGetNodeMemoryProperties;
|
||||
hsaKmtGetNodeCacheProperties;
|
||||
hsaKmtGetNodeIoLinkProperties;
|
||||
hsaKmtCreateEvent;
|
||||
hsaKmtDestroyEvent;
|
||||
hsaKmtSetEvent;
|
||||
hsaKmtResetEvent;
|
||||
hsaKmtQueryEventState;
|
||||
hsaKmtWaitOnEvent;
|
||||
hsaKmtWaitOnMultipleEvents;
|
||||
hsaKmtCreateQueue;
|
||||
hsaKmtUpdateQueue;
|
||||
hsaKmtDestroyQueue;
|
||||
hsaKmtSetQueueCUMask;
|
||||
hsaKmtSetMemoryPolicy;
|
||||
hsaKmtAllocMemory;
|
||||
hsaKmtAllocMemoryAlign;
|
||||
hsaKmtFreeMemory;
|
||||
hsaKmtAvailableMemory;
|
||||
hsaKmtRegisterMemory;
|
||||
hsaKmtRegisterMemoryToNodes;
|
||||
hsaKmtRegisterMemoryWithFlags;
|
||||
hsaKmtRegisterGraphicsHandleToNodes;
|
||||
hsaKmtShareMemory;
|
||||
hsaKmtRegisterSharedHandle;
|
||||
hsaKmtRegisterSharedHandleToNodes;
|
||||
hsaKmtProcessVMRead;
|
||||
hsaKmtProcessVMWrite;
|
||||
hsaKmtDeregisterMemory;
|
||||
hsaKmtMapMemoryToGPU;
|
||||
hsaKmtMapMemoryToGPUNodes;
|
||||
hsaKmtUnmapMemoryToGPU;
|
||||
hsaKmtDbgRegister;
|
||||
hsaKmtDbgUnregister;
|
||||
hsaKmtDbgWavefrontControl;
|
||||
hsaKmtDbgAddressWatch;
|
||||
hsaKmtDbgEnable;
|
||||
hsaKmtDbgDisable;
|
||||
hsaKmtDbgGetDeviceData;
|
||||
hsaKmtDbgGetQueueData;
|
||||
hsaKmtGetClockCounters;
|
||||
hsaKmtPmcGetCounterProperties;
|
||||
hsaKmtPmcRegisterTrace;
|
||||
hsaKmtPmcUnregisterTrace;
|
||||
hsaKmtPmcAcquireTraceAccess;
|
||||
hsaKmtPmcReleaseTraceAccess;
|
||||
hsaKmtPmcStartTrace;
|
||||
hsaKmtPmcQueryTrace;
|
||||
hsaKmtPmcStopTrace;
|
||||
hsaKmtMapGraphicHandle;
|
||||
hsaKmtUnmapGraphicHandle;
|
||||
hsaKmtSetTrapHandler;
|
||||
hsaKmtGetTileConfig;
|
||||
hsaKmtQueryPointerInfo;
|
||||
hsaKmtSetMemoryUserData;
|
||||
hsaKmtGetQueueInfo;
|
||||
hsaKmtAllocQueueGWS;
|
||||
hsaKmtRuntimeEnable;
|
||||
hsaKmtRuntimeDisable;
|
||||
hsaKmtCheckRuntimeDebugSupport;
|
||||
hsaKmtGetRuntimeCapabilities;
|
||||
hsaKmtDebugTrapIoctl;
|
||||
hsaKmtSPMAcquire;
|
||||
hsaKmtSPMRelease;
|
||||
hsaKmtSPMSetDestBuffer;
|
||||
hsaKmtSVMSetAttr;
|
||||
hsaKmtSVMGetAttr;
|
||||
hsaKmtSetXNACKMode;
|
||||
hsaKmtGetXNACKMode;
|
||||
hsaKmtOpenSMI;
|
||||
hsaKmtExportDMABufHandle;
|
||||
hsaKmtWaitOnEvent_Ext;
|
||||
hsaKmtWaitOnMultipleEvents_Ext;
|
||||
hsaKmtReplaceAsanHeaderPage;
|
||||
hsaKmtReturnAsanHeaderPage;
|
||||
hsaKmtGetAMDGPUDeviceHandle;
|
||||
hsaKmtPcSamplingQueryCapabilities;
|
||||
hsaKmtPcSamplingCreate;
|
||||
hsaKmtPcSamplingDestroy;
|
||||
hsaKmtPcSamplingStart;
|
||||
hsaKmtPcSamplingStop;
|
||||
hsaKmtPcSamplingSupport;
|
||||
local: *;
|
||||
};
|
||||
|
||||
@@ -0,0 +1,685 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <assert.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/mman.h>
|
||||
#include <fcntl.h>
|
||||
#include "fmm.h"
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtSetMemoryPolicy(HSAuint32 Node,
|
||||
HSAuint32 DefaultPolicy,
|
||||
HSAuint32 AlternatePolicy,
|
||||
void *MemoryAddressAlternate,
|
||||
HSAuint64 MemorySizeInBytes)
|
||||
{
|
||||
struct kfd_ioctl_set_memory_policy_args args = {0};
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpu_id;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] node %d; default %d; alternate %d\n",
|
||||
__func__, Node, DefaultPolicy, AlternatePolicy);
|
||||
|
||||
result = hsakmt_validate_nodeid(Node, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
if (hsakmt_get_gfxv_by_node_id(Node) != GFX_VERSION_KAVERI)
|
||||
/* This is a legacy API useful on Kaveri only. On dGPU
|
||||
* the alternate aperture is setup and used
|
||||
* automatically for coherent allocations. Don't let
|
||||
* app override it.
|
||||
*/
|
||||
return HSAKMT_STATUS_NOT_IMPLEMENTED;
|
||||
|
||||
/*
|
||||
* We accept any legal policy and alternate address location.
|
||||
* You get CC everywhere anyway.
|
||||
*/
|
||||
if ((DefaultPolicy != HSA_CACHING_CACHED &&
|
||||
DefaultPolicy != HSA_CACHING_NONCACHED) ||
|
||||
(AlternatePolicy != HSA_CACHING_CACHED &&
|
||||
AlternatePolicy != HSA_CACHING_NONCACHED))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
CHECK_PAGE_MULTIPLE(MemoryAddressAlternate);
|
||||
CHECK_PAGE_MULTIPLE(MemorySizeInBytes);
|
||||
|
||||
args.gpu_id = gpu_id;
|
||||
args.default_policy = (DefaultPolicy == HSA_CACHING_CACHED) ?
|
||||
KFD_IOC_CACHE_POLICY_COHERENT :
|
||||
KFD_IOC_CACHE_POLICY_NONCOHERENT;
|
||||
|
||||
args.alternate_policy = (AlternatePolicy == HSA_CACHING_CACHED) ?
|
||||
KFD_IOC_CACHE_POLICY_COHERENT :
|
||||
KFD_IOC_CACHE_POLICY_NONCOHERENT;
|
||||
|
||||
args.alternate_aperture_base = (uintptr_t) MemoryAddressAlternate;
|
||||
args.alternate_aperture_size = MemorySizeInBytes;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SET_MEMORY_POLICY, &args);
|
||||
|
||||
return (err == -1) ? HSAKMT_STATUS_ERROR : HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAuint32 hsakmt_PageSizeFromFlags(unsigned int pageSizeFlags)
|
||||
{
|
||||
switch (pageSizeFlags) {
|
||||
case HSA_PAGE_SIZE_4KB: return 4*1024;
|
||||
case HSA_PAGE_SIZE_64KB: return 64*1024;
|
||||
case HSA_PAGE_SIZE_2MB: return 2*1024*1024;
|
||||
case HSA_PAGE_SIZE_1GB: return 1024*1024*1024;
|
||||
default:
|
||||
assert(false);
|
||||
return 4*1024;
|
||||
}
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtAllocMemory(HSAuint32 PreferredNode,
|
||||
HSAuint64 SizeInBytes,
|
||||
HsaMemFlags MemFlags,
|
||||
void **MemoryAddress)
|
||||
{
|
||||
return hsaKmtAllocMemoryAlign(PreferredNode, SizeInBytes, 0, MemFlags, MemoryAddress);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtAllocMemoryAlign(HSAuint32 PreferredNode,
|
||||
HSAuint64 SizeInBytes,
|
||||
HSAuint64 Alignment,
|
||||
HsaMemFlags MemFlags,
|
||||
void **MemoryAddress)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpu_id;
|
||||
HSAuint64 page_size;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (MemFlags.ui32.Contiguous)
|
||||
CHECK_KFD_MINOR_VERSION(16);
|
||||
|
||||
pr_debug("[%s] node %d\n", __func__, PreferredNode);
|
||||
|
||||
result = hsakmt_validate_nodeid(PreferredNode, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, PreferredNode);
|
||||
return result;
|
||||
}
|
||||
|
||||
page_size = hsakmt_PageSizeFromFlags(MemFlags.ui32.PageSize);
|
||||
|
||||
if (Alignment && (Alignment < page_size || !POWER_OF_2(Alignment)))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (!MemoryAddress || !SizeInBytes || (SizeInBytes & (page_size-1)))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (MemFlags.ui32.FixedAddress) {
|
||||
if (*MemoryAddress == NULL)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
} else
|
||||
*MemoryAddress = NULL;
|
||||
|
||||
if ((MemFlags.ui32.CoarseGrain && MemFlags.ui32.ExtendedCoherent) ||
|
||||
(MemFlags.ui32.ExtendedCoherent && MemFlags.ui32.Uncached))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (MemFlags.ui32.Scratch) {
|
||||
if (Alignment) {
|
||||
// Scratch memory currently forced to SCRATCH_ALIGN
|
||||
pr_err("[%s] Alignment not supported for scratch memory: %d\n", __func__, PreferredNode);
|
||||
return HSAKMT_STATUS_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
*MemoryAddress = hsakmt_fmm_allocate_scratch(gpu_id, *MemoryAddress, SizeInBytes);
|
||||
|
||||
if (!(*MemoryAddress)) {
|
||||
pr_err("[%s] failed to allocate %lu bytes from scratch\n",
|
||||
__func__, SizeInBytes);
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
}
|
||||
|
||||
pr_debug("[%s] node %d address %p size %lu from scratch\n", __func__, PreferredNode, *MemoryAddress, SizeInBytes);
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
/* GPU allocated system memory */
|
||||
if (!gpu_id || !MemFlags.ui32.NonPaged || hsakmt_zfb_support || MemFlags.ui32.GTTAccess
|
||||
|| MemFlags.ui32.OnlyAddress) {
|
||||
/* Backwards compatibility hack: Allocate system memory if app
|
||||
* asks for paged memory from a GPU node.
|
||||
*/
|
||||
|
||||
/* If allocate VRAM under ZFB mode */
|
||||
if (hsakmt_zfb_support && gpu_id && MemFlags.ui32.NonPaged == 1)
|
||||
MemFlags.ui32.CoarseGrain = 1;
|
||||
|
||||
*MemoryAddress = hsakmt_fmm_allocate_host(gpu_id, MemFlags.ui32.GTTAccess ? 0 : PreferredNode,
|
||||
*MemoryAddress, SizeInBytes, Alignment, MemFlags);
|
||||
|
||||
if (!(*MemoryAddress)) {
|
||||
pr_err("[%s] failed to allocate %lu bytes from host\n",
|
||||
__func__, SizeInBytes);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
pr_debug("[%s] node %d address %p size %lu from host\n", __func__, PreferredNode, *MemoryAddress, SizeInBytes);
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
/* GPU allocated VRAM */
|
||||
/* sanity check cannot do OnlyAddress and NoAddress alloc at same time */
|
||||
if (MemFlags.ui32.OnlyAddress && MemFlags.ui32.NoAddress) {
|
||||
pr_err("[%s] allocate addr-only and memory-only at same time\n",
|
||||
__func__);
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
}
|
||||
|
||||
*MemoryAddress = hsakmt_fmm_allocate_device(gpu_id, PreferredNode, *MemoryAddress,
|
||||
SizeInBytes, Alignment, MemFlags);
|
||||
|
||||
if (!(*MemoryAddress)) {
|
||||
pr_err("[%s] failed to allocate %lu bytes from device\n",
|
||||
__func__, SizeInBytes);
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
}
|
||||
|
||||
pr_debug("[%s] node %d address %p size %lu from device\n", __func__, PreferredNode, *MemoryAddress, SizeInBytes);
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtFreeMemory(void *MemoryAddress,
|
||||
HSAuint64 SizeInBytes)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] address %p\n", __func__, MemoryAddress);
|
||||
|
||||
if (!MemoryAddress) {
|
||||
pr_err("FIXME: freeing NULL pointer\n");
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
return hsakmt_fmm_release(MemoryAddress);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtAvailableMemory(HSAuint32 Node,
|
||||
HSAuint64 *AvailableBytes)
|
||||
{
|
||||
struct kfd_ioctl_get_available_memory_args args = {};
|
||||
HSAKMT_STATUS result;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(9);
|
||||
|
||||
pr_debug("[%s] node %d\n", __func__, Node);
|
||||
|
||||
result = hsakmt_validate_nodeid(Node, &args.gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, Node);
|
||||
return result;
|
||||
}
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_AVAILABLE_MEMORY, &args))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
*AvailableBytes = args.available;
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterMemory(void *MemoryAddress,
|
||||
HSAuint64 MemorySizeInBytes)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] address %p size %lu\n", __func__, MemoryAddress, MemorySizeInBytes);
|
||||
|
||||
if (!hsakmt_is_dgpu)
|
||||
/* TODO: support mixed APU and dGPU configurations */
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
return hsakmt_fmm_register_memory(MemoryAddress, MemorySizeInBytes,
|
||||
NULL, 0, true, false);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterMemoryToNodes(void *MemoryAddress,
|
||||
HSAuint64 MemorySizeInBytes,
|
||||
HSAuint64 NumberOfNodes,
|
||||
HSAuint32 *NodeArray)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
uint32_t *gpu_id_array;
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
pr_debug("[%s] address %p size %lu number of nodes %lu\n",
|
||||
__func__, MemoryAddress, MemorySizeInBytes, NumberOfNodes);
|
||||
|
||||
if (!hsakmt_is_dgpu)
|
||||
/* TODO: support mixed APU and dGPU configurations */
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
|
||||
ret = hsakmt_validate_nodeid_array(&gpu_id_array,
|
||||
NumberOfNodes, NodeArray);
|
||||
|
||||
if (ret == HSAKMT_STATUS_SUCCESS) {
|
||||
ret = hsakmt_fmm_register_memory(MemoryAddress, MemorySizeInBytes,
|
||||
gpu_id_array,
|
||||
NumberOfNodes*sizeof(uint32_t),
|
||||
true, false);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
free(gpu_id_array);
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterMemoryWithFlags(void *MemoryAddress,
|
||||
HSAuint64 MemorySizeInBytes,
|
||||
HsaMemFlags MemFlags)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
pr_debug("[%s] address %p size %lu\n",
|
||||
__func__, MemoryAddress, MemorySizeInBytes);
|
||||
|
||||
if (MemFlags.ui32.ExtendedCoherent && MemFlags.ui32.CoarseGrain)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
// Registered memory should be ordinary paged host memory.
|
||||
if ((MemFlags.ui32.HostAccess != 1) || (MemFlags.ui32.NonPaged == 1))
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
|
||||
if (!hsakmt_is_dgpu)
|
||||
/* TODO: support mixed APU and dGPU configurations */
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
|
||||
ret = hsakmt_fmm_register_memory(MemoryAddress, MemorySizeInBytes,
|
||||
NULL, 0, MemFlags.ui32.CoarseGrain, MemFlags.ui32.ExtendedCoherent);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterGraphicsHandleToNodes(HSAuint64 GraphicsResourceHandle,
|
||||
HsaGraphicsResourceInfo *GraphicsResourceInfo,
|
||||
HSAuint64 NumberOfNodes,
|
||||
HSAuint32 *NodeArray)
|
||||
{
|
||||
HSA_REGISTER_MEM_FLAGS regFlags;
|
||||
regFlags.Value = 0;
|
||||
|
||||
return hsaKmtRegisterGraphicsHandleToNodesExt(GraphicsResourceHandle,
|
||||
GraphicsResourceInfo,
|
||||
NumberOfNodes,
|
||||
NodeArray,
|
||||
regFlags);
|
||||
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterGraphicsHandleToNodesExt(HSAuint64 GraphicsResourceHandle,
|
||||
HsaGraphicsResourceInfo *GraphicsResourceInfo,
|
||||
HSAuint64 NumberOfNodes,
|
||||
HSAuint32 *NodeArray,
|
||||
HSA_REGISTER_MEM_FLAGS RegisterFlags)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
uint32_t *gpu_id_array = NULL;
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
pr_debug("[%s] number of nodes %lu\n", __func__, NumberOfNodes);
|
||||
|
||||
if (NodeArray != NULL || NumberOfNodes != 0) {
|
||||
ret = hsakmt_validate_nodeid_array(&gpu_id_array,
|
||||
NumberOfNodes, NodeArray);
|
||||
}
|
||||
|
||||
if (ret == HSAKMT_STATUS_SUCCESS) {
|
||||
ret = hsakmt_fmm_register_graphics_handle(
|
||||
GraphicsResourceHandle, GraphicsResourceInfo,
|
||||
gpu_id_array, NumberOfNodes * sizeof(uint32_t), RegisterFlags);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
free(gpu_id_array);
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtExportDMABufHandle(void *MemoryAddress,
|
||||
HSAuint64 MemorySizeInBytes,
|
||||
int *DMABufFd,
|
||||
HSAuint64 *Offset)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(12);
|
||||
|
||||
pr_debug("[%s] address %p\n", __func__, MemoryAddress);
|
||||
|
||||
return hsakmt_fmm_export_dma_buf_fd(MemoryAddress, MemorySizeInBytes,
|
||||
DMABufFd, Offset);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtShareMemory(void *MemoryAddress,
|
||||
HSAuint64 SizeInBytes,
|
||||
HsaSharedMemoryHandle *SharedMemoryHandle)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] address %p\n", __func__, MemoryAddress);
|
||||
|
||||
if (!SharedMemoryHandle)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
return hsakmt_fmm_share_memory(MemoryAddress, SizeInBytes, SharedMemoryHandle);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterSharedHandle(const HsaSharedMemoryHandle *SharedMemoryHandle,
|
||||
void **MemoryAddress,
|
||||
HSAuint64 *SizeInBytes)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] handle %p\n", __func__, SharedMemoryHandle);
|
||||
|
||||
return hsaKmtRegisterSharedHandleToNodes(SharedMemoryHandle,
|
||||
MemoryAddress,
|
||||
SizeInBytes,
|
||||
0,
|
||||
NULL);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterSharedHandleToNodes(const HsaSharedMemoryHandle *SharedMemoryHandle,
|
||||
void **MemoryAddress,
|
||||
HSAuint64 *SizeInBytes,
|
||||
HSAuint64 NumberOfNodes,
|
||||
HSAuint32 *NodeArray)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
uint32_t *gpu_id_array = NULL;
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
pr_debug("[%s] handle %p number of nodes %lu\n",
|
||||
__func__, SharedMemoryHandle, NumberOfNodes);
|
||||
|
||||
if (!SharedMemoryHandle)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (NodeArray) {
|
||||
ret = hsakmt_validate_nodeid_array(&gpu_id_array, NumberOfNodes, NodeArray);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
goto error;
|
||||
}
|
||||
|
||||
ret = hsakmt_fmm_register_shared_memory(SharedMemoryHandle,
|
||||
SizeInBytes,
|
||||
MemoryAddress,
|
||||
gpu_id_array,
|
||||
NumberOfNodes*sizeof(uint32_t));
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
goto error;
|
||||
|
||||
return ret;
|
||||
|
||||
error:
|
||||
if (gpu_id_array)
|
||||
free(gpu_id_array);
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtProcessVMRead(HSAuint32 Pid,
|
||||
HsaMemoryRange *LocalMemoryArray,
|
||||
HSAuint64 LocalMemoryArrayCount,
|
||||
HsaMemoryRange *RemoteMemoryArray,
|
||||
HSAuint64 RemoteMemoryArrayCount,
|
||||
HSAuint64 *SizeCopied)
|
||||
{
|
||||
pr_err("[%s] Deprecated\n", __func__);
|
||||
|
||||
return HSAKMT_STATUS_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtProcessVMWrite(HSAuint32 Pid,
|
||||
HsaMemoryRange *LocalMemoryArray,
|
||||
HSAuint64 LocalMemoryArrayCount,
|
||||
HsaMemoryRange *RemoteMemoryArray,
|
||||
HSAuint64 RemoteMemoryArrayCount,
|
||||
HSAuint64 *SizeCopied)
|
||||
{
|
||||
pr_err("[%s] Deprecated\n", __func__);
|
||||
|
||||
return HSAKMT_STATUS_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDeregisterMemory(void *MemoryAddress)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] address %p\n", __func__, MemoryAddress);
|
||||
|
||||
return hsakmt_fmm_deregister_memory(MemoryAddress);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtMapMemoryToGPU(void *MemoryAddress,
|
||||
HSAuint64 MemorySizeInBytes,
|
||||
HSAuint64 *AlternateVAGPU)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] address %p\n", __func__, MemoryAddress);
|
||||
|
||||
if (!MemoryAddress) {
|
||||
pr_err("FIXME: mapping NULL pointer\n");
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
if (AlternateVAGPU)
|
||||
*AlternateVAGPU = 0;
|
||||
|
||||
return hsakmt_fmm_map_to_gpu(MemoryAddress, MemorySizeInBytes, AlternateVAGPU);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtMapMemoryToGPUNodes(void *MemoryAddress,
|
||||
HSAuint64 MemorySizeInBytes,
|
||||
HSAuint64 *AlternateVAGPU,
|
||||
HsaMemMapFlags MemMapFlags,
|
||||
HSAuint64 NumberOfNodes,
|
||||
HSAuint32 *NodeArray)
|
||||
{
|
||||
uint32_t *gpu_id_array;
|
||||
HSAKMT_STATUS ret;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] address %p number of nodes %lu\n",
|
||||
__func__, MemoryAddress, NumberOfNodes);
|
||||
|
||||
if (!MemoryAddress) {
|
||||
pr_err("FIXME: mapping NULL pointer\n");
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
if (!hsakmt_is_dgpu && NumberOfNodes == 1)
|
||||
return hsaKmtMapMemoryToGPU(MemoryAddress,
|
||||
MemorySizeInBytes,
|
||||
AlternateVAGPU);
|
||||
|
||||
ret = hsakmt_validate_nodeid_array(&gpu_id_array,
|
||||
NumberOfNodes, NodeArray);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
return ret;
|
||||
|
||||
ret = hsakmt_fmm_map_to_gpu_nodes(MemoryAddress, MemorySizeInBytes,
|
||||
gpu_id_array, NumberOfNodes, AlternateVAGPU);
|
||||
|
||||
if (gpu_id_array)
|
||||
free(gpu_id_array);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtUnmapMemoryToGPU(void *MemoryAddress)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] address %p\n", __func__, MemoryAddress);
|
||||
|
||||
if (!MemoryAddress) {
|
||||
/* Workaround for runtime bug */
|
||||
pr_err("FIXME: Unmapping NULL pointer\n");
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
if (!hsakmt_fmm_unmap_from_gpu(MemoryAddress))
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
else
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtMapGraphicHandle(HSAuint32 NodeId,
|
||||
HSAuint64 GraphicDeviceHandle,
|
||||
HSAuint64 GraphicResourceHandle,
|
||||
HSAuint64 GraphicResourceOffset,
|
||||
HSAuint64 GraphicResourceSize,
|
||||
HSAuint64 *FlatMemoryAddress)
|
||||
{
|
||||
/* This API was only ever implemented in KFD for Kaveri and
|
||||
* was never upstreamed. There are no open-source users of
|
||||
* this interface. It has been superseded by
|
||||
* RegisterGraphicsHandleToNodes.
|
||||
*/
|
||||
return HSAKMT_STATUS_NOT_IMPLEMENTED;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtUnmapGraphicHandle(HSAuint32 NodeId,
|
||||
HSAuint64 FlatMemoryAddress,
|
||||
HSAuint64 SizeInBytes)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
return hsaKmtUnmapMemoryToGPU(PORT_UINT64_TO_VPTR(FlatMemoryAddress));
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtGetTileConfig(HSAuint32 NodeId, HsaGpuTileConfig *config)
|
||||
{
|
||||
struct kfd_ioctl_get_tile_config_args args = {0};
|
||||
uint32_t gpu_id;
|
||||
HSAKMT_STATUS result;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] node %d\n", __func__, NodeId);
|
||||
|
||||
result = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
/* Avoid Valgrind warnings about uninitialized data. Valgrind doesn't
|
||||
* know that KFD writes this.
|
||||
*/
|
||||
memset(config->TileConfig, 0, sizeof(*config->TileConfig) * config->NumTileConfigs);
|
||||
memset(config->MacroTileConfig, 0, sizeof(*config->MacroTileConfig) * config->NumMacroTileConfigs);
|
||||
|
||||
args.gpu_id = gpu_id;
|
||||
args.tile_config_ptr = (uint64_t)config->TileConfig;
|
||||
args.macro_tile_config_ptr = (uint64_t)config->MacroTileConfig;
|
||||
args.num_tile_configs = config->NumTileConfigs;
|
||||
args.num_macro_tile_configs = config->NumMacroTileConfigs;
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_GET_TILE_CONFIG, &args) != 0)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
config->NumTileConfigs = args.num_tile_configs;
|
||||
config->NumMacroTileConfigs = args.num_macro_tile_configs;
|
||||
|
||||
config->GbAddrConfig = args.gb_addr_config;
|
||||
|
||||
config->NumBanks = args.num_banks;
|
||||
config->NumRanks = args.num_ranks;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtQueryPointerInfo(const void *Pointer,
|
||||
HsaPointerInfo *PointerInfo)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] pointer %p\n", __func__, Pointer);
|
||||
|
||||
if (!PointerInfo)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
return hsakmt_fmm_get_mem_info(Pointer, PointerInfo);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtSetMemoryUserData(const void *Pointer,
|
||||
void *UserData)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
pr_debug("[%s] pointer %p\n", __func__, Pointer);
|
||||
|
||||
return hsakmt_fmm_set_mem_user_data(Pointer, UserData);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtReplaceAsanHeaderPage(void *addr)
|
||||
{
|
||||
#ifdef SANITIZER_AMDGPU
|
||||
pr_debug("[%s] address %p\n", __func__, addr);
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
return hsakmt_fmm_replace_asan_header_page(addr);
|
||||
#else
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
#endif
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtReturnAsanHeaderPage(void *addr)
|
||||
{
|
||||
#ifdef SANITIZER_AMDGPU
|
||||
pr_debug("[%s] address %p\n", __func__, addr);
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
return hsakmt_fmm_return_asan_header_page(addr);
|
||||
#else
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
#endif
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtGetAMDGPUDeviceHandle( HSAuint32 NodeId,
|
||||
HsaAMDGPUDeviceHandle *DeviceHandle)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
return hsakmt_fmm_get_amdgpu_device_handle(NodeId, DeviceHandle);
|
||||
}
|
||||
@@ -0,0 +1,260 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
/* glibc macro that enables access some nonstandard GNU/Linux extensions
|
||||
* such as RTLD_DEFAULT used by dlsym
|
||||
*/
|
||||
#define _GNU_SOURCE
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include "hsakmt/hsakmtmodel.h"
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/ioctl.h>
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#include <stdio.h>
|
||||
#include <strings.h>
|
||||
#include "fmm.h"
|
||||
#include <dlfcn.h>
|
||||
#include <string.h>
|
||||
|
||||
int (*hsakmt_fn_amdgpu_device_get_fd)(HsaAMDGPUDeviceHandle device_handle);
|
||||
|
||||
static const char kfd_device_name[] = "/dev/kfd";
|
||||
static pid_t parent_pid = -1;
|
||||
int hsakmt_debug_level;
|
||||
bool hsakmt_forked;
|
||||
|
||||
/* hsakmt_is_forked_child detects when the process has forked since the last
|
||||
* time this function was called. We cannot rely on pthread_atfork
|
||||
* because the process can fork without calling the fork function in
|
||||
* libc (using clone or calling the system call directly).
|
||||
*/
|
||||
bool hsakmt_is_forked_child(void)
|
||||
{
|
||||
pid_t cur_pid;
|
||||
|
||||
if (hsakmt_forked)
|
||||
return true;
|
||||
|
||||
cur_pid = getpid();
|
||||
|
||||
if (parent_pid == -1) {
|
||||
parent_pid = cur_pid;
|
||||
return false;
|
||||
}
|
||||
|
||||
if (parent_pid != cur_pid) {
|
||||
hsakmt_forked = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Callbacks from pthread_atfork */
|
||||
static void prepare_fork_handler(void)
|
||||
{
|
||||
pthread_mutex_lock(&hsakmt_mutex);
|
||||
}
|
||||
static void parent_fork_handler(void)
|
||||
{
|
||||
pthread_mutex_unlock(&hsakmt_mutex);
|
||||
}
|
||||
static void child_fork_handler(void)
|
||||
{
|
||||
pthread_mutex_init(&hsakmt_mutex, NULL);
|
||||
hsakmt_forked = true;
|
||||
}
|
||||
|
||||
/* Call this from the child process after fork. This will clear all
|
||||
* data that is duplicated from the parent process, that is not valid
|
||||
* in the child.
|
||||
* The topology information is duplicated from the parent is valid
|
||||
* in the child process so it is not cleared
|
||||
*/
|
||||
static void clear_after_fork(void)
|
||||
{
|
||||
hsakmt_clear_process_doorbells();
|
||||
hsakmt_clear_events_page();
|
||||
hsakmt_fmm_clear_all_mem();
|
||||
hsakmt_destroy_device_debugging_memory();
|
||||
if (hsakmt_kfd_fd) {
|
||||
close(hsakmt_kfd_fd);
|
||||
hsakmt_kfd_fd = -1;
|
||||
}
|
||||
hsakmt_kfd_open_count = 0;
|
||||
parent_pid = -1;
|
||||
hsakmt_forked = false;
|
||||
}
|
||||
|
||||
static inline void init_page_size(void)
|
||||
{
|
||||
hsakmt_page_size = sysconf(_SC_PAGESIZE);
|
||||
hsakmt_page_shift = ffs(hsakmt_page_size) - 1;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS init_vars_from_env(void)
|
||||
{
|
||||
char *envvar;
|
||||
int debug_level;
|
||||
|
||||
/* Normally libraries don't print messages. For debugging purpose, we'll
|
||||
* print messages if an environment variable, HSAKMT_DEBUG_LEVEL, is set.
|
||||
*/
|
||||
hsakmt_debug_level = HSAKMT_DEBUG_LEVEL_DEFAULT;
|
||||
|
||||
envvar = getenv("HSAKMT_DEBUG_LEVEL");
|
||||
if (envvar) {
|
||||
debug_level = atoi(envvar);
|
||||
if (debug_level >= HSAKMT_DEBUG_LEVEL_ERR &&
|
||||
debug_level <= HSAKMT_DEBUG_LEVEL_DEBUG)
|
||||
hsakmt_debug_level = debug_level;
|
||||
}
|
||||
|
||||
/* Check whether to support Zero frame buffer */
|
||||
envvar = getenv("HSA_ZFB");
|
||||
if (envvar)
|
||||
hsakmt_zfb_support = atoi(envvar);
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtOpenKFD(void)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
int fd = -1;
|
||||
HsaSystemProperties sys_props;
|
||||
char *error;
|
||||
char *useSvmStr;
|
||||
|
||||
pthread_mutex_lock(&hsakmt_mutex);
|
||||
|
||||
/* If the process has forked, the child process must re-initialize
|
||||
* it's connection to KFD. Any references tracked by hsakmt_kfd_open_count
|
||||
* belong to the parent
|
||||
*/
|
||||
if (hsakmt_is_forked_child())
|
||||
clear_after_fork();
|
||||
|
||||
if (hsakmt_kfd_open_count == 0) {
|
||||
static bool atfork_installed = false;
|
||||
|
||||
hsakmt_fn_amdgpu_device_get_fd = dlsym(RTLD_DEFAULT, "amdgpu_device_get_fd");
|
||||
if ((error = dlerror()) != NULL)
|
||||
pr_err("amdgpu_device_get_fd is not available: %s\n", error);
|
||||
else
|
||||
pr_info("amdgpu_device_get_fd is available %p\n", hsakmt_fn_amdgpu_device_get_fd);
|
||||
|
||||
result = init_vars_from_env();
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
goto open_failed;
|
||||
|
||||
// Check if we are using the hsakmtmodel and setup initial state
|
||||
model_init_env_vars();
|
||||
|
||||
if (hsakmt_kfd_fd < 0 && !hsakmt_use_model) {
|
||||
fd = open(kfd_device_name, O_RDWR | O_CLOEXEC);
|
||||
|
||||
if (fd == -1) {
|
||||
result = HSAKMT_STATUS_KERNEL_IO_CHANNEL_NOT_OPENED;
|
||||
goto open_failed;
|
||||
}
|
||||
|
||||
hsakmt_kfd_fd = fd;
|
||||
}
|
||||
|
||||
init_page_size();
|
||||
|
||||
result = hsakmt_init_kfd_version();
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
goto kfd_version_failed;
|
||||
|
||||
useSvmStr = getenv("HSA_USE_SVM");
|
||||
hsakmt_is_svm_api_supported = !(useSvmStr && !strcmp(useSvmStr, "0"));
|
||||
if(!hsakmt_use_model)
|
||||
result = hsakmt_topology_sysfs_get_system_props(&sys_props);
|
||||
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
goto topology_sysfs_failed;
|
||||
|
||||
hsakmt_kfd_open_count = 1;
|
||||
|
||||
if (hsakmt_init_device_debugging_memory(sys_props.NumNodes) != HSAKMT_STATUS_SUCCESS)
|
||||
pr_warn("Insufficient Memory. Debugging unavailable\n");
|
||||
|
||||
hsakmt_init_counter_props(sys_props.NumNodes);
|
||||
|
||||
if (!atfork_installed) {
|
||||
/* Atfork handlers cannot be uninstalled and
|
||||
* must be installed only once. Otherwise
|
||||
* prepare will deadlock when trying to take
|
||||
* the same lock multiple times.
|
||||
*/
|
||||
pthread_atfork(prepare_fork_handler,
|
||||
parent_fork_handler,
|
||||
child_fork_handler);
|
||||
atfork_installed = true;
|
||||
}
|
||||
} else {
|
||||
hsakmt_kfd_open_count++;
|
||||
result = HSAKMT_STATUS_KERNEL_ALREADY_OPENED;
|
||||
}
|
||||
|
||||
pthread_mutex_unlock(&hsakmt_mutex);
|
||||
return result;
|
||||
topology_sysfs_failed:
|
||||
kfd_version_failed:
|
||||
if (fd >= 0)
|
||||
close(fd);
|
||||
open_failed:
|
||||
pthread_mutex_unlock(&hsakmt_mutex);
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtCloseKFD(void)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
|
||||
pthread_mutex_lock(&hsakmt_mutex);
|
||||
|
||||
if (hsakmt_kfd_open_count > 0) {
|
||||
if (--hsakmt_kfd_open_count == 0) {
|
||||
hsakmt_destroy_counter_props();
|
||||
hsakmt_destroy_device_debugging_memory();
|
||||
}
|
||||
|
||||
result = HSAKMT_STATUS_SUCCESS;
|
||||
} else
|
||||
result = HSAKMT_STATUS_KERNEL_IO_CHANNEL_NOT_OPENED;
|
||||
|
||||
pthread_mutex_unlock(&hsakmt_mutex);
|
||||
|
||||
return result;
|
||||
}
|
||||
@@ -0,0 +1,235 @@
|
||||
/*
|
||||
* Copyright © 2023 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
#include <assert.h>
|
||||
#include <errno.h>
|
||||
|
||||
#define INVALID_TRACE_ID 0x0
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPcSamplingSupport(void)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(16);
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPcSamplingQueryCapabilities(HSAuint32 NodeId, void *sample_info,
|
||||
HSAuint32 sample_info_sz, HSAuint32 *size)
|
||||
{
|
||||
struct kfd_ioctl_pc_sample_args args = {0};
|
||||
uint32_t gpu_id;
|
||||
|
||||
if (size == NULL)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(16);
|
||||
|
||||
HSAKMT_STATUS ret = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, NodeId);
|
||||
return ret;
|
||||
}
|
||||
assert(sizeof(HsaPcSamplingInfo) == sizeof(struct kfd_pc_sample_info));
|
||||
|
||||
args.op = KFD_IOCTL_PCS_OP_QUERY_CAPABILITIES;
|
||||
args.gpu_id = gpu_id;
|
||||
args.sample_info_ptr = (uint64_t)sample_info;
|
||||
args.num_sample_info = sample_info_sz;
|
||||
args.flags = 0;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_PC_SAMPLE, &args);
|
||||
|
||||
*size = args.num_sample_info;
|
||||
|
||||
if (err) {
|
||||
switch (errno) {
|
||||
case ENOSPC:
|
||||
return HSAKMT_STATUS_BUFFER_TOO_SMALL;
|
||||
case EINVAL:
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
case EOPNOTSUPP:
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
case EBUSY:
|
||||
return HSAKMT_STATUS_UNAVAILABLE;
|
||||
default:
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPcSamplingCreate(HSAuint32 NodeId, HsaPcSamplingInfo *sample_info,
|
||||
HsaPcSamplingTraceId *traceId)
|
||||
{
|
||||
struct kfd_ioctl_pc_sample_args args = {0};
|
||||
uint32_t gpu_id;
|
||||
|
||||
if (sample_info == NULL || traceId == NULL)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
*traceId = INVALID_TRACE_ID;
|
||||
HSAKMT_STATUS ret = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, NodeId);
|
||||
return ret;
|
||||
}
|
||||
|
||||
args.op = KFD_IOCTL_PCS_OP_CREATE;
|
||||
args.gpu_id = gpu_id;
|
||||
args.sample_info_ptr = (uint64_t)sample_info;
|
||||
args.num_sample_info = 1;
|
||||
args.trace_id = INVALID_TRACE_ID;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_PC_SAMPLE, &args);
|
||||
if (err) {
|
||||
switch (errno) {
|
||||
case EINVAL:
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
case ENOMEM:
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
case EBUSY:
|
||||
return HSAKMT_STATUS_UNAVAILABLE;
|
||||
default:
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
}
|
||||
|
||||
*traceId = args.trace_id;
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPcSamplingDestroy(HSAuint32 NodeId, HsaPcSamplingTraceId traceId)
|
||||
{
|
||||
struct kfd_ioctl_pc_sample_args args = {0};
|
||||
uint32_t gpu_id;
|
||||
|
||||
if (traceId == INVALID_TRACE_ID)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
HSAKMT_STATUS ret = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, NodeId);
|
||||
return ret;
|
||||
}
|
||||
|
||||
hsaKmtPcSamplingStop(NodeId, traceId);
|
||||
|
||||
args.op = KFD_IOCTL_PCS_OP_DESTROY;
|
||||
args.gpu_id = gpu_id;
|
||||
args.trace_id = traceId;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_PC_SAMPLE, &args);
|
||||
if (err) {
|
||||
if (errno == EINVAL)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPcSamplingStart(HSAuint32 NodeId, HsaPcSamplingTraceId traceId)
|
||||
{
|
||||
struct kfd_ioctl_pc_sample_args args = {0};
|
||||
uint32_t gpu_id;
|
||||
|
||||
if (traceId == INVALID_TRACE_ID)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
HSAKMT_STATUS ret = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, NodeId);
|
||||
return ret;
|
||||
}
|
||||
|
||||
args.op = KFD_IOCTL_PCS_OP_START;
|
||||
args.gpu_id = gpu_id;
|
||||
args.trace_id = traceId;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_PC_SAMPLE, &args);
|
||||
if (err) {
|
||||
switch (errno) {
|
||||
case EINVAL:
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
case ENOMEM:
|
||||
return HSAKMT_STATUS_OUT_OF_RESOURCES;
|
||||
case EBUSY:
|
||||
return HSAKMT_STATUS_UNAVAILABLE;
|
||||
case EALREADY:
|
||||
return HSAKMT_STATUS_KERNEL_ALREADY_OPENED;
|
||||
default:
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPcSamplingStop(HSAuint32 NodeId, HsaPcSamplingTraceId traceId)
|
||||
{
|
||||
struct kfd_ioctl_pc_sample_args args = {0};
|
||||
uint32_t gpu_id;
|
||||
|
||||
if (traceId == INVALID_TRACE_ID)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
HSAKMT_STATUS ret = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, NodeId);
|
||||
return ret;
|
||||
}
|
||||
|
||||
args.op = KFD_IOCTL_PCS_OP_STOP;
|
||||
args.gpu_id = gpu_id;
|
||||
args.trace_id = traceId;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_PC_SAMPLE, &args);
|
||||
if (err) {
|
||||
switch (errno) {
|
||||
case EINVAL:
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
case EALREADY:
|
||||
return HSAKMT_STATUS_KERNEL_ALREADY_OPENED;
|
||||
default:
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
}
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
@@ -0,0 +1,694 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <linux/perf_event.h>
|
||||
#include <sys/syscall.h>
|
||||
#include "libhsakmt.h"
|
||||
#include "pmc_table.h"
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
#include <unistd.h>
|
||||
#include <sys/ioctl.h>
|
||||
#include <errno.h>
|
||||
#include <sys/mman.h>
|
||||
#include <fcntl.h>
|
||||
#include <semaphore.h>
|
||||
|
||||
#define BITS_PER_BYTE CHAR_BIT
|
||||
|
||||
#define HSA_PERF_MAGIC4CC 0x54415348
|
||||
|
||||
enum perf_trace_state {
|
||||
PERF_TRACE_STATE__STOPPED = 0,
|
||||
PERF_TRACE_STATE__STARTED
|
||||
};
|
||||
|
||||
struct perf_trace_block {
|
||||
enum perf_block_id block_id;
|
||||
uint32_t num_counters;
|
||||
uint64_t *counter_id;
|
||||
int *perf_event_fd;
|
||||
};
|
||||
|
||||
struct perf_trace {
|
||||
uint32_t magic4cc;
|
||||
uint32_t gpu_id;
|
||||
enum perf_trace_state state;
|
||||
uint32_t num_blocks;
|
||||
void *buf;
|
||||
uint64_t buf_size;
|
||||
struct perf_trace_block blocks[0];
|
||||
};
|
||||
|
||||
struct perf_counts_values {
|
||||
union {
|
||||
struct {
|
||||
uint64_t val;
|
||||
uint64_t ena;
|
||||
uint64_t run;
|
||||
};
|
||||
uint64_t values[3];
|
||||
};
|
||||
};
|
||||
|
||||
static HsaCounterProperties **counter_props;
|
||||
static unsigned int counter_props_count;
|
||||
|
||||
static ssize_t readn(int fd, void *buf, size_t n)
|
||||
{
|
||||
size_t left = n;
|
||||
ssize_t bytes;
|
||||
|
||||
while (left) {
|
||||
bytes = read(fd, buf, left);
|
||||
if (!bytes) /* reach EOF */
|
||||
return (n - left);
|
||||
if (bytes < 0) {
|
||||
if (errno == EINTR) /* read got interrupted */
|
||||
continue;
|
||||
else
|
||||
return -errno;
|
||||
}
|
||||
left -= bytes;
|
||||
buf = VOID_PTR_ADD(buf, bytes);
|
||||
}
|
||||
return n;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS hsakmt_init_counter_props(unsigned int NumNodes)
|
||||
{
|
||||
counter_props = calloc(NumNodes, sizeof(struct HsaCounterProperties *));
|
||||
if (!counter_props) {
|
||||
pr_warn("Profiling is not available.\n");
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
}
|
||||
|
||||
counter_props_count = NumNodes;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
void hsakmt_destroy_counter_props(void)
|
||||
{
|
||||
unsigned int i;
|
||||
|
||||
if (!counter_props)
|
||||
return;
|
||||
|
||||
for (i = 0; i < counter_props_count; i++)
|
||||
if (counter_props[i]) {
|
||||
free(counter_props[i]);
|
||||
counter_props[i] = NULL;
|
||||
}
|
||||
|
||||
free(counter_props);
|
||||
}
|
||||
|
||||
static int blockid2uuid(enum perf_block_id block_id, HSA_UUID *uuid)
|
||||
{
|
||||
int rc = 0;
|
||||
|
||||
switch (block_id) {
|
||||
case PERFCOUNTER_BLOCKID__CB:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_CB;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__CPF:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_CPF;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__CPG:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_CPG;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__DB:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_DB;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__GDS:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_GDS;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__GRBM:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_GRBM;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__GRBMSE:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_GRBMSE;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__IA:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_IA;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__MC:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_MC;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__PASC:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_PASC;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__PASU:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_PASU;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__SPI:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_SPI;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__SRBM:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_SRBM;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__SQ:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_SQ;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__SX:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_SX;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__TA:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_TA;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__TCA:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_TCA;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__TCC:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_TCC;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__TCP:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_TCP;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__TCS:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_TCS;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__TD:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_TD;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__VGT:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_VGT;
|
||||
break;
|
||||
case PERFCOUNTER_BLOCKID__WD:
|
||||
*uuid = HSA_PROFILEBLOCK_AMD_WD;
|
||||
break;
|
||||
default:
|
||||
/* If we reach this point, it's a bug */
|
||||
rc = -1;
|
||||
break;
|
||||
}
|
||||
|
||||
return rc;
|
||||
}
|
||||
|
||||
static HSAuint32 get_block_concurrent_limit(uint32_t node_id,
|
||||
HSAuint32 block_id)
|
||||
{
|
||||
uint32_t i;
|
||||
HsaCounterBlockProperties *block = &counter_props[node_id]->Blocks[0];
|
||||
|
||||
for (i = 0; i < PERFCOUNTER_BLOCKID__MAX; i++) {
|
||||
if (block->Counters[0].BlockIndex == block_id)
|
||||
return block->NumConcurrent;
|
||||
block = (HsaCounterBlockProperties *)&block->Counters[block->NumCounters];
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS perf_trace_ioctl(struct perf_trace_block *block,
|
||||
uint32_t cmd)
|
||||
{
|
||||
uint32_t i;
|
||||
|
||||
for (i = 0; i < block->num_counters; i++) {
|
||||
if (block->perf_event_fd[i] < 0)
|
||||
return HSAKMT_STATUS_UNAVAILABLE;
|
||||
if (ioctl(block->perf_event_fd[i], cmd, NULL))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS query_trace(int fd, uint64_t *buf)
|
||||
{
|
||||
struct perf_counts_values content;
|
||||
|
||||
if (fd < 0)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
if (readn(fd, &content, sizeof(content)) != sizeof(content))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
*buf = content.val;
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPmcGetCounterProperties(HSAuint32 NodeId,
|
||||
HsaCounterProperties **CounterProperties)
|
||||
{
|
||||
HSAKMT_STATUS rc = HSAKMT_STATUS_SUCCESS;
|
||||
uint32_t gpu_id, i, block_id;
|
||||
uint32_t counter_props_size = 0;
|
||||
uint32_t total_counters = 0;
|
||||
uint32_t total_concurrent = 0;
|
||||
struct perf_counter_block block = {0};
|
||||
uint32_t total_blocks = 0;
|
||||
HsaCounterBlockProperties *block_prop;
|
||||
|
||||
if (!counter_props)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
if (!CounterProperties)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (hsakmt_validate_nodeid(NodeId, &gpu_id) != HSAKMT_STATUS_SUCCESS)
|
||||
return HSAKMT_STATUS_INVALID_NODE_UNIT;
|
||||
|
||||
if (counter_props[NodeId]) {
|
||||
*CounterProperties = counter_props[NodeId];
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
for (i = 0; i < PERFCOUNTER_BLOCKID__MAX; i++) {
|
||||
rc = hsakmt_get_block_properties(NodeId, i, &block);
|
||||
if (rc != HSAKMT_STATUS_SUCCESS)
|
||||
return rc;
|
||||
total_concurrent += block.num_of_slots;
|
||||
total_counters += block.num_of_counters;
|
||||
/* If num_of_slots=0, this block doesn't exist */
|
||||
if (block.num_of_slots)
|
||||
total_blocks++;
|
||||
}
|
||||
|
||||
counter_props_size = sizeof(HsaCounterProperties) +
|
||||
sizeof(HsaCounterBlockProperties) * (total_blocks - 1) +
|
||||
sizeof(HsaCounter) * (total_counters - total_blocks);
|
||||
|
||||
counter_props[NodeId] = malloc(counter_props_size);
|
||||
if (!counter_props[NodeId])
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
counter_props[NodeId]->NumBlocks = total_blocks;
|
||||
counter_props[NodeId]->NumConcurrent = total_concurrent;
|
||||
|
||||
block_prop = &counter_props[NodeId]->Blocks[0];
|
||||
for (block_id = 0; block_id < PERFCOUNTER_BLOCKID__MAX; block_id++) {
|
||||
rc = hsakmt_get_block_properties(NodeId, block_id, &block);
|
||||
if (rc != HSAKMT_STATUS_SUCCESS) {
|
||||
free(counter_props[NodeId]);
|
||||
counter_props[NodeId] = NULL;
|
||||
return rc;
|
||||
}
|
||||
|
||||
if (!block.num_of_slots) /* not a valid block */
|
||||
continue;
|
||||
|
||||
blockid2uuid(block_id, &block_prop->BlockId);
|
||||
block_prop->NumCounters = block.num_of_counters;
|
||||
block_prop->NumConcurrent = block.num_of_slots;
|
||||
for (i = 0; i < block.num_of_counters; i++) {
|
||||
block_prop->Counters[i].BlockIndex = block_id;
|
||||
block_prop->Counters[i].CounterId = block.counter_ids[i];
|
||||
block_prop->Counters[i].CounterSizeInBits = block.counter_size_in_bits;
|
||||
block_prop->Counters[i].CounterMask = block.counter_mask;
|
||||
block_prop->Counters[i].Flags.ui32.Global = 1;
|
||||
block_prop->Counters[i].Type = HSA_PROFILE_TYPE_NONPRIV_IMMEDIATE;
|
||||
}
|
||||
|
||||
block_prop = (HsaCounterBlockProperties *)&block_prop->Counters[block_prop->NumCounters];
|
||||
}
|
||||
|
||||
*CounterProperties = counter_props[NodeId];
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
/* Registers a set of (HW) counters to be used for tracing/profiling */
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPmcRegisterTrace(HSAuint32 NodeId,
|
||||
HSAuint32 NumberOfCounters,
|
||||
HsaCounter *Counters,
|
||||
HsaPmcTraceRoot *TraceRoot)
|
||||
{
|
||||
uint32_t gpu_id, i, j;
|
||||
uint64_t min_buf_size = 0;
|
||||
struct perf_trace *trace = NULL;
|
||||
uint32_t concurrent_limit;
|
||||
const uint32_t MAX_COUNTERS = 512;
|
||||
|
||||
/* Declare performance counter ID 2D array as a contiguous block */
|
||||
uint64_t *counter_id = malloc(
|
||||
PERFCOUNTER_BLOCKID__MAX * MAX_COUNTERS * sizeof(uint64_t));
|
||||
uint32_t num_counters[PERFCOUNTER_BLOCKID__MAX] = {0};
|
||||
uint32_t block, num_blocks = 0, total_counters = 0;
|
||||
uint64_t *counter_id_ptr;
|
||||
int *fd_ptr;
|
||||
|
||||
pr_debug("[%s] Number of counters %d\n", __func__, NumberOfCounters);
|
||||
|
||||
if (counter_id == NULL) {
|
||||
pr_err("Failed to allocate memory for counter_id. Requested %zu bytes.\n",
|
||||
PERFCOUNTER_BLOCKID__MAX * MAX_COUNTERS * sizeof(uint64_t));
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
}
|
||||
|
||||
if (!counter_props) {
|
||||
pr_err("Profiling is not available, counter_props is NULL.\n");
|
||||
goto no_memory_exit;
|
||||
}
|
||||
|
||||
if (!Counters || !TraceRoot || NumberOfCounters == 0)
|
||||
goto invalid_parameter_exit;
|
||||
|
||||
if (hsakmt_validate_nodeid(NodeId, &gpu_id) != HSAKMT_STATUS_SUCCESS) {
|
||||
free(counter_id);
|
||||
return HSAKMT_STATUS_INVALID_NODE_UNIT;
|
||||
}
|
||||
|
||||
if (NumberOfCounters > MAX_COUNTERS) {
|
||||
pr_err("MAX_COUNTERS is too small for %d.\n", NumberOfCounters);
|
||||
goto no_memory_exit;
|
||||
}
|
||||
|
||||
/* Calculating the minimum buffer size */
|
||||
for (i = 0; i < NumberOfCounters; i++) {
|
||||
if (Counters[i].BlockIndex >= PERFCOUNTER_BLOCKID__MAX)
|
||||
goto invalid_parameter_exit;
|
||||
/* Only privileged counters need to register */
|
||||
if (Counters[i].Type > HSA_PROFILE_TYPE_PRIVILEGED_STREAMING)
|
||||
continue;
|
||||
min_buf_size += Counters[i].CounterSizeInBits/BITS_PER_BYTE;
|
||||
/* j: the first blank entry in the block to record counter_id */
|
||||
j = num_counters[Counters[i].BlockIndex];
|
||||
/* Make sure counter_id stays within bounds */
|
||||
if (j >= MAX_COUNTERS) {
|
||||
pr_err("Counter ID exceeded MAX_COUNTERS for block %d.\n",
|
||||
Counters[i].BlockIndex);
|
||||
goto invalid_parameter_exit;
|
||||
}
|
||||
/* Initialize counter_id */
|
||||
counter_id[Counters[i].BlockIndex * MAX_COUNTERS + j] = Counters[i].CounterId;
|
||||
num_counters[Counters[i].BlockIndex]++;
|
||||
total_counters++;
|
||||
}
|
||||
|
||||
/* Verify that the number of counters per block is not larger than the
|
||||
* number of slots.
|
||||
*/
|
||||
for (i = 0; i < PERFCOUNTER_BLOCKID__MAX; i++) {
|
||||
if (!num_counters[i])
|
||||
continue;
|
||||
concurrent_limit = get_block_concurrent_limit(NodeId, i);
|
||||
if (!concurrent_limit) {
|
||||
pr_err("Invalid block ID: %d\n", i);
|
||||
goto invalid_parameter_exit;
|
||||
}
|
||||
if (num_counters[i] > concurrent_limit) {
|
||||
pr_err("Counters exceed the limit.\n");
|
||||
goto invalid_parameter_exit;
|
||||
}
|
||||
num_blocks++;
|
||||
}
|
||||
|
||||
if (!num_blocks)
|
||||
goto invalid_parameter_exit;
|
||||
|
||||
/* Now we have sorted blocks/counters information in
|
||||
* num_counters[block_id] and counter_id[block_id][]. Allocate trace
|
||||
* and record the information.
|
||||
*/
|
||||
trace = (struct perf_trace *)calloc(sizeof(struct perf_trace)
|
||||
+ sizeof(struct perf_trace_block) * num_blocks
|
||||
+ sizeof(uint64_t) * total_counters
|
||||
+ sizeof(int) * total_counters,
|
||||
1);
|
||||
if (!trace) {
|
||||
pr_err("Failed to allocate memory for trace. Requested %zu bytes.\n",
|
||||
sizeof(struct perf_trace)
|
||||
+ sizeof(struct perf_trace_block) * num_blocks
|
||||
+ sizeof(uint64_t) * total_counters
|
||||
+ sizeof(int) * total_counters);
|
||||
goto no_memory_exit;
|
||||
}
|
||||
|
||||
/* Allocated area is partitioned as:
|
||||
* +---------------------------------+ trace
|
||||
* | perf_trace |
|
||||
* |---------------------------------| trace->blocks[0]
|
||||
* | perf_trace_block 0 |
|
||||
* | .... |
|
||||
* | perf_trace_block N-1 | trace->blocks[N-1]
|
||||
* |---------------------------------| <-- counter_id_ptr starts here
|
||||
* | block 0's counter IDs(uint64_t) |
|
||||
* | ...... |
|
||||
* | block N-1's counter IDs |
|
||||
* |---------------------------------| <-- perf_event_fd starts here
|
||||
* | block 0's perf_event_fds(int) |
|
||||
* | ...... |
|
||||
* | block N-1's perf_event_fds |
|
||||
* +---------------------------------+
|
||||
*/
|
||||
block = 0;
|
||||
counter_id_ptr = (uint64_t *)((char *)
|
||||
trace + sizeof(struct perf_trace)
|
||||
+ sizeof(struct perf_trace_block) * num_blocks);
|
||||
fd_ptr = (int *)(counter_id_ptr + total_counters);
|
||||
/* Fill in each block's information to the TraceId */
|
||||
for (i = 0; i < PERFCOUNTER_BLOCKID__MAX; i++) {
|
||||
if (!num_counters[i]) /* not a block to trace */
|
||||
continue;
|
||||
/* Following perf_trace + perf_trace_block x N are those
|
||||
* counter_id arrays. Assign the counter_id array belonging to
|
||||
* this block.
|
||||
*/
|
||||
trace->blocks[block].counter_id = counter_id_ptr;
|
||||
/* Fill in counter IDs to the counter_id array. */
|
||||
for (j = 0; j < num_counters[i]; j++)
|
||||
trace->blocks[block].counter_id[j] = counter_id[i * MAX_COUNTERS + j];
|
||||
trace->blocks[block].perf_event_fd = fd_ptr;
|
||||
/* how many counters to trace */
|
||||
trace->blocks[block].num_counters = num_counters[i];
|
||||
/* block index in "enum perf_block_id" */
|
||||
trace->blocks[block].block_id = i;
|
||||
block++; /* move to next */
|
||||
counter_id_ptr += num_counters[i];
|
||||
fd_ptr += num_counters[i];
|
||||
}
|
||||
|
||||
trace->magic4cc = HSA_PERF_MAGIC4CC;
|
||||
trace->gpu_id = gpu_id;
|
||||
trace->state = PERF_TRACE_STATE__STOPPED;
|
||||
trace->num_blocks = num_blocks;
|
||||
|
||||
TraceRoot->NumberOfPasses = 1;
|
||||
TraceRoot->TraceBufferMinSizeBytes = PAGE_ALIGN_UP(min_buf_size);
|
||||
TraceRoot->TraceId = PORT_VPTR_TO_UINT64(trace);
|
||||
|
||||
free(trace);
|
||||
free(counter_id);
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
no_memory_exit:
|
||||
free(counter_id);
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
invalid_parameter_exit:
|
||||
free(counter_id);
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
}
|
||||
|
||||
/* Unregisters a set of (HW) counters used for tracing/profiling */
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPmcUnregisterTrace(HSAuint32 NodeId,
|
||||
HSATraceId TraceId)
|
||||
{
|
||||
uint32_t gpu_id;
|
||||
struct perf_trace *trace;
|
||||
|
||||
pr_debug("[%s] Trace ID 0x%lx\n", __func__, TraceId);
|
||||
|
||||
if (TraceId == 0)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (hsakmt_validate_nodeid(NodeId, &gpu_id) != HSAKMT_STATUS_SUCCESS)
|
||||
return HSAKMT_STATUS_INVALID_NODE_UNIT;
|
||||
|
||||
trace = (struct perf_trace *)PORT_UINT64_TO_VPTR(TraceId);
|
||||
|
||||
if (trace->magic4cc != HSA_PERF_MAGIC4CC)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
if (trace->gpu_id != gpu_id)
|
||||
return HSAKMT_STATUS_INVALID_NODE_UNIT;
|
||||
|
||||
/* If the trace is in the running state, stop it */
|
||||
if (trace->state == PERF_TRACE_STATE__STARTED) {
|
||||
HSAKMT_STATUS status = hsaKmtPmcStopTrace(TraceId);
|
||||
|
||||
if (status != HSAKMT_STATUS_SUCCESS)
|
||||
return status;
|
||||
}
|
||||
|
||||
free(trace);
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPmcAcquireTraceAccess(HSAuint32 NodeId,
|
||||
HSATraceId TraceId)
|
||||
{
|
||||
struct perf_trace *trace;
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
uint32_t gpu_id;
|
||||
|
||||
pr_debug("[%s] Trace ID 0x%lx\n", __func__, TraceId);
|
||||
|
||||
if (TraceId == 0)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
trace = (struct perf_trace *)PORT_UINT64_TO_VPTR(TraceId);
|
||||
|
||||
if (trace->magic4cc != HSA_PERF_MAGIC4CC)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
if (hsakmt_validate_nodeid(NodeId, &gpu_id) != HSAKMT_STATUS_SUCCESS)
|
||||
return HSAKMT_STATUS_INVALID_NODE_UNIT;
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPmcReleaseTraceAccess(HSAuint32 NodeId,
|
||||
HSATraceId TraceId)
|
||||
{
|
||||
struct perf_trace *trace;
|
||||
|
||||
pr_debug("[%s] Trace ID 0x%lx\n", __func__, TraceId);
|
||||
|
||||
if (TraceId == 0)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
trace = (struct perf_trace *)PORT_UINT64_TO_VPTR(TraceId);
|
||||
|
||||
if (trace->magic4cc != HSA_PERF_MAGIC4CC)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
|
||||
/* Starts tracing operation on a previously established set of performance counters */
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPmcStartTrace(HSATraceId TraceId,
|
||||
void *TraceBuffer,
|
||||
HSAuint64 TraceBufferSizeBytes)
|
||||
{
|
||||
struct perf_trace *trace =
|
||||
(struct perf_trace *)PORT_UINT64_TO_VPTR(TraceId);
|
||||
uint32_t i;
|
||||
int32_t j;
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
pr_debug("[%s] Trace ID 0x%lx\n", __func__, TraceId);
|
||||
|
||||
if (TraceId == 0 || !TraceBuffer || TraceBufferSizeBytes == 0)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (trace->magic4cc != HSA_PERF_MAGIC4CC)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
for (i = 0; i < trace->num_blocks; i++) {
|
||||
ret = perf_trace_ioctl(&trace->blocks[i],
|
||||
PERF_EVENT_IOC_ENABLE);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
break;
|
||||
}
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
/* Disable enabled blocks before returning the failure. */
|
||||
j = (int32_t)i;
|
||||
while (--j >= 0)
|
||||
perf_trace_ioctl(&trace->blocks[j],
|
||||
PERF_EVENT_IOC_DISABLE);
|
||||
return ret;
|
||||
}
|
||||
|
||||
trace->state = PERF_TRACE_STATE__STARTED;
|
||||
trace->buf = TraceBuffer;
|
||||
trace->buf_size = TraceBufferSizeBytes;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
|
||||
/*Forces an update of all the counters that a previously started trace operation has registered */
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPmcQueryTrace(HSATraceId TraceId)
|
||||
{
|
||||
struct perf_trace *trace =
|
||||
(struct perf_trace *)PORT_UINT64_TO_VPTR(TraceId);
|
||||
uint32_t i, j;
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
uint64_t *buf;
|
||||
uint64_t buf_filled = 0;
|
||||
|
||||
if (TraceId == 0)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (trace->magic4cc != HSA_PERF_MAGIC4CC)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
buf = (uint64_t *)trace->buf;
|
||||
pr_debug("[%s] Trace buffer(%p): ", __func__, buf);
|
||||
for (i = 0; i < trace->num_blocks; i++)
|
||||
for (j = 0; j < trace->blocks[i].num_counters; j++) {
|
||||
buf_filled += sizeof(uint64_t);
|
||||
if (buf_filled > trace->buf_size)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
ret = query_trace(trace->blocks[i].perf_event_fd[j],
|
||||
buf);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
return ret;
|
||||
pr_debug("%lu_", *buf);
|
||||
buf++;
|
||||
}
|
||||
pr_debug("\n");
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
|
||||
/* Stops tracing operation on a previously established set of performance counters */
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtPmcStopTrace(HSATraceId TraceId)
|
||||
{
|
||||
struct perf_trace *trace =
|
||||
(struct perf_trace *)PORT_UINT64_TO_VPTR(TraceId);
|
||||
uint32_t i;
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
pr_debug("[%s] Trace ID 0x%lx\n", __func__, TraceId);
|
||||
|
||||
if (TraceId == 0)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (trace->magic4cc != HSA_PERF_MAGIC4CC)
|
||||
return HSAKMT_STATUS_INVALID_HANDLE;
|
||||
|
||||
for (i = 0; i < trace->num_blocks; i++) {
|
||||
ret = perf_trace_ioctl(&trace->blocks[i],
|
||||
PERF_EVENT_IOC_DISABLE);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS)
|
||||
return ret;
|
||||
}
|
||||
|
||||
trace->state = PERF_TRACE_STATE__STOPPED;
|
||||
|
||||
return ret;
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,74 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#ifndef PMC_TABLE_H
|
||||
#define PMC_TABLE_H
|
||||
|
||||
#include "libhsakmt.h"
|
||||
|
||||
enum perf_block_id {
|
||||
PERFCOUNTER_BLOCKID__FIRST = 0,
|
||||
/* non-privileged */
|
||||
PERFCOUNTER_BLOCKID__CB = PERFCOUNTER_BLOCKID__FIRST,
|
||||
PERFCOUNTER_BLOCKID__CPC,
|
||||
PERFCOUNTER_BLOCKID__CPF,
|
||||
PERFCOUNTER_BLOCKID__CPG,
|
||||
PERFCOUNTER_BLOCKID__DB,
|
||||
PERFCOUNTER_BLOCKID__GDS,
|
||||
PERFCOUNTER_BLOCKID__GRBM,
|
||||
PERFCOUNTER_BLOCKID__GRBMSE,
|
||||
PERFCOUNTER_BLOCKID__IA,
|
||||
PERFCOUNTER_BLOCKID__MC,
|
||||
PERFCOUNTER_BLOCKID__PASC,
|
||||
PERFCOUNTER_BLOCKID__PASU,
|
||||
PERFCOUNTER_BLOCKID__SPI,
|
||||
PERFCOUNTER_BLOCKID__SRBM,
|
||||
PERFCOUNTER_BLOCKID__SQ,
|
||||
PERFCOUNTER_BLOCKID__SX,
|
||||
PERFCOUNTER_BLOCKID__TA,
|
||||
PERFCOUNTER_BLOCKID__TCA,
|
||||
PERFCOUNTER_BLOCKID__TCC,
|
||||
PERFCOUNTER_BLOCKID__TCP,
|
||||
PERFCOUNTER_BLOCKID__TCS,
|
||||
PERFCOUNTER_BLOCKID__TD,
|
||||
PERFCOUNTER_BLOCKID__VGT,
|
||||
PERFCOUNTER_BLOCKID__WD,
|
||||
/* privileged */
|
||||
PERFCOUNTER_BLOCKID__MAX
|
||||
};
|
||||
|
||||
struct perf_counter_block {
|
||||
uint32_t num_of_slots;
|
||||
uint32_t num_of_counters;
|
||||
uint32_t *counter_ids;
|
||||
uint32_t counter_size_in_bits;
|
||||
uint64_t counter_mask;
|
||||
};
|
||||
|
||||
HSAKMT_STATUS hsakmt_get_block_properties(uint32_t node_id,
|
||||
enum perf_block_id block_id,
|
||||
struct perf_counter_block *block);
|
||||
|
||||
#endif // PMC_TABLE_H
|
||||
@@ -0,0 +1,954 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include "fmm.h"
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <math.h>
|
||||
#include <stdio.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/mman.h>
|
||||
#include <fcntl.h>
|
||||
#include <errno.h>
|
||||
|
||||
/* 1024 doorbells, 4 or 8 bytes each doorbell depending on ASIC generation */
|
||||
#define DOORBELL_SIZE(gfxv) (((gfxv) >= 0x90000) ? 8 : 4)
|
||||
#define DOORBELLS_PAGE_SIZE(ds) (1024 * (ds))
|
||||
|
||||
#define WG_CONTEXT_DATA_SIZE_PER_CU(gfxv, node) \
|
||||
(hsakmt_get_vgpr_size_per_cu(gfxv) + SGPR_SIZE_PER_CU + \
|
||||
(node.LDSSizeInKB << 10) + HWREG_SIZE_PER_CU)
|
||||
|
||||
#define CNTL_STACK_BYTES_PER_WAVE(gfxv) \
|
||||
((gfxv) >= GFX_VERSION_NAVI10 ? 12 : 8)
|
||||
|
||||
#define HWREG_SIZE_PER_CU 0x1000
|
||||
#define DEBUGGER_BYTES_ALIGN 64
|
||||
#define DEBUGGER_BYTES_PER_WAVE 32
|
||||
|
||||
struct queue {
|
||||
uint32_t queue_id;
|
||||
uint64_t wptr;
|
||||
uint64_t rptr;
|
||||
void *eop_buffer;
|
||||
void *ctx_save_restore;
|
||||
uint32_t ctx_save_restore_size;
|
||||
uint32_t ctl_stack_size;
|
||||
uint32_t debug_memory_size;
|
||||
uint32_t eop_buffer_size;
|
||||
uint32_t total_mem_alloc_size;
|
||||
uint32_t gfxv;
|
||||
bool use_ats;
|
||||
bool unified_ctx_save_restore;
|
||||
/* This queue structure is allocated from GPU with page aligned size
|
||||
* but only small bytes are used. We use the extra space in the end for
|
||||
* cu_mask bits array.
|
||||
*/
|
||||
uint32_t cu_mask_count; /* in bits */
|
||||
uint32_t cu_mask[0];
|
||||
};
|
||||
|
||||
struct process_doorbells {
|
||||
bool use_gpuvm;
|
||||
uint32_t size;
|
||||
void *mapping;
|
||||
pthread_mutex_t mutex;
|
||||
};
|
||||
|
||||
static unsigned int num_doorbells;
|
||||
static struct process_doorbells *doorbells;
|
||||
|
||||
uint32_t hsakmt_get_vgpr_size_per_cu(uint32_t gfxv)
|
||||
{
|
||||
uint32_t vgpr_size = 0x40000;
|
||||
|
||||
if (gfxv == GFX_VERSION_GFX950 ||
|
||||
(gfxv & ~(0xff)) == GFX_VERSION_AQUA_VANJARAM ||
|
||||
gfxv == GFX_VERSION_ALDEBARAN ||
|
||||
gfxv == GFX_VERSION_ARCTURUS)
|
||||
vgpr_size = 0x80000;
|
||||
|
||||
else if (gfxv == GFX_VERSION_PLUM_BONITO ||
|
||||
gfxv == GFX_VERSION_WHEAT_NAS ||
|
||||
gfxv == GFX_VERSION_GFX1200 ||
|
||||
gfxv == GFX_VERSION_GFX1201)
|
||||
vgpr_size = 0x60000;
|
||||
|
||||
return vgpr_size;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS hsakmt_init_process_doorbells(unsigned int NumNodes)
|
||||
{
|
||||
unsigned int i;
|
||||
HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
/* doorbells[] is accessed using Topology NodeId. This means doorbells[0],
|
||||
* which corresponds to CPU only Node, might not be used
|
||||
*/
|
||||
doorbells = malloc(NumNodes * sizeof(struct process_doorbells));
|
||||
if (!doorbells)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
for (i = 0; i < NumNodes; i++) {
|
||||
doorbells[i].use_gpuvm = false;
|
||||
doorbells[i].size = 0;
|
||||
doorbells[i].mapping = NULL;
|
||||
pthread_mutex_init(&doorbells[i].mutex, NULL);
|
||||
}
|
||||
|
||||
num_doorbells = NumNodes;
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
static void get_doorbell_map_info(uint32_t node_id,
|
||||
struct process_doorbells *doorbell)
|
||||
{
|
||||
/*
|
||||
* GPUVM doorbell on Tonga requires a workaround for VM TLB ACTIVE bit
|
||||
* lookup bug. Remove ASIC check when this is implemented in amdgpu.
|
||||
*/
|
||||
uint32_t gfxv = hsakmt_get_gfxv_by_node_id(node_id);
|
||||
doorbell->use_gpuvm = (hsakmt_is_dgpu && gfxv != GFX_VERSION_TONGA);
|
||||
doorbell->size = DOORBELLS_PAGE_SIZE(DOORBELL_SIZE(gfxv));
|
||||
|
||||
if (doorbell->size < (uint32_t) PAGE_SIZE) {
|
||||
doorbell->size = PAGE_SIZE;
|
||||
}
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
void hsakmt_destroy_process_doorbells(void)
|
||||
{
|
||||
unsigned int i;
|
||||
|
||||
if (!doorbells)
|
||||
return;
|
||||
|
||||
for (i = 0; i < num_doorbells; i++) {
|
||||
if (!doorbells[i].size)
|
||||
continue;
|
||||
|
||||
if (doorbells[i].use_gpuvm) {
|
||||
hsakmt_fmm_unmap_from_gpu(doorbells[i].mapping);
|
||||
hsakmt_fmm_release(doorbells[i].mapping);
|
||||
} else
|
||||
munmap(doorbells[i].mapping, doorbells[i].size);
|
||||
}
|
||||
|
||||
free(doorbells);
|
||||
doorbells = NULL;
|
||||
num_doorbells = 0;
|
||||
}
|
||||
|
||||
/* This is a special funcion that should be called only from the child process
|
||||
* after a fork(). This will clear doorbells duplicated from the parent.
|
||||
*/
|
||||
void hsakmt_clear_process_doorbells(void)
|
||||
{
|
||||
unsigned int i;
|
||||
|
||||
if (!doorbells)
|
||||
return;
|
||||
|
||||
for (i = 0; i < num_doorbells; i++) {
|
||||
if (!doorbells[i].size)
|
||||
continue;
|
||||
|
||||
if (!doorbells[i].use_gpuvm)
|
||||
munmap(doorbells[i].mapping, doorbells[i].size);
|
||||
}
|
||||
|
||||
free(doorbells);
|
||||
doorbells = NULL;
|
||||
num_doorbells = 0;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS map_doorbell_apu(HSAuint32 NodeId, HSAuint32 gpu_id,
|
||||
HSAuint64 doorbell_mmap_offset)
|
||||
{
|
||||
void *ptr;
|
||||
|
||||
ptr = mmap(0, doorbells[NodeId].size, PROT_READ|PROT_WRITE,
|
||||
MAP_SHARED, hsakmt_kfd_fd, doorbell_mmap_offset);
|
||||
|
||||
if (ptr == MAP_FAILED)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
doorbells[NodeId].mapping = ptr;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS map_doorbell_dgpu(HSAuint32 NodeId, HSAuint32 gpu_id,
|
||||
HSAuint64 doorbell_mmap_offset)
|
||||
{
|
||||
void *ptr;
|
||||
|
||||
ptr = hsakmt_fmm_allocate_doorbell(gpu_id, doorbells[NodeId].size,
|
||||
doorbell_mmap_offset);
|
||||
|
||||
if (!ptr)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
/* map for GPU access */
|
||||
if (hsakmt_fmm_map_to_gpu(ptr, doorbells[NodeId].size, NULL)) {
|
||||
hsakmt_fmm_release(ptr);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
doorbells[NodeId].mapping = ptr;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS map_doorbell(HSAuint32 NodeId, HSAuint32 gpu_id,
|
||||
HSAuint64 doorbell_mmap_offset)
|
||||
{
|
||||
HSAKMT_STATUS status = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
pthread_mutex_lock(&doorbells[NodeId].mutex);
|
||||
if (doorbells[NodeId].size) {
|
||||
pthread_mutex_unlock(&doorbells[NodeId].mutex);
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
get_doorbell_map_info(NodeId, &doorbells[NodeId]);
|
||||
|
||||
if (doorbells[NodeId].use_gpuvm) {
|
||||
status = map_doorbell_dgpu(NodeId, gpu_id, doorbell_mmap_offset);
|
||||
if (status != HSAKMT_STATUS_SUCCESS) {
|
||||
/* Fall back to the old method if KFD doesn't
|
||||
* support doorbells in GPUVM
|
||||
*/
|
||||
doorbells[NodeId].use_gpuvm = false;
|
||||
status = map_doorbell_apu(NodeId, gpu_id, doorbell_mmap_offset);
|
||||
}
|
||||
} else
|
||||
status = map_doorbell_apu(NodeId, gpu_id, doorbell_mmap_offset);
|
||||
|
||||
if (status != HSAKMT_STATUS_SUCCESS)
|
||||
doorbells[NodeId].size = 0;
|
||||
|
||||
pthread_mutex_unlock(&doorbells[NodeId].mutex);
|
||||
|
||||
return status;
|
||||
}
|
||||
|
||||
static void *allocate_exec_aligned_memory_cpu(uint32_t size)
|
||||
{
|
||||
void *ptr;
|
||||
|
||||
/* mmap will return a pointer with alignment equal to
|
||||
* sysconf(_SC_PAGESIZE).
|
||||
*
|
||||
* MAP_ANONYMOUS initializes the memory to zero.
|
||||
*/
|
||||
ptr = mmap(NULL, size, PROT_READ | PROT_WRITE | PROT_EXEC,
|
||||
MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
|
||||
|
||||
if (ptr == MAP_FAILED)
|
||||
return NULL;
|
||||
return ptr;
|
||||
}
|
||||
|
||||
/* The bool return indicate whether the queue needs a context-save-restore area*/
|
||||
static bool update_ctx_save_restore_size(uint32_t nodeid, struct queue *q)
|
||||
{
|
||||
HsaNodeProperties node;
|
||||
|
||||
if (q->gfxv < GFX_VERSION_CARRIZO)
|
||||
return false;
|
||||
if (hsaKmtGetNodeProperties(nodeid, &node))
|
||||
return false;
|
||||
if (node.NumFComputeCores && node.NumSIMDPerCU) {
|
||||
uint32_t ctl_stack_size, wg_data_size;
|
||||
uint32_t cu_num = node.NumFComputeCores / node.NumSIMDPerCU / node.NumXcc;
|
||||
uint32_t wave_num = (q->gfxv < GFX_VERSION_NAVI10)
|
||||
? MIN(cu_num * 40, node.NumShaderBanks / node.NumArrays * 512)
|
||||
: cu_num * 32;
|
||||
|
||||
ctl_stack_size = wave_num * CNTL_STACK_BYTES_PER_WAVE(q->gfxv) + 8;
|
||||
wg_data_size = cu_num * WG_CONTEXT_DATA_SIZE_PER_CU(q->gfxv, node);
|
||||
q->ctl_stack_size = PAGE_ALIGN_UP(sizeof(HsaUserContextSaveAreaHeader)
|
||||
+ ctl_stack_size);
|
||||
if ((q->gfxv & 0x3f0000) == 0xA0000) {
|
||||
/* HW design limits control stack size to 0x7000.
|
||||
* This is insufficient for theoretical PM4 cases
|
||||
* but sufficient for AQL, limited by SPI events.
|
||||
*/
|
||||
q->ctl_stack_size = MIN(q->ctl_stack_size, 0x7000);
|
||||
}
|
||||
|
||||
q->debug_memory_size =
|
||||
ALIGN_UP(wave_num * DEBUGGER_BYTES_PER_WAVE, DEBUGGER_BYTES_ALIGN);
|
||||
|
||||
q->ctx_save_restore_size = q->ctl_stack_size
|
||||
+ PAGE_ALIGN_UP(wg_data_size);
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void *hsakmt_allocate_exec_aligned_memory_gpu(uint32_t size, uint32_t align, uint32_t gpu_id,
|
||||
uint32_t NodeId, bool nonPaged,
|
||||
bool DeviceLocal,
|
||||
bool Uncached)
|
||||
{
|
||||
void *mem = NULL;
|
||||
HSAuint64 gpu_va;
|
||||
HsaMemFlags flags;
|
||||
HSAuint32 cpu_id = 0;
|
||||
|
||||
flags.Value = 0;
|
||||
flags.ui32.HostAccess = !DeviceLocal;
|
||||
flags.ui32.ExecuteAccess = 1;
|
||||
flags.ui32.NonPaged = nonPaged;
|
||||
flags.ui32.PageSize = HSA_PAGE_SIZE_4KB;
|
||||
flags.ui32.CoarseGrain = DeviceLocal;
|
||||
flags.ui32.Uncached = Uncached;
|
||||
|
||||
size = ALIGN_UP(size, align);
|
||||
|
||||
if (DeviceLocal && !hsakmt_zfb_support)
|
||||
mem = hsakmt_fmm_allocate_device(gpu_id, NodeId, mem, size, 0, flags);
|
||||
else {
|
||||
/* VRAM under ZFB mode should be supported here without any
|
||||
* additional code
|
||||
*/
|
||||
/* Get the closest cpu_id to GPU NodeId for system memory allocation
|
||||
* nonPaged=0 system memory allocation uses GTT path
|
||||
*/
|
||||
if (!nonPaged) {
|
||||
cpu_id = hsakmt_get_direct_link_cpu(NodeId);
|
||||
if (cpu_id == INVALID_NODEID) {
|
||||
flags.ui32.NoNUMABind = 1;
|
||||
cpu_id = 0;
|
||||
}
|
||||
}
|
||||
mem = hsakmt_fmm_allocate_host(gpu_id, cpu_id, mem, size, 0, flags);
|
||||
}
|
||||
|
||||
if (!mem) {
|
||||
pr_err("Alloc %s memory failed size %d\n",
|
||||
DeviceLocal ? "VRAM" : "GTT", size);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
if (NodeId != 0) {
|
||||
uint32_t nodes_array[1] = {NodeId};
|
||||
HsaMemMapFlags map_flags = {0};
|
||||
HSAKMT_STATUS result;
|
||||
|
||||
result = hsaKmtMapMemoryToGPUNodes(mem, size, &gpu_va, map_flags, 1, nodes_array);
|
||||
if (result != HSAKMT_STATUS_SUCCESS) {
|
||||
hsaKmtFreeMemory(mem, size);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
return mem;
|
||||
}
|
||||
|
||||
if (hsaKmtMapMemoryToGPU(mem, size, &gpu_va) != HSAKMT_STATUS_SUCCESS) {
|
||||
hsaKmtFreeMemory(mem, size);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
return mem;
|
||||
}
|
||||
|
||||
void hsakmt_free_exec_aligned_memory_gpu(void *addr, uint32_t size, uint32_t align)
|
||||
{
|
||||
size = ALIGN_UP(size, align);
|
||||
|
||||
if (hsaKmtUnmapMemoryToGPU(addr) == HSAKMT_STATUS_SUCCESS)
|
||||
hsaKmtFreeMemory(addr, size);
|
||||
}
|
||||
|
||||
/*
|
||||
* Allocates memory aligned to sysconf(_SC_PAGESIZE)
|
||||
*/
|
||||
static void *allocate_exec_aligned_memory(uint32_t size,
|
||||
bool use_ats,
|
||||
uint32_t gpu_id,
|
||||
uint32_t NodeId,
|
||||
bool nonPaged,
|
||||
bool DeviceLocal,
|
||||
bool Uncached)
|
||||
{
|
||||
if (!use_ats)
|
||||
return hsakmt_allocate_exec_aligned_memory_gpu(size, PAGE_SIZE, gpu_id, NodeId,
|
||||
nonPaged, DeviceLocal,
|
||||
Uncached);
|
||||
return allocate_exec_aligned_memory_cpu(size);
|
||||
}
|
||||
|
||||
static void free_exec_aligned_memory(void *addr, uint32_t size, uint32_t align,
|
||||
bool use_ats)
|
||||
{
|
||||
if (!use_ats)
|
||||
hsakmt_free_exec_aligned_memory_gpu(addr, size, align);
|
||||
else
|
||||
munmap(addr, size);
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS register_svm_range(void *mem, uint32_t size,
|
||||
uint32_t gpuNode, uint32_t prefetchNode,
|
||||
uint32_t preferredNode, bool alwaysMapped)
|
||||
{
|
||||
HSA_SVM_ATTRIBUTE *attrs;
|
||||
HSAuint64 s_attr;
|
||||
HSAuint32 nattr;
|
||||
HSAuint32 flags;
|
||||
|
||||
flags = HSA_SVM_FLAG_HOST_ACCESS | HSA_SVM_FLAG_GPU_EXEC;
|
||||
|
||||
if (alwaysMapped) {
|
||||
CHECK_KFD_MINOR_VERSION(11);
|
||||
flags |= HSA_SVM_FLAG_GPU_ALWAYS_MAPPED;
|
||||
}
|
||||
|
||||
nattr = 6;
|
||||
s_attr = sizeof(*attrs) * nattr;
|
||||
attrs = (HSA_SVM_ATTRIBUTE *)alloca(s_attr);
|
||||
|
||||
attrs[0].type = HSA_SVM_ATTR_PREFETCH_LOC;
|
||||
attrs[0].value = prefetchNode;
|
||||
attrs[1].type = HSA_SVM_ATTR_PREFERRED_LOC;
|
||||
attrs[1].value = preferredNode;
|
||||
attrs[2].type = HSA_SVM_ATTR_CLR_FLAGS;
|
||||
attrs[2].value = ~flags;
|
||||
attrs[3].type = HSA_SVM_ATTR_SET_FLAGS;
|
||||
attrs[3].value = flags;
|
||||
attrs[4].type = HSA_SVM_ATTR_ACCESS;
|
||||
attrs[4].value = gpuNode;
|
||||
attrs[5].type = HSA_SVM_ATTR_GRANULARITY;
|
||||
attrs[5].value = 0xFF;
|
||||
|
||||
return hsaKmtSVMSetAttr(mem, size, nattr, attrs);
|
||||
}
|
||||
|
||||
static void free_queue(struct queue *q)
|
||||
{
|
||||
if (q->eop_buffer)
|
||||
free_exec_aligned_memory(q->eop_buffer,
|
||||
q->eop_buffer_size,
|
||||
PAGE_SIZE, q->use_ats);
|
||||
if (q->unified_ctx_save_restore)
|
||||
munmap(q->ctx_save_restore, q->total_mem_alloc_size);
|
||||
else if (q->ctx_save_restore)
|
||||
free_exec_aligned_memory(q->ctx_save_restore,
|
||||
q->total_mem_alloc_size,
|
||||
PAGE_SIZE, q->use_ats);
|
||||
|
||||
free_exec_aligned_memory((void *)q, sizeof(*q), PAGE_SIZE, q->use_ats);
|
||||
}
|
||||
|
||||
static inline void fill_cwsr_header(struct queue *q, void *addr,
|
||||
HsaEvent *Event, volatile HSAint64 *ErrPayload, HSAuint32 NumXcc)
|
||||
{
|
||||
uint32_t i;
|
||||
HsaUserContextSaveAreaHeader *header;
|
||||
|
||||
for (i = 0; i < NumXcc; i++) {
|
||||
header = (HsaUserContextSaveAreaHeader *)
|
||||
((uintptr_t)addr + (i * q->ctx_save_restore_size));
|
||||
header->ErrorEventId = 0;
|
||||
if (Event)
|
||||
header->ErrorEventId = Event->EventId;
|
||||
header->ErrorReason = ErrPayload;
|
||||
header->DebugOffset = (NumXcc - i) * q->ctx_save_restore_size;
|
||||
header->DebugSize = q->debug_memory_size * NumXcc;
|
||||
}
|
||||
}
|
||||
|
||||
static int handle_concrete_asic(struct queue *q,
|
||||
struct kfd_ioctl_create_queue_args *args,
|
||||
uint32_t gpu_id,
|
||||
uint32_t NodeId,
|
||||
HsaEvent *Event,
|
||||
volatile HSAint64 *ErrPayload)
|
||||
{
|
||||
bool ret;
|
||||
|
||||
if (args->queue_type == KFD_IOC_QUEUE_TYPE_SDMA ||
|
||||
args->queue_type == KFD_IOC_QUEUE_TYPE_SDMA_XGMI)
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
if (q->eop_buffer_size > 0) {
|
||||
pr_info("Allocating VRAM for EOP\n");
|
||||
q->eop_buffer = allocate_exec_aligned_memory(q->eop_buffer_size,
|
||||
q->use_ats, gpu_id,
|
||||
NodeId, true, true, /* Unused for VRAM */false);
|
||||
if (!q->eop_buffer)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
args->eop_buffer_address = (uintptr_t)q->eop_buffer;
|
||||
args->eop_buffer_size = q->eop_buffer_size;
|
||||
}
|
||||
|
||||
ret = update_ctx_save_restore_size(NodeId, q);
|
||||
|
||||
if (ret) {
|
||||
HsaNodeProperties node;
|
||||
|
||||
if (hsaKmtGetNodeProperties(NodeId, &node))
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
args->ctx_save_restore_size = q->ctx_save_restore_size;
|
||||
args->ctl_stack_size = q->ctl_stack_size;
|
||||
|
||||
/* Total memory to be allocated is =
|
||||
* (Control Stack size + WG size +
|
||||
* Debug memory area size) * num_xcc
|
||||
*/
|
||||
q->total_mem_alloc_size = (q->ctx_save_restore_size +
|
||||
q->debug_memory_size) * node.NumXcc;
|
||||
|
||||
/* Allocate unified memory for context save restore
|
||||
* area on dGPU.
|
||||
*/
|
||||
if (!q->use_ats && hsakmt_is_svm_api_supported) {
|
||||
uint32_t size = PAGE_ALIGN_UP(q->total_mem_alloc_size);
|
||||
|
||||
pr_info("Allocating GTT for CWSR\n");
|
||||
void *addr = hsakmt_mmap_allocate_aligned(PROT_READ | PROT_WRITE,
|
||||
MAP_ANONYMOUS | MAP_PRIVATE,
|
||||
size, GPU_HUGE_PAGE_SIZE, 0,
|
||||
0, (void *)LONG_MAX);
|
||||
if (!addr) {
|
||||
pr_err("mmap failed to alloc ctx area size 0x%x: %s\n",
|
||||
size, strerror(errno));
|
||||
} else {
|
||||
/*
|
||||
* To avoid fork child process COW MMU notifier
|
||||
* callback evict parent process queues.
|
||||
*/
|
||||
if (madvise(addr, size, MADV_DONTFORK))
|
||||
pr_err("madvise failed -%d\n", errno);
|
||||
|
||||
fill_cwsr_header(q, addr, Event, ErrPayload, node.NumXcc);
|
||||
|
||||
HSAKMT_STATUS r = register_svm_range(addr, size,
|
||||
NodeId, NodeId, 0, true);
|
||||
|
||||
if (r == HSAKMT_STATUS_SUCCESS) {
|
||||
q->ctx_save_restore = addr;
|
||||
q->unified_ctx_save_restore = true;
|
||||
} else {
|
||||
munmap(addr, size);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!q->unified_ctx_save_restore) {
|
||||
q->ctx_save_restore = allocate_exec_aligned_memory(
|
||||
q->total_mem_alloc_size,
|
||||
q->use_ats, gpu_id, NodeId,
|
||||
false, false, false);
|
||||
|
||||
if (!q->ctx_save_restore)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
fill_cwsr_header(q, q->ctx_save_restore, Event, ErrPayload, node.NumXcc);
|
||||
}
|
||||
|
||||
args->ctx_save_restore_address = (uintptr_t)q->ctx_save_restore;
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
/* A map to translate thunk queue priority (-3 to +3)
|
||||
* to KFD queue priority (0 to 15)
|
||||
* Indexed by thunk_queue_priority+3
|
||||
*/
|
||||
static uint32_t priority_map[] = {0, 3, 5, 7, 9, 11, 15};
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtCreateQueue(HSAuint32 NodeId,
|
||||
HSA_QUEUE_TYPE Type,
|
||||
HSAuint32 QueuePercentage,
|
||||
HSA_QUEUE_PRIORITY Priority,
|
||||
void *QueueAddress,
|
||||
HSAuint64 QueueSizeInBytes,
|
||||
HsaEvent *Event,
|
||||
HsaQueueResource *QueueResource)
|
||||
{
|
||||
if (Type == HSA_QUEUE_SDMA_BY_ENG_ID)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return hsaKmtCreateQueueExt(NodeId, Type, QueuePercentage, Priority, 0,
|
||||
QueueAddress, QueueSizeInBytes, Event,
|
||||
QueueResource);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtCreateQueueExt(HSAuint32 NodeId,
|
||||
HSA_QUEUE_TYPE Type,
|
||||
HSAuint32 QueuePercentage,
|
||||
HSA_QUEUE_PRIORITY Priority,
|
||||
HSAuint32 SdmaEngineId,
|
||||
void *QueueAddress,
|
||||
HSAuint64 QueueSizeInBytes,
|
||||
HsaEvent *Event,
|
||||
HsaQueueResource *QueueResource)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpu_id;
|
||||
uint64_t doorbell_mmap_offset;
|
||||
unsigned int doorbell_offset;
|
||||
int err;
|
||||
HsaNodeProperties props;
|
||||
uint32_t cu_num, i;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (Priority < HSA_QUEUE_PRIORITY_MINIMUM ||
|
||||
Priority > HSA_QUEUE_PRIORITY_MAXIMUM)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
result = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
struct queue *q = allocate_exec_aligned_memory(sizeof(*q),
|
||||
false, gpu_id, NodeId, true, false, true);
|
||||
if (!q)
|
||||
return HSAKMT_STATUS_NO_MEMORY;
|
||||
|
||||
memset(q, 0, sizeof(*q));
|
||||
|
||||
q->gfxv = hsakmt_get_gfxv_by_node_id(NodeId);
|
||||
q->use_ats = false;
|
||||
|
||||
if (q->gfxv == GFX_VERSION_TONGA)
|
||||
q->eop_buffer_size = TONGA_PAGE_SIZE;
|
||||
else if ((q->gfxv & ~(0xff)) == GFX_VERSION_AQUA_VANJARAM)
|
||||
q->eop_buffer_size = ((Type == HSA_QUEUE_COMPUTE) ? 4096 : 0);
|
||||
else if (q->gfxv >= 0x80000)
|
||||
q->eop_buffer_size = 4096;
|
||||
|
||||
/* By default, CUs are all turned on. Initialize cu_mask to '1
|
||||
* for all CU bits.
|
||||
*/
|
||||
if (hsaKmtGetNodeProperties(NodeId, &props))
|
||||
q->cu_mask_count = 0;
|
||||
else {
|
||||
cu_num = props.NumFComputeCores / props.NumSIMDPerCU;
|
||||
/* cu_mask_count counts bits. It must be multiple of 32 */
|
||||
q->cu_mask_count = ALIGN_UP_32(cu_num, 32);
|
||||
for (i = 0; i < cu_num; i++)
|
||||
q->cu_mask[i/32] |= (1 << (i % 32));
|
||||
}
|
||||
|
||||
struct kfd_ioctl_create_queue_args args = {0};
|
||||
|
||||
args.gpu_id = gpu_id;
|
||||
|
||||
switch (Type) {
|
||||
case HSA_QUEUE_COMPUTE:
|
||||
args.queue_type = KFD_IOC_QUEUE_TYPE_COMPUTE;
|
||||
break;
|
||||
case HSA_QUEUE_SDMA:
|
||||
args.queue_type = KFD_IOC_QUEUE_TYPE_SDMA;
|
||||
break;
|
||||
case HSA_QUEUE_SDMA_XGMI:
|
||||
args.queue_type = KFD_IOC_QUEUE_TYPE_SDMA_XGMI;
|
||||
break;
|
||||
case HSA_QUEUE_SDMA_BY_ENG_ID:
|
||||
args.queue_type = KFD_IOC_QUEUE_TYPE_SDMA_BY_ENG_ID;
|
||||
break;
|
||||
case HSA_QUEUE_COMPUTE_AQL:
|
||||
args.queue_type = KFD_IOC_QUEUE_TYPE_COMPUTE_AQL;
|
||||
break;
|
||||
default:
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
}
|
||||
|
||||
if (Type != HSA_QUEUE_COMPUTE_AQL) {
|
||||
QueueResource->QueueRptrValue = (uintptr_t)&q->rptr;
|
||||
QueueResource->QueueWptrValue = (uintptr_t)&q->wptr;
|
||||
}
|
||||
|
||||
err = handle_concrete_asic(q, &args, gpu_id, NodeId, Event, QueueResource->ErrorReason);
|
||||
if (err != HSAKMT_STATUS_SUCCESS) {
|
||||
free_queue(q);
|
||||
return err;
|
||||
}
|
||||
|
||||
args.read_pointer_address = QueueResource->QueueRptrValue;
|
||||
args.write_pointer_address = QueueResource->QueueWptrValue;
|
||||
args.ring_base_address = (uintptr_t)QueueAddress;
|
||||
args.ring_size = QueueSizeInBytes;
|
||||
args.queue_percentage = QueuePercentage;
|
||||
args.queue_priority = priority_map[Priority+3];
|
||||
args.sdma_engine_id = SdmaEngineId;
|
||||
|
||||
err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_CREATE_QUEUE, &args);
|
||||
|
||||
if (err == -1) {
|
||||
free_queue(q);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
q->queue_id = args.queue_id;
|
||||
|
||||
if (IS_SOC15(q->gfxv)) {
|
||||
HSAuint64 mask = DOORBELLS_PAGE_SIZE(DOORBELL_SIZE(q->gfxv)) - 1;
|
||||
|
||||
/* On SOC15 chips, the doorbell offset within the
|
||||
* doorbell page is included in the doorbell offset
|
||||
* returned by KFD. This allows CP queue doorbells to be
|
||||
* allocated dynamically (while SDMA queue doorbells fixed)
|
||||
* rather than based on the its process queue ID.
|
||||
*/
|
||||
doorbell_mmap_offset = args.doorbell_offset & ~mask;
|
||||
doorbell_offset = args.doorbell_offset & mask;
|
||||
} else {
|
||||
/* On older chips, the doorbell offset within the
|
||||
* doorbell page is based on the queue ID.
|
||||
*/
|
||||
doorbell_mmap_offset = args.doorbell_offset;
|
||||
doorbell_offset = q->queue_id * DOORBELL_SIZE(q->gfxv);
|
||||
}
|
||||
|
||||
err = map_doorbell(NodeId, gpu_id, doorbell_mmap_offset);
|
||||
if (err != HSAKMT_STATUS_SUCCESS) {
|
||||
hsaKmtDestroyQueue(q->queue_id);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
QueueResource->QueueId = PORT_VPTR_TO_UINT64(q);
|
||||
QueueResource->Queue_DoorBell = VOID_PTR_ADD(doorbells[NodeId].mapping,
|
||||
doorbell_offset);
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtUpdateQueue(HSA_QUEUEID QueueId,
|
||||
HSAuint32 QueuePercentage,
|
||||
HSA_QUEUE_PRIORITY Priority,
|
||||
void *QueueAddress,
|
||||
HSAuint64 QueueSize,
|
||||
HsaEvent *Event)
|
||||
{
|
||||
struct kfd_ioctl_update_queue_args arg = {0};
|
||||
struct queue *q = PORT_UINT64_TO_VPTR(QueueId);
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (Priority < HSA_QUEUE_PRIORITY_MINIMUM ||
|
||||
Priority > HSA_QUEUE_PRIORITY_MAXIMUM)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (!q)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
arg.queue_id = (HSAuint32)q->queue_id;
|
||||
arg.ring_base_address = (uintptr_t)QueueAddress;
|
||||
arg.ring_size = QueueSize;
|
||||
arg.queue_percentage = QueuePercentage;
|
||||
arg.queue_priority = priority_map[Priority+3];
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_UPDATE_QUEUE, &arg);
|
||||
|
||||
if (err == -1)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtDestroyQueue(HSA_QUEUEID QueueId)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
struct queue *q = PORT_UINT64_TO_VPTR(QueueId);
|
||||
struct kfd_ioctl_destroy_queue_args args = {0};
|
||||
|
||||
if (!q)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
args.queue_id = q->queue_id;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_DESTROY_QUEUE, &args);
|
||||
|
||||
if (err == -1) {
|
||||
pr_err("Failed to destroy queue: %s\n", strerror(errno));
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
free_queue(q);
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtSetQueueCUMask(HSA_QUEUEID QueueId,
|
||||
HSAuint32 CUMaskCount,
|
||||
HSAuint32 *QueueCUMask)
|
||||
{
|
||||
struct queue *q = PORT_UINT64_TO_VPTR(QueueId);
|
||||
struct kfd_ioctl_set_cu_mask_args args = {0};
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (CUMaskCount == 0 || !QueueCUMask || ((CUMaskCount % 32) != 0))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
args.queue_id = q->queue_id;
|
||||
args.num_cu_mask = CUMaskCount;
|
||||
args.cu_mask_ptr = (uintptr_t)QueueCUMask;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SET_CU_MASK, &args);
|
||||
|
||||
if (err == -1)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
memcpy(q->cu_mask, QueueCUMask, CUMaskCount / 8);
|
||||
q->cu_mask_count = CUMaskCount;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS
|
||||
HSAKMTAPI
|
||||
hsaKmtGetQueueInfo(
|
||||
HSA_QUEUEID QueueId,
|
||||
HsaQueueInfo *QueueInfo
|
||||
)
|
||||
{
|
||||
struct queue *q = PORT_UINT64_TO_VPTR(QueueId);
|
||||
struct kfd_ioctl_get_queue_wave_state_args args = {0};
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
if (QueueInfo == NULL || q == NULL)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
if (q->ctx_save_restore == NULL)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
args.queue_id = q->queue_id;
|
||||
args.ctl_stack_address = (uintptr_t)q->ctx_save_restore;
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_GET_QUEUE_WAVE_STATE, &args) < 0)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
QueueInfo->ControlStackTop = (void *)(args.ctl_stack_address +
|
||||
q->ctl_stack_size - args.ctl_stack_used_size);
|
||||
QueueInfo->UserContextSaveArea = (void *)
|
||||
(args.ctl_stack_address + q->ctl_stack_size);
|
||||
QueueInfo->SaveAreaSizeInBytes = args.save_area_used_size;
|
||||
QueueInfo->ControlStackUsedInBytes = args.ctl_stack_used_size;
|
||||
QueueInfo->NumCUAssigned = q->cu_mask_count;
|
||||
QueueInfo->CUMaskInfo = q->cu_mask;
|
||||
QueueInfo->QueueDetailError = 0;
|
||||
QueueInfo->QueueTypeExtended = 0;
|
||||
QueueInfo->SaveAreaHeader = q->ctx_save_restore;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtSetTrapHandler(HSAuint32 Node,
|
||||
void *TrapHandlerBaseAddress,
|
||||
HSAuint64 TrapHandlerSizeInBytes,
|
||||
void *TrapBufferBaseAddress,
|
||||
HSAuint64 TrapBufferSizeInBytes)
|
||||
{
|
||||
struct kfd_ioctl_set_trap_handler_args args = {0};
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpu_id;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
result = hsakmt_validate_nodeid(Node, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
args.gpu_id = gpu_id;
|
||||
args.tba_addr = (uintptr_t)TrapHandlerBaseAddress;
|
||||
args.tma_addr = (uintptr_t)TrapBufferBaseAddress;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SET_TRAP_HANDLER, &args);
|
||||
|
||||
return (err == -1) ? HSAKMT_STATUS_ERROR : HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
uint32_t *hsakmt_convert_queue_ids(HSAuint32 NumQueues, HSA_QUEUEID *Queues)
|
||||
{
|
||||
uint32_t *queue_ids_ptr;
|
||||
unsigned int i;
|
||||
|
||||
if (NumQueues == 0 || Queues == NULL)
|
||||
return NULL;
|
||||
|
||||
queue_ids_ptr = malloc(NumQueues * sizeof(uint32_t));
|
||||
if (!queue_ids_ptr)
|
||||
return NULL;
|
||||
|
||||
for (i = 0; i < NumQueues; i++) {
|
||||
struct queue *q = PORT_UINT64_TO_VPTR(Queues[i]);
|
||||
|
||||
if (q == NULL) {
|
||||
free(queue_ids_ptr);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
queue_ids_ptr[i] = q->queue_id;
|
||||
}
|
||||
return queue_ids_ptr;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS
|
||||
HSAKMTAPI
|
||||
hsaKmtAllocQueueGWS(
|
||||
HSA_QUEUEID QueueId,
|
||||
HSAuint32 nGWS,
|
||||
HSAuint32 *firstGWS)
|
||||
{
|
||||
struct kfd_ioctl_alloc_queue_gws_args args = {0};
|
||||
struct queue *q = PORT_UINT64_TO_VPTR(QueueId);
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
args.queue_id = (HSAuint32)q->queue_id;
|
||||
args.num_gws = nGWS;
|
||||
|
||||
int err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_ALLOC_QUEUE_GWS, &args);
|
||||
|
||||
if (!err && firstGWS)
|
||||
*firstGWS = args.first_gws;
|
||||
|
||||
if (!err)
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
else if (errno == EINVAL)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
else if (errno == EBUSY)
|
||||
return HSAKMT_STATUS_OUT_OF_RESOURCES;
|
||||
else if (errno == ENODEV)
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
else
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
@@ -0,0 +1,402 @@
|
||||
/*
|
||||
* Copyright (C) 2002-2018 Igor Sysoev
|
||||
* Copyright (C) 2011-2018 Nginx, Inc.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* 1. Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* 2. Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
* ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
|
||||
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
|
||||
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
|
||||
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
|
||||
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
|
||||
* SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#include "rbtree.h"
|
||||
|
||||
static inline void rbtree_left_rotate(rbtree_node_t **root,
|
||||
rbtree_node_t *sentinel, rbtree_node_t *node);
|
||||
static inline void rbtree_right_rotate(rbtree_node_t **root,
|
||||
rbtree_node_t *sentinel, rbtree_node_t *node);
|
||||
|
||||
static void
|
||||
hsakmt_rbtree_insert_value(rbtree_node_t *temp, rbtree_node_t *node,
|
||||
rbtree_node_t *sentinel)
|
||||
{
|
||||
rbtree_node_t **p;
|
||||
|
||||
for ( ;; ) {
|
||||
|
||||
p = rbtree_key_compare(LKP_ALL, &node->key, &temp->key) < 0 ?
|
||||
&temp->left : &temp->right;
|
||||
|
||||
if (*p == sentinel) {
|
||||
break;
|
||||
}
|
||||
|
||||
temp = *p;
|
||||
}
|
||||
|
||||
*p = node;
|
||||
node->parent = temp;
|
||||
node->left = sentinel;
|
||||
node->right = sentinel;
|
||||
rbt_red(node);
|
||||
}
|
||||
|
||||
|
||||
void
|
||||
hsakmt_rbtree_insert(rbtree_t *tree, rbtree_node_t *node)
|
||||
{
|
||||
rbtree_node_t **root, *temp, *sentinel;
|
||||
|
||||
/* a binary tree insert */
|
||||
|
||||
root = &tree->root;
|
||||
sentinel = &tree->sentinel;
|
||||
|
||||
if (*root == sentinel) {
|
||||
node->parent = NULL;
|
||||
node->left = sentinel;
|
||||
node->right = sentinel;
|
||||
rbt_black(node);
|
||||
*root = node;
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
hsakmt_rbtree_insert_value(*root, node, sentinel);
|
||||
|
||||
/* re-balance tree */
|
||||
|
||||
while (node != *root && rbt_is_red(node->parent)) {
|
||||
|
||||
if (node->parent == node->parent->parent->left) {
|
||||
temp = node->parent->parent->right;
|
||||
|
||||
if (rbt_is_red(temp)) {
|
||||
rbt_black(node->parent);
|
||||
rbt_black(temp);
|
||||
rbt_red(node->parent->parent);
|
||||
node = node->parent->parent;
|
||||
|
||||
} else {
|
||||
if (node == node->parent->right) {
|
||||
node = node->parent;
|
||||
rbtree_left_rotate(root, sentinel, node);
|
||||
}
|
||||
|
||||
rbt_black(node->parent);
|
||||
rbt_red(node->parent->parent);
|
||||
rbtree_right_rotate(root, sentinel, node->parent->parent);
|
||||
}
|
||||
|
||||
} else {
|
||||
temp = node->parent->parent->left;
|
||||
|
||||
if (rbt_is_red(temp)) {
|
||||
rbt_black(node->parent);
|
||||
rbt_black(temp);
|
||||
rbt_red(node->parent->parent);
|
||||
node = node->parent->parent;
|
||||
|
||||
} else {
|
||||
if (node == node->parent->left) {
|
||||
node = node->parent;
|
||||
rbtree_right_rotate(root, sentinel, node);
|
||||
}
|
||||
|
||||
rbt_black(node->parent);
|
||||
rbt_red(node->parent->parent);
|
||||
rbtree_left_rotate(root, sentinel, node->parent->parent);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
rbt_black(*root);
|
||||
}
|
||||
|
||||
|
||||
void
|
||||
hsakmt_rbtree_delete(rbtree_t *tree, rbtree_node_t *node)
|
||||
{
|
||||
unsigned int red;
|
||||
rbtree_node_t **root, *sentinel, *subst, *temp, *w;
|
||||
|
||||
/* a binary tree delete */
|
||||
|
||||
root = &tree->root;
|
||||
sentinel = &tree->sentinel;
|
||||
|
||||
if (node->left == sentinel) {
|
||||
temp = node->right;
|
||||
subst = node;
|
||||
|
||||
} else if (node->right == sentinel) {
|
||||
temp = node->left;
|
||||
subst = node;
|
||||
|
||||
} else {
|
||||
subst = rbtree_min(node->right, sentinel);
|
||||
|
||||
if (subst->left != sentinel) {
|
||||
temp = subst->left;
|
||||
} else {
|
||||
temp = subst->right;
|
||||
}
|
||||
}
|
||||
|
||||
if (subst == *root) {
|
||||
*root = temp;
|
||||
rbt_black(temp);
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
red = rbt_is_red(subst);
|
||||
|
||||
if (subst == subst->parent->left) {
|
||||
subst->parent->left = temp;
|
||||
|
||||
} else {
|
||||
subst->parent->right = temp;
|
||||
}
|
||||
|
||||
if (subst == node) {
|
||||
|
||||
temp->parent = subst->parent;
|
||||
|
||||
} else {
|
||||
|
||||
if (subst->parent == node) {
|
||||
temp->parent = subst;
|
||||
|
||||
} else {
|
||||
temp->parent = subst->parent;
|
||||
}
|
||||
|
||||
subst->left = node->left;
|
||||
subst->right = node->right;
|
||||
subst->parent = node->parent;
|
||||
rbt_copy_color(subst, node);
|
||||
|
||||
if (node == *root) {
|
||||
*root = subst;
|
||||
|
||||
} else {
|
||||
if (node == node->parent->left) {
|
||||
node->parent->left = subst;
|
||||
} else {
|
||||
node->parent->right = subst;
|
||||
}
|
||||
}
|
||||
|
||||
if (subst->left != sentinel) {
|
||||
subst->left->parent = subst;
|
||||
}
|
||||
|
||||
if (subst->right != sentinel) {
|
||||
subst->right->parent = subst;
|
||||
}
|
||||
}
|
||||
|
||||
if (red) {
|
||||
return;
|
||||
}
|
||||
|
||||
/* a delete fixup */
|
||||
|
||||
while (temp != *root && rbt_is_black(temp)) {
|
||||
|
||||
if (temp == temp->parent->left) {
|
||||
w = temp->parent->right;
|
||||
|
||||
if (rbt_is_red(w)) {
|
||||
rbt_black(w);
|
||||
rbt_red(temp->parent);
|
||||
rbtree_left_rotate(root, sentinel, temp->parent);
|
||||
w = temp->parent->right;
|
||||
}
|
||||
|
||||
if (rbt_is_black(w->left) && rbt_is_black(w->right)) {
|
||||
rbt_red(w);
|
||||
temp = temp->parent;
|
||||
|
||||
} else {
|
||||
if (rbt_is_black(w->right)) {
|
||||
rbt_black(w->left);
|
||||
rbt_red(w);
|
||||
rbtree_right_rotate(root, sentinel, w);
|
||||
w = temp->parent->right;
|
||||
}
|
||||
|
||||
rbt_copy_color(w, temp->parent);
|
||||
rbt_black(temp->parent);
|
||||
rbt_black(w->right);
|
||||
rbtree_left_rotate(root, sentinel, temp->parent);
|
||||
temp = *root;
|
||||
}
|
||||
|
||||
} else {
|
||||
w = temp->parent->left;
|
||||
|
||||
if (rbt_is_red(w)) {
|
||||
rbt_black(w);
|
||||
rbt_red(temp->parent);
|
||||
rbtree_right_rotate(root, sentinel, temp->parent);
|
||||
w = temp->parent->left;
|
||||
}
|
||||
|
||||
if (rbt_is_black(w->left) && rbt_is_black(w->right)) {
|
||||
rbt_red(w);
|
||||
temp = temp->parent;
|
||||
|
||||
} else {
|
||||
if (rbt_is_black(w->left)) {
|
||||
rbt_black(w->right);
|
||||
rbt_red(w);
|
||||
rbtree_left_rotate(root, sentinel, w);
|
||||
w = temp->parent->left;
|
||||
}
|
||||
|
||||
rbt_copy_color(w, temp->parent);
|
||||
rbt_black(temp->parent);
|
||||
rbt_black(w->left);
|
||||
rbtree_right_rotate(root, sentinel, temp->parent);
|
||||
temp = *root;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
rbt_black(temp);
|
||||
}
|
||||
|
||||
|
||||
static inline void
|
||||
rbtree_left_rotate(rbtree_node_t **root, rbtree_node_t *sentinel,
|
||||
rbtree_node_t *node)
|
||||
{
|
||||
rbtree_node_t *temp;
|
||||
|
||||
temp = node->right;
|
||||
node->right = temp->left;
|
||||
|
||||
if (temp->left != sentinel) {
|
||||
temp->left->parent = node;
|
||||
}
|
||||
|
||||
temp->parent = node->parent;
|
||||
|
||||
if (node == *root) {
|
||||
*root = temp;
|
||||
|
||||
} else if (node == node->parent->left) {
|
||||
node->parent->left = temp;
|
||||
|
||||
} else {
|
||||
node->parent->right = temp;
|
||||
}
|
||||
|
||||
temp->left = node;
|
||||
node->parent = temp;
|
||||
}
|
||||
|
||||
|
||||
static inline void
|
||||
rbtree_right_rotate(rbtree_node_t **root, rbtree_node_t *sentinel,
|
||||
rbtree_node_t *node)
|
||||
{
|
||||
rbtree_node_t *temp;
|
||||
|
||||
temp = node->left;
|
||||
node->left = temp->right;
|
||||
|
||||
if (temp->right != sentinel) {
|
||||
temp->right->parent = node;
|
||||
}
|
||||
|
||||
temp->parent = node->parent;
|
||||
|
||||
if (node == *root) {
|
||||
*root = temp;
|
||||
|
||||
} else if (node == node->parent->right) {
|
||||
node->parent->right = temp;
|
||||
|
||||
} else {
|
||||
node->parent->left = temp;
|
||||
}
|
||||
|
||||
temp->right = node;
|
||||
node->parent = temp;
|
||||
}
|
||||
|
||||
|
||||
rbtree_node_t *
|
||||
hsakmt_rbtree_next(rbtree_t *tree, rbtree_node_t *node)
|
||||
{
|
||||
rbtree_node_t *root, *sentinel, *parent;
|
||||
|
||||
sentinel = &tree->sentinel;
|
||||
|
||||
if (node->right != sentinel) {
|
||||
return rbtree_min(node->right, sentinel);
|
||||
}
|
||||
|
||||
root = tree->root;
|
||||
|
||||
for ( ;; ) {
|
||||
parent = node->parent;
|
||||
|
||||
if (node == root) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
if (node == parent->left) {
|
||||
return parent;
|
||||
}
|
||||
|
||||
node = parent;
|
||||
}
|
||||
}
|
||||
|
||||
rbtree_node_t *
|
||||
hsakmt_rbtree_prev(rbtree_t *tree, rbtree_node_t *node)
|
||||
{
|
||||
rbtree_node_t *root, *sentinel, *parent;
|
||||
|
||||
sentinel = &tree->sentinel;
|
||||
|
||||
if (node->left != sentinel) {
|
||||
return rbtree_max(node->left, sentinel);
|
||||
}
|
||||
|
||||
root = tree->root;
|
||||
|
||||
for ( ;; ) {
|
||||
parent = node->parent;
|
||||
|
||||
if (node == root) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
if (node == parent->right) {
|
||||
return parent;
|
||||
}
|
||||
|
||||
node = parent;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
/*
|
||||
* Copyright (C) 2002-2018 Igor Sysoev
|
||||
* Copyright (C) 2011-2018 Nginx, Inc.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* 1. Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* 2. Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
|
||||
* ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||
* ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
|
||||
* FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
||||
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
|
||||
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
|
||||
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
|
||||
* LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
|
||||
* OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
|
||||
* SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#ifndef _RBTREE_H_
|
||||
#define _RBTREE_H_
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#include <inttypes.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/time.h>
|
||||
#include <errno.h>
|
||||
#include "rbtree_amd.h"
|
||||
|
||||
typedef struct rbtree_node_s rbtree_node_t;
|
||||
|
||||
struct rbtree_node_s {
|
||||
rbtree_key_t key;
|
||||
rbtree_node_t *left;
|
||||
rbtree_node_t *right;
|
||||
rbtree_node_t *parent;
|
||||
unsigned char color;
|
||||
unsigned char data;
|
||||
};
|
||||
|
||||
typedef struct rbtree_s rbtree_t;
|
||||
|
||||
struct rbtree_s {
|
||||
rbtree_node_t *root;
|
||||
rbtree_node_t sentinel;
|
||||
};
|
||||
|
||||
#define rbtree_init(tree) \
|
||||
rbtree_sentinel_init(&(tree)->sentinel); \
|
||||
(tree)->root = &(tree)->sentinel;
|
||||
|
||||
void hsakmt_rbtree_insert(rbtree_t *tree, rbtree_node_t *node);
|
||||
void hsakmt_rbtree_delete(rbtree_t *tree, rbtree_node_t *node);
|
||||
rbtree_node_t *hsakmt_rbtree_prev(rbtree_t *tree,
|
||||
rbtree_node_t *node);
|
||||
rbtree_node_t *hsakmt_rbtree_next(rbtree_t *tree,
|
||||
rbtree_node_t *node);
|
||||
|
||||
#define rbt_red(node) ((node)->color = 1)
|
||||
#define rbt_black(node) ((node)->color = 0)
|
||||
#define rbt_is_red(node) ((node)->color)
|
||||
#define rbt_is_black(node) (!rbt_is_red(node))
|
||||
#define rbt_copy_color(n1, n2) (n1->color = n2->color)
|
||||
|
||||
/* a sentinel must be black */
|
||||
|
||||
#define rbtree_sentinel_init(node) rbt_black(node)
|
||||
|
||||
static inline rbtree_node_t *
|
||||
rbtree_min(rbtree_node_t *node, rbtree_node_t *sentinel)
|
||||
{
|
||||
while (node->left != sentinel) {
|
||||
node = node->left;
|
||||
}
|
||||
|
||||
return node;
|
||||
}
|
||||
|
||||
#include "rbtree_amd.h"
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,155 @@
|
||||
/*
|
||||
* Copyright © 2018 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#ifndef _RBTREE_AMD_H_
|
||||
#define _RBTREE_AMD_H_
|
||||
|
||||
typedef struct rbtree_key_s rbtree_key_t;
|
||||
struct rbtree_key_s {
|
||||
#define ADDR_BIT 0
|
||||
#define SIZE_BIT 1
|
||||
unsigned long addr;
|
||||
unsigned long size;
|
||||
};
|
||||
#define BIT(x) (1<<(x))
|
||||
#define LKP_ALL (BIT(ADDR_BIT) | BIT(SIZE_BIT))
|
||||
#define LKP_ADDR (BIT(ADDR_BIT))
|
||||
#define LKP_ADDR_SIZE (BIT(ADDR_BIT) | BIT(SIZE_BIT))
|
||||
|
||||
static inline rbtree_key_t
|
||||
rbtree_key(unsigned long addr, unsigned long size)
|
||||
{
|
||||
return (rbtree_key_t){addr, size};
|
||||
}
|
||||
|
||||
/*
|
||||
* compare addr, size one by one
|
||||
*/
|
||||
static inline int
|
||||
rbtree_key_compare(unsigned int type, rbtree_key_t *key1, rbtree_key_t *key2)
|
||||
{
|
||||
if ((type & 1 << ADDR_BIT) && (key1->addr != key2->addr))
|
||||
return key1->addr > key2->addr ? 1 : -1;
|
||||
|
||||
if ((type & 1 << SIZE_BIT) && (key1->size != key2->size))
|
||||
return key1->size > key2->size ? 1 : -1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
#endif /*_RBTREE_AMD_H_*/
|
||||
|
||||
/*inlcude this file again with RBTREE_HELPER defined*/
|
||||
#ifndef RBTREE_HELPER
|
||||
#define RBTREE_HELPER
|
||||
#else
|
||||
#ifndef _RBTREE_AMD_H_HELPER_
|
||||
#define _RBTREE_AMD_H_HELPER_
|
||||
static inline rbtree_node_t *
|
||||
rbtree_max(rbtree_node_t *node, rbtree_node_t *sentinel)
|
||||
{
|
||||
while (node->right != sentinel)
|
||||
node = node->right;
|
||||
|
||||
return node;
|
||||
}
|
||||
|
||||
#define LEFT 0
|
||||
#define RIGHT 1
|
||||
#define MID 2
|
||||
static inline rbtree_node_t *
|
||||
rbtree_min_max(rbtree_t *tree, int lr)
|
||||
{
|
||||
rbtree_node_t *sentinel = &tree->sentinel;
|
||||
rbtree_node_t *node = tree->root;
|
||||
|
||||
if (node == sentinel)
|
||||
return NULL;
|
||||
|
||||
if (lr == LEFT)
|
||||
node = rbtree_min(node, sentinel);
|
||||
else if (lr == RIGHT)
|
||||
node = rbtree_max(node, sentinel);
|
||||
|
||||
return node;
|
||||
}
|
||||
|
||||
static inline rbtree_node_t *
|
||||
rbtree_node_any(rbtree_t *tree, int lmr)
|
||||
{
|
||||
rbtree_node_t *sentinel = &tree->sentinel;
|
||||
rbtree_node_t *node = tree->root;
|
||||
|
||||
if (node == sentinel)
|
||||
return NULL;
|
||||
|
||||
if (lmr == MID)
|
||||
return node;
|
||||
|
||||
return rbtree_min_max(tree, lmr);
|
||||
}
|
||||
|
||||
static inline rbtree_node_t *
|
||||
rbtree_lookup_nearest(rbtree_t *rbtree, rbtree_key_t *key,
|
||||
unsigned int type, int lr)
|
||||
{
|
||||
int rc;
|
||||
rbtree_node_t *node, *sentinel, *n = NULL;
|
||||
|
||||
node = rbtree->root;
|
||||
sentinel = &rbtree->sentinel;
|
||||
|
||||
while (node != sentinel) {
|
||||
rc = rbtree_key_compare(type, key, &node->key);
|
||||
|
||||
if (rc < 0) {
|
||||
if (lr == RIGHT)
|
||||
n = node;
|
||||
node = node->left;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (rc > 0) {
|
||||
if (lr == LEFT)
|
||||
n = node;
|
||||
node = node->right;
|
||||
continue;
|
||||
}
|
||||
|
||||
return node;
|
||||
}
|
||||
|
||||
return n;
|
||||
}
|
||||
|
||||
static inline rbtree_node_t *
|
||||
rbtree_lookup(rbtree_t *rbtree, rbtree_key_t *key,
|
||||
unsigned int type)
|
||||
{
|
||||
return rbtree_lookup_nearest(rbtree, key, type, -1);
|
||||
}
|
||||
#endif /*_RBTREE_AMD_H_HELPER_*/
|
||||
|
||||
#endif /*RBTREE_HELPER*/
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
/*
|
||||
* Copyright © 2020 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtSPMAcquire(HSAuint32 PreferredNode)
|
||||
{
|
||||
int ret;
|
||||
struct kfd_ioctl_spm_args args = {0};
|
||||
uint32_t gpu_id;
|
||||
|
||||
ret = hsakmt_validate_nodeid(PreferredNode, &gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, PreferredNode);
|
||||
return ret;
|
||||
}
|
||||
|
||||
ret = HSAKMT_STATUS_SUCCESS;
|
||||
args.op = KFD_IOCTL_SPM_OP_ACQUIRE;
|
||||
args.gpu_id = gpu_id;
|
||||
|
||||
ret = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_RLC_SPM, &args);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtSPMSetDestBuffer(HSAuint32 PreferredNode,
|
||||
HSAuint32 SizeInBytes,
|
||||
HSAuint32 * timeout,
|
||||
HSAuint32 * SizeCopied,
|
||||
void *DestMemoryAddress,
|
||||
bool *isSPMDataLoss)
|
||||
{
|
||||
int ret;
|
||||
struct kfd_ioctl_spm_args args = {0};
|
||||
uint32_t gpu_id = 0;
|
||||
|
||||
ret = hsakmt_validate_nodeid(PreferredNode, &gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
return ret;
|
||||
}
|
||||
|
||||
args.timeout = *timeout;
|
||||
args.dest_buf = (uint64_t)DestMemoryAddress;
|
||||
args.buf_size = SizeInBytes;
|
||||
args.op = KFD_IOCTL_SPM_OP_SET_DEST_BUF;
|
||||
args.gpu_id = gpu_id;
|
||||
|
||||
ret = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_RLC_SPM, &args);
|
||||
|
||||
*SizeCopied = args.bytes_copied;
|
||||
*isSPMDataLoss = args.has_data_loss;
|
||||
*timeout = args.timeout;
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtSPMRelease(HSAuint32 PreferredNode)
|
||||
{
|
||||
int ret = HSAKMT_STATUS_SUCCESS;
|
||||
struct kfd_ioctl_spm_args args = {0};
|
||||
uint32_t gpu_id;
|
||||
|
||||
ret = hsakmt_validate_nodeid(PreferredNode, &gpu_id);
|
||||
if (ret != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_err("[%s] invalid node ID: %d\n", __func__, PreferredNode);
|
||||
return ret;
|
||||
}
|
||||
|
||||
args.op = KFD_IOCTL_SPM_OP_RELEASE;
|
||||
args.gpu_id = gpu_id;
|
||||
|
||||
ret = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_RLC_SPM, &args);
|
||||
|
||||
return ret;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,227 @@
|
||||
/*
|
||||
* Copyright © 2020 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
#include "libhsakmt.h"
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <unistd.h>
|
||||
#include <inttypes.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/time.h>
|
||||
#include <errno.h>
|
||||
|
||||
/* Helper functions for calling KFD SVM ioctl */
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI
|
||||
hsaKmtSVMSetAttr(void *start_addr, HSAuint64 size, unsigned int nattr,
|
||||
HSA_SVM_ATTRIBUTE *attrs)
|
||||
{
|
||||
struct kfd_ioctl_svm_args *args;
|
||||
HSAuint64 s_attr;
|
||||
HSAKMT_STATUS r;
|
||||
HSAuint32 i;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(5);
|
||||
|
||||
pr_debug("%s: address 0x%p size 0x%lx\n", __func__, start_addr, size);
|
||||
|
||||
if (!start_addr || !size)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
if ((uint64_t)start_addr & (PAGE_SIZE - 1))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
if (size & (PAGE_SIZE - 1))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
s_attr = sizeof(*attrs) * nattr;
|
||||
args = alloca(sizeof(*args) + s_attr);
|
||||
|
||||
args->start_addr = (uint64_t)start_addr;
|
||||
args->size = size;
|
||||
args->op = KFD_IOCTL_SVM_OP_SET_ATTR;
|
||||
args->nattr = nattr;
|
||||
memcpy(args->attrs, attrs, s_attr);
|
||||
|
||||
for (i = 0; i < nattr; i++) {
|
||||
if (attrs[i].type != KFD_IOCTL_SVM_ATTR_PREFERRED_LOC &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_PREFETCH_LOC &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_ACCESS &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_ACCESS_IN_PLACE &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_NO_ACCESS)
|
||||
continue;
|
||||
|
||||
if (attrs[i].type == KFD_IOCTL_SVM_ATTR_PREFERRED_LOC &&
|
||||
attrs[i].value == INVALID_NODEID) {
|
||||
args->attrs[i].value = KFD_IOCTL_SVM_LOCATION_UNDEFINED;
|
||||
continue;
|
||||
}
|
||||
|
||||
r = hsakmt_validate_nodeid(attrs[i].value, &args->attrs[i].value);
|
||||
if (r != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_debug("invalid node ID: %d\n", attrs[i].value);
|
||||
return r;
|
||||
} else if (!args->attrs[i].value &&
|
||||
(attrs[i].type == KFD_IOCTL_SVM_ATTR_ACCESS ||
|
||||
attrs[i].type == KFD_IOCTL_SVM_ATTR_ACCESS_IN_PLACE ||
|
||||
attrs[i].type == KFD_IOCTL_SVM_ATTR_NO_ACCESS)) {
|
||||
pr_debug("CPU node invalid for access attribute\n");
|
||||
return HSAKMT_STATUS_INVALID_NODE_UNIT;
|
||||
}
|
||||
}
|
||||
|
||||
/* Driver does one copy_from_user, with extra attrs size */
|
||||
r = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SVM + (s_attr << _IOC_SIZESHIFT), args);
|
||||
if (r) {
|
||||
pr_debug("op set range attrs failed %s\n", strerror(errno));
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI
|
||||
hsaKmtSVMGetAttr(void *start_addr, HSAuint64 size, unsigned int nattr,
|
||||
HSA_SVM_ATTRIBUTE *attrs)
|
||||
{
|
||||
struct kfd_ioctl_svm_args *args;
|
||||
HSAuint64 s_attr;
|
||||
HSAKMT_STATUS r;
|
||||
HSAuint32 i;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(5);
|
||||
|
||||
pr_debug("%s: address 0x%p size 0x%lx\n", __func__, start_addr, size);
|
||||
|
||||
if (!start_addr || !size)
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
if ((uint64_t)start_addr & (PAGE_SIZE - 1))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
if (size & (PAGE_SIZE - 1))
|
||||
return HSAKMT_STATUS_INVALID_PARAMETER;
|
||||
|
||||
s_attr = sizeof(*attrs) * nattr;
|
||||
args = alloca(sizeof(*args) + s_attr);
|
||||
|
||||
args->start_addr = (uint64_t)start_addr;
|
||||
args->size = size;
|
||||
args->op = KFD_IOCTL_SVM_OP_GET_ATTR;
|
||||
args->nattr = nattr;
|
||||
memcpy(args->attrs, attrs, s_attr);
|
||||
|
||||
for (i = 0; i < nattr; i++) {
|
||||
if (attrs[i].type != KFD_IOCTL_SVM_ATTR_ACCESS &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_ACCESS_IN_PLACE &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_NO_ACCESS)
|
||||
continue;
|
||||
|
||||
r = hsakmt_validate_nodeid(attrs[i].value, &args->attrs[i].value);
|
||||
if (r != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_debug("invalid node ID: %d\n", attrs[i].value);
|
||||
return r;
|
||||
} else if (!args->attrs[i].value) {
|
||||
pr_debug("CPU node invalid for access attribute\n");
|
||||
return HSAKMT_STATUS_INVALID_NODE_UNIT;
|
||||
}
|
||||
}
|
||||
|
||||
/* Driver does one copy_from_user, with extra attrs size */
|
||||
r = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SVM + (s_attr << _IOC_SIZESHIFT), args);
|
||||
if (r) {
|
||||
pr_debug("op get range attrs failed %s\n", strerror(errno));
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
memcpy(attrs, args->attrs, s_attr);
|
||||
|
||||
for (i = 0; i < nattr; i++) {
|
||||
if (attrs[i].type != KFD_IOCTL_SVM_ATTR_PREFERRED_LOC &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_PREFETCH_LOC &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_ACCESS &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_ACCESS_IN_PLACE &&
|
||||
attrs[i].type != KFD_IOCTL_SVM_ATTR_NO_ACCESS)
|
||||
continue;
|
||||
|
||||
switch (attrs[i].value) {
|
||||
case KFD_IOCTL_SVM_LOCATION_SYSMEM:
|
||||
attrs[i].value = 0;
|
||||
break;
|
||||
case KFD_IOCTL_SVM_LOCATION_UNDEFINED:
|
||||
attrs[i].value = INVALID_NODEID;
|
||||
break;
|
||||
default:
|
||||
r = hsakmt_gpuid_to_nodeid(attrs[i].value, &attrs[i].value);
|
||||
if (r != HSAKMT_STATUS_SUCCESS) {
|
||||
pr_debug("invalid GPU ID: %d\n",
|
||||
attrs[i].value);
|
||||
return r;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
static HSAKMT_STATUS
|
||||
hsaKmtSetGetXNACKMode(HSAint32 * enable)
|
||||
{
|
||||
struct kfd_ioctl_set_xnack_mode_args args;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
CHECK_KFD_MINOR_VERSION(5);
|
||||
|
||||
args.xnack_enabled = *enable;
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_SET_XNACK_MODE, &args)) {
|
||||
if (errno == EPERM) {
|
||||
pr_debug("set mode not supported %s\n",
|
||||
strerror(errno));
|
||||
return HSAKMT_STATUS_NOT_SUPPORTED;
|
||||
} else if (errno == EBUSY) {
|
||||
pr_debug("hsakmt_ioctl queues not empty %s\n",
|
||||
strerror(errno));
|
||||
}
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
*enable = args.xnack_enabled;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI
|
||||
hsaKmtSetXNACKMode(HSAint32 enable)
|
||||
{
|
||||
return hsaKmtSetGetXNACKMode(&enable);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI
|
||||
hsaKmtGetXNACKMode(HSAint32 * enable)
|
||||
{
|
||||
*enable = -1;
|
||||
return hsaKmtSetGetXNACKMode(enable);
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtGetClockCounters(HSAuint32 NodeId,
|
||||
HsaClockCounters *Counters)
|
||||
{
|
||||
HSAKMT_STATUS result;
|
||||
uint32_t gpu_id;
|
||||
struct kfd_ioctl_get_clock_counters_args args = {0};
|
||||
int err;
|
||||
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
result = hsakmt_validate_nodeid(NodeId, &gpu_id);
|
||||
if (result != HSAKMT_STATUS_SUCCESS)
|
||||
return result;
|
||||
|
||||
args.gpu_id = gpu_id;
|
||||
|
||||
err = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_GET_CLOCK_COUNTERS, &args);
|
||||
if (err < 0) {
|
||||
result = HSAKMT_STATUS_ERROR;
|
||||
} else {
|
||||
/* At this point the result is already HSAKMT_STATUS_SUCCESS */
|
||||
Counters->GPUClockCounter = args.gpu_clock_counter;
|
||||
Counters->CPUClockCounter = args.cpu_clock_counter;
|
||||
Counters->SystemClockCounter = args.system_clock_counter;
|
||||
Counters->SystemClockFrequencyHz = args.system_clock_freq;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,56 @@
|
||||
/*
|
||||
* Copyright © 2014 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person
|
||||
* obtaining a copy of this software and associated documentation
|
||||
* files (the "Software"), to deal in the Software without
|
||||
* restriction, including without limitation the rights to use, copy,
|
||||
* modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
* of the Software, and to permit persons to whom the Software is
|
||||
* furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice (including
|
||||
* the next paragraph) shall be included in all copies or substantial
|
||||
* portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
|
||||
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
|
||||
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
* DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#include "libhsakmt.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include "hsakmt/linux/kfd_ioctl.h"
|
||||
|
||||
HsaVersionInfo hsakmt_kfd_version_info;
|
||||
|
||||
HSAKMT_STATUS HSAKMTAPI hsaKmtGetVersion(HsaVersionInfo *VersionInfo)
|
||||
{
|
||||
CHECK_KFD_OPEN();
|
||||
|
||||
*VersionInfo = hsakmt_kfd_version_info;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS hsakmt_init_kfd_version(void)
|
||||
{
|
||||
struct kfd_ioctl_get_version_args args = {0};
|
||||
|
||||
if (hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_GET_VERSION, &args) == -1)
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
|
||||
hsakmt_kfd_version_info.KernelInterfaceMajorVersion = args.major_version;
|
||||
hsakmt_kfd_version_info.KernelInterfaceMinorVersion = args.minor_version;
|
||||
|
||||
if (args.major_version != 1)
|
||||
return HSAKMT_STATUS_DRIVER_MISMATCH;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
@@ -0,0 +1,264 @@
|
||||
#
|
||||
# Copyright (C) 2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
#
|
||||
# Permission is hereby granted, free of charge, to any person obtaining a
|
||||
# copy of this software and associated documentation files (the "Software"),
|
||||
# to deal in the Software without restriction, including without limitation
|
||||
# the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
# and/or sell copies of the Software, and to permit persons to whom the
|
||||
# Software is furnished to do so, subject to the following conditions:
|
||||
#
|
||||
# The above copyright notice and this permission notice shall be included in
|
||||
# all copies or substantial portions of the Software.
|
||||
#
|
||||
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
# THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
# OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
# ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
# OTHER DEALINGS IN THE SOFTWARE.
|
||||
#
|
||||
#
|
||||
|
||||
# If environment variable DRM_DIR or LIBHSAKMT_PATH is set, the script
|
||||
# will pick up the corresponding libraries from those pathes.
|
||||
|
||||
cmake_minimum_required(VERSION 3.5 FATAL_ERROR)
|
||||
|
||||
project(KFDTest)
|
||||
|
||||
# For DEB/RPM generation
|
||||
set ( CPACK_PACKAGE_NAME "kfdtest" )
|
||||
set ( CPACK_PACKAGE_CONTACT "Advanced Micro Devices Inc." )
|
||||
set ( CPACK_PACKAGE_DESCRIPTION "This package includes kfdtest, the list of excluded tests for each ASIC, and a convenience script to run the test suite" )
|
||||
set ( CPACK_PACKAGE_DESCRIPTION_SUMMARY "Test suite for ROCK/KFD" )
|
||||
|
||||
# Make proper version for appending
|
||||
# Default Value is 99999, setting it first
|
||||
set(ROCM_VERSION_FOR_PACKAGE "99999")
|
||||
if(DEFINED ENV{ROCM_LIBPATCH_VERSION})
|
||||
set(ROCM_VERSION_FOR_PACKAGE $ENV{ROCM_LIBPATCH_VERSION})
|
||||
endif()
|
||||
|
||||
set ( CPACK_PACKAGE_VERSION_MAJOR "1" )
|
||||
set ( CPACK_PACKAGE_VERSION_MINOR "0" )
|
||||
set ( CPACK_PACKAGE_VERSION_PATCH "0" )
|
||||
set ( CPACK_PACKAGE_HOMEPAGE_URL "https://github.com/ROCm/ROCR-Runtime/" )
|
||||
set ( CPACK_DEBIAN_FILE_NAME "DEB-DEFAULT")
|
||||
set ( CPACK_RPM_FILE_NAME "RPM-DEFAULT")
|
||||
|
||||
## Debian package values
|
||||
set ( CPACK_DEBIAN_PACKAGE_RELEASE "local" )
|
||||
if( DEFINED ENV{CPACK_DEBIAN_PACKAGE_RELEASE} )
|
||||
set ( CPACK_DEBIAN_PACKAGE_RELEASE $ENV{CPACK_DEBIAN_PACKAGE_RELEASE} )
|
||||
endif()
|
||||
## RPM package variables
|
||||
set ( CPACK_RPM_PACKAGE_RELEASE "local" )
|
||||
if( DEFINED ENV{CPACK_RPM_PACKAGE_RELEASE} )
|
||||
set ( CPACK_RPM_PACKAGE_RELEASE $ENV{CPACK_RPM_PACKAGE_RELEASE} )
|
||||
endif()
|
||||
|
||||
## Note: rpm --eval %{?dist} will evaluate to NULL in Debian
|
||||
## So Debian distros won't append dist tag to CPACK_RPM_PACKAGE_RELEASE.
|
||||
## Also for debian package name , the dist tag is added from build env
|
||||
execute_process( COMMAND rpm --eval %{?dist}
|
||||
RESULT_VARIABLE PROC_RESULT
|
||||
OUTPUT_VARIABLE EVAL_RESULT
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE )
|
||||
message("RESULT_VARIABLE ${PROC_RESULT} OUTPUT_VARIABLE: ${EVAL_RESULT}")
|
||||
## Add distribution tag to rpm package name
|
||||
if ( PROC_RESULT EQUAL "0" AND NOT EVAL_RESULT STREQUAL "" )
|
||||
string ( APPEND CPACK_RPM_PACKAGE_RELEASE "%{?dist}" )
|
||||
endif()
|
||||
|
||||
set(PACKAGE_VERSION_STR "${CPACK_PACKAGE_VERSION_MAJOR}.${CPACK_PACKAGE_VERSION_MINOR}.${CPACK_PACKAGE_VERSION_PATCH}.${ROCM_VERSION_FOR_PACKAGE}")
|
||||
set(CPACK_PACKAGE_VERSION "${PACKAGE_VERSION_STR}")
|
||||
|
||||
## Define default variable and variables for the optional build target hsakmt-dev
|
||||
set ( SOURCE_DIR ${CMAKE_CURRENT_SOURCE_DIR} CACHE STRING "Location of hsakmt source code." )
|
||||
set ( CMAKE_INSTALL_PREFIX "/opt/rocm" CACHE STRING "Default installation directory." )
|
||||
set ( CPACK_PACKAGING_INSTALL_PREFIX "${CMAKE_INSTALL_PREFIX}" CACHE STRING "Default packaging prefix." )
|
||||
set ( CPACK_GENERATOR "DEB;RPM" CACHE STRING "Default packaging generators." )
|
||||
|
||||
# Debian package specific variables
|
||||
set ( CPACK_DEBIAN_PACKAGE_HOMEPAGE "https://github.com/ROCm/ROCR-Runtime/" )
|
||||
set ( CPACK_DEBIAN_PACKAGE_DEPENDS "rocm-core" )
|
||||
|
||||
# RPM package specific variables
|
||||
set (CPACK_RPM_PACKAGE_REQUIRES "rocm-core")
|
||||
|
||||
#set ( CMAKE_VERBOSE_MAKEFILE on )
|
||||
|
||||
find_package(PkgConfig)
|
||||
|
||||
list (PREPEND CMAKE_PREFIX_PATH "${DRM_DIR}")
|
||||
# The module name passed to pkg_check_modules() is determined by the
|
||||
# name of file *.pc
|
||||
pkg_check_modules(DRM REQUIRED libdrm)
|
||||
pkg_check_modules(DRM_AMDGPU REQUIRED libdrm_amdgpu)
|
||||
include_directories(${DRM_AMDGPU_INCLUDE_DIRS})
|
||||
|
||||
if( DEFINED ENV{LIBHSAKMT_PATH} )
|
||||
set ( LIBHSAKMT_PATH $ENV{LIBHSAKMT_PATH} )
|
||||
message ( "LIBHSAKMT_PATH environment variable is set" )
|
||||
else()
|
||||
if ( ${ROCM_INSTALL_PATH} )
|
||||
set ( ENV{PKG_CONFIG_PATH} ${ROCM_INSTALL_PATH}/share/pkgconfig )
|
||||
else()
|
||||
set ( ENV{PKG_CONFIG_PATH} /opt/rocm/share/pkgconfig )
|
||||
endif()
|
||||
|
||||
pkg_check_modules(HSAKMT libhsakmt)
|
||||
|
||||
if( NOT HSAKMT_FOUND )
|
||||
set ( LIBHSAKMT_PATH $ENV{OUT_DIR} )
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if( DEFINED LIBHSAKMT_PATH )
|
||||
set ( HSAKMT_LIBRARY_DIRS ${LIBHSAKMT_PATH} )
|
||||
set ( HSAKMT_LIBRARIES hsakmt )
|
||||
endif()
|
||||
|
||||
message ( "Find libhsakmt at ${HSAKMT_LIBRARY_DIRS}" )
|
||||
|
||||
if ( POLICY CMP0074 )
|
||||
cmake_policy( SET CMP0074 NEW )
|
||||
endif()
|
||||
|
||||
find_path( LIGHTNING_CMAKE_DIR NAMES LLVMConfig.cmake
|
||||
PATHS $ENV{OUT_DIR}/llvm/lib/cmake/llvm NO_CACHE NO_DEFAULT_PATH)
|
||||
|
||||
if ( DEFINED LIGHTNING_CMAKE_DIR AND EXISTS ${LIGHTNING_CMAKE_DIR} )
|
||||
set ( LLVM_DIR ${LIGHTNING_CMAKE_DIR} )
|
||||
else()
|
||||
message( STATUS "Couldn't find Lightning build in compute directory. "
|
||||
"Searching LLVM_DIR then defaulting to system LLVM install if still not found..." )
|
||||
endif()
|
||||
|
||||
find_package( LLVM REQUIRED CONFIG )
|
||||
|
||||
if( ${LLVM_PACKAGE_VERSION} VERSION_LESS "7.0" )
|
||||
message( FATAL_ERROR "Requires LLVM 7.0 or greater "
|
||||
"(found ${LLVM_PACKAGE_VERSION})" )
|
||||
elseif( ${LLVM_PACKAGE_VERSION} VERSION_LESS "14.0" )
|
||||
message( WARNING "Not using latest LLVM version. "
|
||||
"Some ASIC targets may not work!" )
|
||||
endif()
|
||||
|
||||
message( STATUS "Found LLVM ${LLVM_PACKAGE_VERSION}" )
|
||||
message( STATUS "Using LLVMConfig.cmake in: ${LLVM_DIR}" )
|
||||
|
||||
include_directories(${LLVM_INCLUDE_DIRS})
|
||||
separate_arguments(LLVM_DEFINITIONS_LIST NATIVE_COMMAND ${LLVM_DEFINITIONS})
|
||||
add_definitions(${LLVM_DEFINITIONS_LIST})
|
||||
|
||||
if (LLVM_LINK_LLVM_DYLIB)
|
||||
set(llvm_libs LLVM)
|
||||
else()
|
||||
llvm_map_components_to_libnames(llvm_libs AMDGPUAsmParser Core Support)
|
||||
endif()
|
||||
|
||||
include_directories(${PROJECT_SOURCE_DIR}/gtest-1.6.0)
|
||||
include_directories(${PROJECT_SOURCE_DIR}/include)
|
||||
include_directories(${PROJECT_SOURCE_DIR}/../../include)
|
||||
include_directories(${PROJECT_SOURCE_DIR}/../../libhsakmt/include)
|
||||
|
||||
include_directories(${DRM_INCLUDE_DIRS})
|
||||
|
||||
set (SRC_FILES gtest-1.6.0/gtest-all.cpp
|
||||
|
||||
src/AqlQueue.cpp
|
||||
src/BasePacket.cpp
|
||||
src/BaseDebug.cpp
|
||||
src/BaseQueue.cpp
|
||||
src/Dispatch.cpp
|
||||
src/GoogleTestExtension.cpp
|
||||
src/IndirectBuffer.cpp
|
||||
src/Assemble.cpp
|
||||
src/ShaderStore.cpp
|
||||
src/LinuxOSWrapper.cpp
|
||||
src/PM4Packet.cpp
|
||||
src/PM4Queue.cpp
|
||||
src/RDMAUtil.cpp
|
||||
src/SDMAPacket.cpp
|
||||
src/SDMAQueue.cpp
|
||||
src/KFDBaseComponentTest.cpp
|
||||
src/KFDMultiProcessTest.cpp
|
||||
src/KFDTestMain.cpp
|
||||
src/KFDTestUtil.cpp
|
||||
src/KFDTestUtilQueue.cpp
|
||||
|
||||
src/KFDOpenCloseKFDTest.cpp
|
||||
src/KFDTopologyTest.cpp
|
||||
src/KFDMemoryTest.cpp
|
||||
src/KFDLocalMemoryTest.cpp
|
||||
src/KFDEventTest.cpp
|
||||
src/KFDQMTest.cpp
|
||||
src/KFDCWSRTest.cpp
|
||||
src/KFDExceptionTest.cpp
|
||||
src/KFDGraphicsInterop.cpp
|
||||
src/KFDPerfCounters.cpp
|
||||
src/KFDDBGTest.cpp
|
||||
src/KFDGWSTest.cpp
|
||||
src/KFDIPCTest.cpp
|
||||
src/KFDASMTest.cpp
|
||||
|
||||
src/KFDEvictTest.cpp
|
||||
src/KFDHWSTest.cpp
|
||||
src/KFDPerformanceTest.cpp
|
||||
src/KFDPMTest.cpp
|
||||
src/KFDSVMRangeTest.cpp
|
||||
src/KFDSVMEvictTest.cpp
|
||||
src/KFDRASTest.cpp
|
||||
src/KFDPCSamplingTest.cpp
|
||||
src/KFDNegativeTest.cpp
|
||||
src/RDMATest.cpp)
|
||||
|
||||
message( STATUS "PROJECT_SOURCE_DIR:" ${PROJECT_SOURCE_DIR} )
|
||||
#message( STATUS "SRC_FILES: ")
|
||||
#foreach(file ${SRC_FILES})
|
||||
# message(STATUS "${file}")
|
||||
#endforeach()
|
||||
|
||||
#add_definitions(-Wall -std=c++11)
|
||||
|
||||
if ( "${CMAKE_C_COMPILER_VERSION}" STRGREATER "4.8.0")
|
||||
## Add --enable-new-dtags to generate DT_RUNPATH
|
||||
set ( CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -std=gnu++17 -Wl,--enable-new-dtags" )
|
||||
endif()
|
||||
if ( "${CMAKE_BUILD_TYPE}" STREQUAL Release )
|
||||
set ( CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -O2" )
|
||||
else ()
|
||||
set ( CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -g" )
|
||||
endif ()
|
||||
|
||||
## Address Sanitize Flag
|
||||
if ( ${ADDRESS_SANITIZER} )
|
||||
set ( CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -fsanitize=address" )
|
||||
set ( CMAKE_EXE_LINKER_FLAGS -fsanitize=address )
|
||||
endif ()
|
||||
|
||||
# link_directories() has to be put before add_executable()
|
||||
# The modules found by pkg_check_modules() in the default pkg config
|
||||
# path do not need to use link_directories() here.
|
||||
link_directories(${HSAKMT_LIBRARY_DIRS})
|
||||
|
||||
add_executable(kfdtest ${SRC_FILES})
|
||||
|
||||
target_link_libraries(kfdtest ${HSAKMT_LIBRARIES} ${DRM_LDFLAGS} ${DRM_AMDGPU_LDFLAGS} ${llvm_libs} pthread m stdc++ rt numa)
|
||||
|
||||
configure_file ( scripts/kfdtest.exclude kfdtest.exclude COPYONLY )
|
||||
configure_file ( scripts/run_kfdtest.sh run_kfdtest.sh COPYONLY )
|
||||
|
||||
install( PROGRAMS ${CMAKE_CURRENT_BINARY_DIR}/kfdtest ${CMAKE_CURRENT_BINARY_DIR}/run_kfdtest.sh
|
||||
DESTINATION bin )
|
||||
install( FILES ${CMAKE_CURRENT_BINARY_DIR}/kfdtest.exclude
|
||||
DESTINATION share/kfdtest )
|
||||
# Remove dependency on rocm-core if -DROCM_DEP_ROCMCORE=ON not given to cmake
|
||||
if(NOT ROCM_DEP_ROCMCORE)
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_RPM_PACKAGE_REQUIRES ${CPACK_RPM_PACKAGE_REQUIRES})
|
||||
string(REGEX REPLACE ",? ?rocm-core" "" CPACK_DEBIAN_PACKAGE_DEPENDS ${CPACK_DEBIAN_PACKAGE_DEPENDS})
|
||||
endif()
|
||||
include ( CPack )
|
||||
@@ -0,0 +1,22 @@
|
||||
KFDTest - KFD unit tests LICENSE
|
||||
Copyright (C) 2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
|
||||
MIT LICENSE:
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
|
||||
CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
|
||||
TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||
SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
@@ -0,0 +1,19 @@
|
||||
1. Note on building kfdtest
|
||||
|
||||
To build this kfdtest application, the following libraries should be already
|
||||
installed on the building machine:
|
||||
libdrm libdrm_amdgpu libhsakmt
|
||||
|
||||
If libhsakmt is not installed, but the headers and libraries are present
|
||||
locally, you can specify its directory by
|
||||
export LIBHSAKMT_PATH=/path/to/libhsakmt.a
|
||||
With that, CMake/make will look for the lib at LIBHSAKMT_PATH/libhsakmt.a
|
||||
Note that this assumes that you will be building kfdtest from the same thunk found in ../..
|
||||
|
||||
2. How to run kfdtest
|
||||
|
||||
Just run "./run_kfdtest.sh" under the building output folder. You may need
|
||||
to specify library path through:
|
||||
export LD_LIBRARY_PATH=/path/to/libhsakmt.a
|
||||
|
||||
Note: you can use "run_kfdtest.sh -h" to see more options.
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,71 @@
|
||||
/*
|
||||
* Copyright 2015-2024 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
#ifndef AMDP2PTEST_H_
|
||||
#define AMDP2PTEST_H_
|
||||
|
||||
#include <linux/ioctl.h>
|
||||
|
||||
#define AMDP2PTEST_IOCTL_MAGIC 'A'
|
||||
|
||||
|
||||
#define AMDP2PTEST_DEVICE_NAME "amdp2ptest"
|
||||
#define AMDP2PTEST_DEVICE_PATH "/dev/amdp2ptest"
|
||||
|
||||
struct AMDRDMA_IOCTL_GET_PAGE_SIZE_PARAM {
|
||||
/* Input parameters */
|
||||
uint64_t addr;
|
||||
uint64_t length;
|
||||
|
||||
/* Output parameters */
|
||||
uint64_t page_size;
|
||||
};
|
||||
|
||||
struct AMDRDMA_IOCTL_GET_PAGES_PARAM {
|
||||
/* Input parameters */
|
||||
uint64_t addr;
|
||||
uint64_t length;
|
||||
uint64_t is_local; /* 1 if this is the pointer to local
|
||||
allocation */
|
||||
|
||||
/* Output parameters */
|
||||
uint64_t cpu_ptr;
|
||||
};
|
||||
|
||||
|
||||
struct AMDRDMA_IOCTL_PUT_PAGES_PARAM {
|
||||
/* Input parameters */
|
||||
uint64_t addr;
|
||||
uint64_t length;
|
||||
};
|
||||
|
||||
|
||||
#define AMD2P2PTEST_IOCTL_GET_PAGE_SIZE \
|
||||
_IOWR(AMDP2PTEST_IOCTL_MAGIC, 1, struct AMDRDMA_IOCTL_GET_PAGE_SIZE_PARAM *)
|
||||
|
||||
#define AMD2P2PTEST_IOCTL_GET_PAGES \
|
||||
_IOWR(AMDP2PTEST_IOCTL_MAGIC, 2, struct AMDRDMA_IOCTL_GET_PAGES_PARAM *)
|
||||
|
||||
#define AMD2P2PTEST_IOCTL_PUT_PAGES \
|
||||
_IOW(AMDP2PTEST_IOCTL_MAGIC, 3, struct AMDRDMA_IOCTL_PUT_PAGES_PARAM *)
|
||||
|
||||
|
||||
#endif /* AMDP2PTEST_H */
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,107 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
|
||||
#ifndef KFD_PM4_OPCODES_H
|
||||
#define KFD_PM4_OPCODES_H
|
||||
|
||||
enum it_opcode_type {
|
||||
IT_NOP = 0x10,
|
||||
IT_SET_BASE = 0x11,
|
||||
IT_CLEAR_STATE = 0x12,
|
||||
IT_INDEX_BUFFER_SIZE = 0x13,
|
||||
IT_DISPATCH_DIRECT = 0x15,
|
||||
IT_DISPATCH_INDIRECT = 0x16,
|
||||
IT_ATOMIC_GDS = 0x1D,
|
||||
IT_OCCLUSION_QUERY = 0x1F,
|
||||
IT_SET_PREDICATION = 0x20,
|
||||
IT_REG_RMW = 0x21,
|
||||
IT_COND_EXEC = 0x22,
|
||||
IT_PRED_EXEC = 0x23,
|
||||
IT_DRAW_INDIRECT = 0x24,
|
||||
IT_DRAW_INDEX_INDIRECT = 0x25,
|
||||
IT_INDEX_BASE = 0x26,
|
||||
IT_DRAW_INDEX_2 = 0x27,
|
||||
IT_CONTEXT_CONTROL = 0x28,
|
||||
IT_INDEX_TYPE = 0x2A,
|
||||
IT_DRAW_INDIRECT_MULTI = 0x2C,
|
||||
IT_DRAW_INDEX_AUTO = 0x2D,
|
||||
IT_NUM_INSTANCES = 0x2F,
|
||||
IT_DRAW_INDEX_MULTI_AUTO = 0x30,
|
||||
IT_INDIRECT_BUFFER_CNST = 0x33,
|
||||
IT_STRMOUT_BUFFER_UPDATE = 0x34,
|
||||
IT_DRAW_INDEX_OFFSET_2 = 0x35,
|
||||
IT_DRAW_PREAMBLE = 0x36,
|
||||
IT_WRITE_DATA = 0x37,
|
||||
IT_DRAW_INDEX_INDIRECT_MULTI = 0x38,
|
||||
IT_MEM_SEMAPHORE = 0x39,
|
||||
IT_COPY_DW = 0x3B,
|
||||
IT_WAIT_REG_MEM = 0x3C,
|
||||
IT_INDIRECT_BUFFER = 0x3F,
|
||||
IT_COPY_DATA = 0x40,
|
||||
IT_PFP_SYNC_ME = 0x42,
|
||||
IT_SURFACE_SYNC = 0x43,
|
||||
IT_COND_WRITE = 0x45,
|
||||
IT_EVENT_WRITE = 0x46,
|
||||
IT_EVENT_WRITE_EOP = 0x47,
|
||||
IT_EVENT_WRITE_EOS = 0x48,
|
||||
IT_RELEASE_MEM = 0x49,
|
||||
IT_PREAMBLE_CNTL = 0x4A,
|
||||
IT_DMA_DATA = 0x50,
|
||||
IT_ACQUIRE_MEM = 0x58,
|
||||
IT_REWIND = 0x59,
|
||||
IT_LOAD_UCONFIG_REG = 0x5E,
|
||||
IT_LOAD_SH_REG = 0x5F,
|
||||
IT_LOAD_CONFIG_REG = 0x60,
|
||||
IT_LOAD_CONTEXT_REG = 0x61,
|
||||
IT_SET_CONFIG_REG = 0x68,
|
||||
IT_SET_CONTEXT_REG = 0x69,
|
||||
IT_SET_CONTEXT_REG_INDIRECT = 0x73,
|
||||
IT_SET_SH_REG = 0x76,
|
||||
IT_SET_SH_REG_OFFSET = 0x77,
|
||||
IT_SET_QUEUE_REG = 0x78,
|
||||
IT_SET_UCONFIG_REG = 0x79,
|
||||
IT_SCRATCH_RAM_WRITE = 0x7D,
|
||||
IT_SCRATCH_RAM_READ = 0x7E,
|
||||
IT_LOAD_CONST_RAM = 0x80,
|
||||
IT_WRITE_CONST_RAM = 0x81,
|
||||
IT_DUMP_CONST_RAM = 0x83,
|
||||
IT_INCREMENT_CE_COUNTER = 0x84,
|
||||
IT_INCREMENT_DE_COUNTER = 0x85,
|
||||
IT_WAIT_ON_CE_COUNTER = 0x86,
|
||||
IT_WAIT_ON_DE_COUNTER_DIFF = 0x88,
|
||||
IT_SWITCH_BUFFER = 0x8B,
|
||||
IT_SET_RESOURCES = 0xA0,
|
||||
IT_MAP_PROCESS = 0xA1,
|
||||
IT_MAP_QUEUES = 0xA2,
|
||||
IT_UNMAP_QUEUES = 0xA3,
|
||||
IT_QUERY_STATUS = 0xA4,
|
||||
IT_RUN_LIST = 0xA5,
|
||||
};
|
||||
|
||||
#define PM4_TYPE_0 0
|
||||
#define PM4_TYPE_2 2
|
||||
#define PM4_TYPE_3 3
|
||||
|
||||
#endif /* KFD_PM4_OPCODES_H */
|
||||
|
||||
@@ -0,0 +1,160 @@
|
||||
/*
|
||||
* Copyright (C) 2016-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __PM4_PKT_STRUCT_AI_H__
|
||||
#define __PM4_PKT_STRUCT_AI_H__
|
||||
|
||||
#ifndef PM4_MEC_RELEASE_MEM_AI_DEFINED
|
||||
#define PM4_MEC_RELEASE_MEM_AI_DEFINED
|
||||
|
||||
enum AI_MEC_RELEASE_MEM_event_index_enum {
|
||||
event_index__mec_release_mem__end_of_pipe = 5,
|
||||
event_index__mec_release_mem__shader_done = 6 };
|
||||
|
||||
enum AI_MEC_RELEASE_MEM_cache_policy_enum {
|
||||
cache_policy__mec_release_mem__lru = 0,
|
||||
cache_policy__mec_release_mem__stream = 1 };
|
||||
|
||||
enum AI_MEC_RELEASE_MEM_pq_exe_status_enum {
|
||||
pq_exe_status__mec_release_mem__default = 0,
|
||||
pq_exe_status__mec_release_mem__phase_update = 1 };
|
||||
|
||||
enum AI_MEC_RELEASE_MEM_dst_sel_enum {
|
||||
dst_sel__mec_release_mem__memory_controller = 0,
|
||||
dst_sel__mec_release_mem__tc_l2 = 1,
|
||||
dst_sel__mec_release_mem__queue_write_pointer_register = 2,
|
||||
dst_sel__mec_release_mem__queue_write_pointer_poll_mask_bit = 3 };
|
||||
|
||||
enum AI_MEC_RELEASE_MEM_int_sel_enum {
|
||||
int_sel__mec_release_mem__none = 0,
|
||||
int_sel__mec_release_mem__send_interrupt_only = 1,
|
||||
int_sel__mec_release_mem__send_interrupt_after_write_confirm = 2,
|
||||
int_sel__mec_release_mem__send_data_after_write_confirm = 3,
|
||||
int_sel__mec_release_mem__unconditionally_send_int_ctxid = 4,
|
||||
int_sel__mec_release_mem__conditionally_send_int_ctxid_based_on_32_bit_compare = 5,
|
||||
int_sel__mec_release_mem__conditionally_send_int_ctxid_based_on_64_bit_compare = 6 };
|
||||
|
||||
enum AI_MEC_RELEASE_MEM_data_sel_enum {
|
||||
data_sel__mec_release_mem__none = 0,
|
||||
data_sel__mec_release_mem__send_32_bit_low = 1,
|
||||
data_sel__mec_release_mem__send_64_bit_data = 2,
|
||||
data_sel__mec_release_mem__send_gpu_clock_counter = 3,
|
||||
data_sel__mec_release_mem__send_cp_perfcounter_hi_lo = 4,
|
||||
data_sel__mec_release_mem__store_gds_data_to_memory = 5 };
|
||||
|
||||
|
||||
typedef struct PM4_MEC_RELEASE_MEM_AI {
|
||||
union {
|
||||
PM4_TYPE_3_HEADER header;
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int event_type:6;
|
||||
unsigned int reserved1:2;
|
||||
AI_MEC_RELEASE_MEM_event_index_enum event_index:4;
|
||||
unsigned int tcl1_vol_action_ena:1;
|
||||
unsigned int tc_vol_action_ena:1;
|
||||
unsigned int reserved2:1;
|
||||
unsigned int tc_wb_action_ena:1;
|
||||
unsigned int tcl1_action_ena:1;
|
||||
unsigned int tc_action_ena:1;
|
||||
unsigned int reserved3:1;
|
||||
unsigned int tc_nc_action_ena:1;
|
||||
unsigned int tc_wc_action_ena:1;
|
||||
unsigned int tc_md_action_ena:1;
|
||||
unsigned int reserved4:3;
|
||||
AI_MEC_RELEASE_MEM_cache_policy_enum cache_policy:2;
|
||||
unsigned int reserved5:2;
|
||||
AI_MEC_RELEASE_MEM_pq_exe_status_enum pq_exe_status:1;
|
||||
unsigned int reserved6:2;
|
||||
} bitfields2;
|
||||
unsigned int ordinal2;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int reserved7:16;
|
||||
AI_MEC_RELEASE_MEM_dst_sel_enum dst_sel:2;
|
||||
unsigned int reserved8:6;
|
||||
AI_MEC_RELEASE_MEM_int_sel_enum int_sel:3;
|
||||
unsigned int reserved9:2;
|
||||
AI_MEC_RELEASE_MEM_data_sel_enum data_sel:3;
|
||||
} bitfields3;
|
||||
unsigned int ordinal3;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int reserved10:2;
|
||||
unsigned int address_lo_32b:30;
|
||||
} bitfields4a;
|
||||
struct {
|
||||
unsigned int reserved11:3;
|
||||
unsigned int address_lo_64b:29;
|
||||
} bitfields4b;
|
||||
unsigned int reserved12;
|
||||
|
||||
unsigned int ordinal4;
|
||||
};
|
||||
|
||||
union {
|
||||
unsigned int address_hi;
|
||||
|
||||
unsigned int reserved13;
|
||||
|
||||
unsigned int ordinal5;
|
||||
};
|
||||
|
||||
union {
|
||||
unsigned int data_lo;
|
||||
|
||||
unsigned int cmp_data_lo;
|
||||
|
||||
struct {
|
||||
unsigned int dw_offset:16;
|
||||
unsigned int num_dwords:16;
|
||||
} bitfields6c;
|
||||
unsigned int reserved14;
|
||||
|
||||
unsigned int ordinal6;
|
||||
};
|
||||
|
||||
union {
|
||||
unsigned int data_hi;
|
||||
|
||||
unsigned int cmp_data_hi;
|
||||
|
||||
unsigned int reserved15;
|
||||
|
||||
unsigned int reserved16;
|
||||
|
||||
unsigned int ordinal7;
|
||||
};
|
||||
|
||||
unsigned int int_ctxid;
|
||||
} PM4MEC_RELEASE_MEM_AI, *PPM4MEC_RELEASE_MEM_AI;
|
||||
|
||||
#endif // PM4_MEC_RELEASE_MEM_AI_DEFINED
|
||||
#endif // __PM4_PKT_STRUCT_AI_H__
|
||||
@@ -0,0 +1,129 @@
|
||||
/*
|
||||
* Copyright (C) 2012-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __PM4_PKT_STRUCT_CI_H__
|
||||
#define __PM4_PKT_STRUCT_CI_H__
|
||||
|
||||
|
||||
enum WRITE_DATA_CI_atc_enum { atc_write_data_NOT_USE_ATC_0 = 0, atc_write_data_USE_ATC_1 = 1 };
|
||||
enum WRITE_DATA_CI_engine_sel { engine_sel_write_data_ci_MICRO_ENGINE_0 = 0, engine_sel_write_data_ci_PREFETCH_PARSER_1 = 1, engine_sel_write_data_ci_CONST_ENG_2 = 2 };
|
||||
|
||||
typedef struct _PM4WRITE_DATA_CI {
|
||||
union {
|
||||
PM4_TYPE_3_HEADER header;
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int reserved1:8;
|
||||
MEC_WRITE_DATA_dst_sel_enum dst_sel:4;
|
||||
unsigned int reserved2:4;
|
||||
MEC_WRITE_DATA_addr_incr_enum addr_incr:1;
|
||||
unsigned int reserved3:3;
|
||||
MEC_WRITE_DATA_wr_confirm_enum wr_confirm:1;
|
||||
unsigned int reserved4:3;
|
||||
WRITE_DATA_CI_atc_enum atc:1;
|
||||
MEC_WRITE_DATA_cache_policy_enum cache_policy:2;
|
||||
unsigned int volatile_setting:1;
|
||||
unsigned int reserved5:2;
|
||||
WRITE_DATA_CI_engine_sel engine_sel:2;
|
||||
} bitfields2;
|
||||
unsigned int ordinal2;
|
||||
};
|
||||
|
||||
unsigned int dst_addr_lo;
|
||||
|
||||
unsigned int dst_address_hi;
|
||||
|
||||
unsigned int data[1]; // 1..N of these fields
|
||||
} PM4WRITE_DATA_CI, *PPM4WRITE_DATA_CI;
|
||||
|
||||
|
||||
enum MEC_RELEASE_MEM_CI_atc_enum { atc_mec_release_mem_ci_NOT_USE_ATC_0 = 0, atc_mec_release_mem_ci_USE_ATC_1 = 1 };
|
||||
|
||||
typedef struct _PM4_RELEASE_MEM_CI {
|
||||
union {
|
||||
PM4_TYPE_3_HEADER header;
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int event_type:6;
|
||||
unsigned int reserved1:2;
|
||||
MEC_RELEASE_MEM_event_index_enum event_index:4;
|
||||
unsigned int l1_vol:1;
|
||||
unsigned int l2_vol:1;
|
||||
unsigned int reserved:1;
|
||||
unsigned int l2_wb:1;
|
||||
unsigned int l1_inv:1;
|
||||
unsigned int l2_inv:1;
|
||||
unsigned int reserved2:6;
|
||||
MEC_RELEASE_MEM_CI_atc_enum atc:1;
|
||||
MEC_RELEASE_MEM_cache_policy_enum cache_policy:2;
|
||||
unsigned int volatile_setting:1;
|
||||
unsigned int reserved3:4;
|
||||
} bitfields2;
|
||||
unsigned int ordinal2;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int reserved4:16;
|
||||
MEC_RELEASE_MEM_dst_sel_enum dst_sel:2;
|
||||
unsigned int reserved5:6;
|
||||
MEC_RELEASE_MEM_int_sel_enum int_sel:3;
|
||||
unsigned int reserved6:2;
|
||||
MEC_RELEASE_MEM_data_sel_enum data_sel:3;
|
||||
} bitfields3;
|
||||
unsigned int ordinal3;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int reserved7:2;
|
||||
unsigned int address_lo_dword_aligned:30;
|
||||
} bitfields4a;
|
||||
struct {
|
||||
unsigned int reserved8:3;
|
||||
unsigned int address_lo_qword_aligned:29;
|
||||
} bitfields4b;
|
||||
unsigned int ordinal4;
|
||||
};
|
||||
|
||||
unsigned int addr_hi;
|
||||
|
||||
union {
|
||||
unsigned int data_lo;
|
||||
struct {
|
||||
unsigned int offset:16;
|
||||
unsigned int num_dwords:16;
|
||||
} bitfields5b;
|
||||
unsigned int ordinal6;
|
||||
};
|
||||
|
||||
unsigned int data_hi;
|
||||
} PM4_RELEASE_MEM_CI, *PPM4_RELEASE_MEM_CI;
|
||||
|
||||
#endif // __PM4_PKT_STRUCT_CI_H__
|
||||
@@ -0,0 +1,366 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __PM4_PKT_STRUCT_COMMON_H__
|
||||
#define __PM4_PKT_STRUCT_COMMON_H__
|
||||
|
||||
#ifndef PM4_HEADER_DEFINED
|
||||
#define PM4_HEADER_DEFINED
|
||||
typedef union PM4_TYPE_3_HEADER
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int predicate : 1; ///< predicated version of packet when set
|
||||
unsigned int shaderType: 1; ///< 0: Graphics, 1: Compute Shader
|
||||
unsigned int reserved1 : 6; ///< reserved
|
||||
unsigned int opcode : 8; ///< IT opcode
|
||||
unsigned int count : 14;///< number of DWORDs - 1 in the information body.
|
||||
unsigned int type : 2; ///< packet identifier. It should be 3 for type 3 packets
|
||||
};
|
||||
unsigned int u32All;
|
||||
} PM4_TYPE_3_HEADER;
|
||||
#endif // PM4_HEADER_DEFINED
|
||||
|
||||
//--------------------DISPATCH_DIRECT--------------------
|
||||
|
||||
|
||||
typedef struct _PM4_DISPATCH_DIRECT
|
||||
{
|
||||
union
|
||||
{
|
||||
PM4_TYPE_3_HEADER header; ///header
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
unsigned int dim_x;
|
||||
|
||||
|
||||
unsigned int dim_y;
|
||||
|
||||
|
||||
unsigned int dim_z;
|
||||
|
||||
|
||||
unsigned int dispatch_initiator;
|
||||
|
||||
|
||||
} PM4DISPATCH_DIRECT, *PPM4DISPATCH_DIRECT;
|
||||
|
||||
//--------------------INDIRECT_BUFFER--------------------
|
||||
|
||||
enum INDIRECT_BUFFER_cache_policy_enum { cache_policy_indirect_buffer_LRU_0 = 0, cache_policy_indirect_buffer_STREAM_1 = 1, cache_policy_indirect_buffer_BYPASS_2 = 2 };
|
||||
|
||||
|
||||
//--------------------EVENT_WRITE--------------------
|
||||
|
||||
enum EVENT_WRITE_event_index_enum { event_index_event_write_OTHER_0 = 0, event_index_event_write_ZPASS_DONE_1 = 1, event_index_event_write_SAMPLE_PIPELINESTAT_2 = 2, event_index_event_write_SAMPLE_STREAMOUTSTAT_3 = 3, event_index_event_write_CS_VS_PS_PARTIAL_FLUSH_4 = 4, event_index_event_write_RESERVED_EOP_5 = 5, event_index_event_write_RESERVED_EOS_6 = 6, event_index_event_write_CACHE_FLUSH_7 = 7 };
|
||||
|
||||
typedef struct _PM4_EVENT_WRITE
|
||||
{
|
||||
union
|
||||
{
|
||||
PM4_TYPE_3_HEADER header; ///header
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int event_type:6;
|
||||
unsigned int reserved1:2;
|
||||
EVENT_WRITE_event_index_enum event_index:4;
|
||||
unsigned int reserved2:20;
|
||||
} bitfields2;
|
||||
unsigned int ordinal2;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int reserved3:3;
|
||||
unsigned int address_lo:29;
|
||||
} bitfields3;
|
||||
unsigned int ordinal3;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int address_hi:16;
|
||||
unsigned int reserved4:16;
|
||||
} bitfields4;
|
||||
unsigned int ordinal4;
|
||||
};
|
||||
|
||||
} PM4EVENT_WRITE, *PPM4EVENT_WRITE;
|
||||
|
||||
|
||||
//--------------------SET_SH_REG--------------------
|
||||
|
||||
|
||||
typedef struct _PM4_SET_SH_REG
|
||||
{
|
||||
union
|
||||
{
|
||||
PM4_TYPE_3_HEADER header; ///header
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int reg_offset:16;
|
||||
unsigned int reserved1:16;
|
||||
} bitfields2;
|
||||
unsigned int ordinal2;
|
||||
};
|
||||
|
||||
unsigned int reg_data[1]; //1..N of these fields
|
||||
|
||||
|
||||
} PM4SET_SH_REG, *PPM4SET_SH_REG;
|
||||
|
||||
|
||||
//--------------------ACQUIRE_MEM--------------------
|
||||
|
||||
enum ACQUIRE_MEM_engine_enum { engine_acquire_mem_PFP_0 = 0, engine_acquire_mem_ME_1 = 1 };
|
||||
|
||||
|
||||
typedef struct _PM4_ACQUIRE_MEM
|
||||
{
|
||||
union
|
||||
{
|
||||
PM4_TYPE_3_HEADER header; ///header
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int coher_cntl:31;
|
||||
ACQUIRE_MEM_engine_enum engine:1;
|
||||
} bitfields2;
|
||||
unsigned int ordinal2;
|
||||
};
|
||||
|
||||
unsigned int coher_size;
|
||||
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int coher_size_hi:8;
|
||||
unsigned int reserved1:24;
|
||||
} bitfields3;
|
||||
unsigned int ordinal4;
|
||||
};
|
||||
|
||||
unsigned int coher_base_lo;
|
||||
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int coher_base_hi:25;
|
||||
unsigned int reserved2:7;
|
||||
} bitfields4;
|
||||
unsigned int ordinal6;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int poll_interval:16;
|
||||
unsigned int reserved3:16;
|
||||
} bitfields5;
|
||||
unsigned int ordinal7;
|
||||
};
|
||||
|
||||
} PM4ACQUIRE_MEM, *PPM4ACQUIRE_MEM;
|
||||
|
||||
|
||||
//--------------------MEC_INDIRECT_BUFFER--------------------
|
||||
|
||||
typedef struct _PM4_MEC_INDIRECT_BUFFER
|
||||
{
|
||||
union
|
||||
{
|
||||
PM4_TYPE_3_HEADER header; ///header
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int swap_function:2;
|
||||
unsigned int ib_base_lo:30;
|
||||
} bitfields2;
|
||||
unsigned int ordinal2;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int ib_base_hi:16;
|
||||
unsigned int reserved1:16;
|
||||
} bitfields3;
|
||||
unsigned int ordinal3;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int ib_size:20;
|
||||
unsigned int chain:1;
|
||||
unsigned int offload_polling:1;
|
||||
unsigned int volatile_setting:1;
|
||||
unsigned int valid:1;
|
||||
unsigned int vmid:4;
|
||||
INDIRECT_BUFFER_cache_policy_enum cache_policy:2;
|
||||
unsigned int reserved4:2;
|
||||
} bitfields4;
|
||||
unsigned int ordinal4;
|
||||
};
|
||||
|
||||
} PM4MEC_INDIRECT_BUFFER, *PPM4MEC_INDIRECT_BUFFER;
|
||||
|
||||
//--------------------MEC_WAIT_REG_MEM--------------------
|
||||
|
||||
enum MEC_WAIT_REG_MEM_function_enum {
|
||||
function__mec_wait_reg_mem__always_pass = 0,
|
||||
function__mec_wait_reg_mem__less_than_ref_value = 1,
|
||||
function__mec_wait_reg_mem__less_than_equal_to_the_ref_value = 2,
|
||||
function__mec_wait_reg_mem__equal_to_the_reference_value = 3,
|
||||
function__mec_wait_reg_mem__not_equal_reference_value = 4,
|
||||
function__mec_wait_reg_mem__greater_than_or_equal_reference_value = 5,
|
||||
function__mec_wait_reg_mem__greater_than_reference_value = 6 };
|
||||
|
||||
enum MEC_WAIT_REG_MEM_mem_space_enum {
|
||||
mem_space__mec_wait_reg_mem__register_space = 0,
|
||||
mem_space__mec_wait_reg_mem__memory_space = 1 };
|
||||
|
||||
enum MEC_WAIT_REG_MEM_operation_enum {
|
||||
operation__mec_wait_reg_mem__wait_reg_mem = 0,
|
||||
operation__mec_wait_reg_mem__wr_wait_wr_reg = 1,
|
||||
operation__mec_wait_reg_mem__wait_mem_preemptable = 3 };
|
||||
|
||||
|
||||
typedef struct PM4_MEC_WAIT_REG_MEM
|
||||
{
|
||||
union
|
||||
{
|
||||
PM4_TYPE_3_HEADER header; ///header
|
||||
uint32_t ordinal1;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
MEC_WAIT_REG_MEM_function_enum function:3;
|
||||
uint32_t reserved1:1;
|
||||
MEC_WAIT_REG_MEM_mem_space_enum mem_space:2;
|
||||
MEC_WAIT_REG_MEM_operation_enum operation:2;
|
||||
uint32_t reserved2:24;
|
||||
} bitfields2;
|
||||
uint32_t ordinal2;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
uint32_t reserved3:2;
|
||||
uint32_t mem_poll_addr_lo:30;
|
||||
} bitfields3a;
|
||||
struct
|
||||
{
|
||||
uint32_t reg_poll_addr:18;
|
||||
uint32_t reserved4:14;
|
||||
} bitfields3b;
|
||||
struct
|
||||
{
|
||||
uint32_t reg_write_addr1:18;
|
||||
uint32_t reserved5:14;
|
||||
} bitfields3c;
|
||||
uint32_t ordinal3;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
uint32_t mem_poll_addr_hi;
|
||||
|
||||
struct
|
||||
{
|
||||
uint32_t reg_write_addr2:18;
|
||||
uint32_t reserved6:14;
|
||||
} bitfields4b;
|
||||
uint32_t ordinal4;
|
||||
};
|
||||
|
||||
uint32_t reference;
|
||||
|
||||
uint32_t mask;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
uint32_t poll_interval:16;
|
||||
uint32_t reserved7:15;
|
||||
uint32_t optimize_ace_offload_mode:1;
|
||||
} bitfields7;
|
||||
uint32_t ordinal7;
|
||||
};
|
||||
|
||||
} PM4MEC_WAIT_REG_MEM, *PPM4MEC_WAIT_REG_MEM;
|
||||
|
||||
//--------------------MEC_WRITE_DATA--------------------
|
||||
|
||||
enum MEC_WRITE_DATA_dst_sel_enum { dst_sel_mec_write_data_MEM_MAPPED_REGISTER_0 = 0, dst_sel_mec_write_data_TC_L2_2 = 2, dst_sel_mec_write_data_GDS_3 = 3, dst_sel_mec_write_data_MEMORY_5 = 5 };
|
||||
enum MEC_WRITE_DATA_addr_incr_enum { addr_incr_mec_write_data_INCREMENT_ADDR_0 = 0, addr_incr_mec_write_data_DO_NOT_INCREMENT_ADDR_1 = 1 };
|
||||
enum MEC_WRITE_DATA_wr_confirm_enum { wr_confirm_mec_write_data_DO_NOT_WAIT_FOR_CONFIRMATION_0 = 0, wr_confirm_mec_write_data_WAIT_FOR_CONFIRMATION_1 = 1 };
|
||||
enum MEC_WRITE_DATA_cache_policy_enum { cache_policy_mec_write_data_LRU_0 = 0, cache_policy_mec_write_data_STREAM_1 = 1, cache_policy_mec_write_data_BYPASS_2 = 2 };
|
||||
|
||||
//--------------------MEC_RELEASE_MEM--------------------
|
||||
|
||||
enum MEC_RELEASE_MEM_event_index_enum { event_index_mec_release_mem_EVENT_WRITE_EOP_5 = 5, event_index_mec_release_mem_CS_Done_6 = 6 };
|
||||
enum MEC_RELEASE_MEM_cache_policy_enum { cache_policy_mec_release_mem_LRU_0 = 0, cache_policy_mec_release_mem_STREAM_1 = 1, cache_policy_mec_release_mem_BYPASS_2 = 2 };
|
||||
enum MEC_RELEASE_MEM_dst_sel_enum { dst_sel_mec_release_mem_MEMORY_CONTROLLER_0 = 0, dst_sel_mec_release_mem_TC_L2_1 = 1 };
|
||||
enum MEC_RELEASE_MEM_int_sel_enum { int_sel_mec_release_mem_NONE_0 = 0, int_sel_mec_release_mem_SEND_INTERRUPT_ONLY_1 = 1, int_sel_mec_release_mem_SEND_INTERRUPT_AFTER_WRITE_CONFIRM_2 = 2, int_sel_mec_release_mem_SEND_DATA_AFTER_WRITE_CONFIRM_3 = 3 };
|
||||
enum MEC_RELEASE_MEM_data_sel_enum { data_sel_mec_release_mem_NONE_0 = 0, data_sel_mec_release_mem_SEND_32_BIT_LOW_1 = 1, data_sel_mec_release_mem_SEND_64_BIT_DATA_2 = 2, data_sel_mec_release_mem_SEND_GPU_CLOCK_COUNTER_3 = 3, data_sel_mec_release_mem_SEND_CP_PERFCOUNTER_HI_LO_4 = 4, data_sel_mec_release_mem_STORE_GDS_DATA_TO_MEMORY_5 = 5 };
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
/*
|
||||
* Copyright 2018 Advanced Micro Devices, Inc.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __PM4__PKT__STRUCT__NV__HPP__
|
||||
#define __PM4__PKT__STRUCT__NV__HPP__
|
||||
|
||||
#include "pm4_pkt_struct_ai.h"
|
||||
|
||||
typedef struct _PM4_ACQUIRE_MEM_NV
|
||||
{
|
||||
union
|
||||
{
|
||||
PM4_TYPE_3_HEADER header; ///header
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
unsigned int reserved;
|
||||
|
||||
unsigned int coher_size;
|
||||
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int coher_size_hi:8;
|
||||
unsigned int reserved1:24;
|
||||
} bitfields3;
|
||||
unsigned int ordinal4;
|
||||
};
|
||||
|
||||
unsigned int coher_base_lo;
|
||||
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int coher_base_hi:24;
|
||||
unsigned int reserved2:8;
|
||||
} bitfields4;
|
||||
unsigned int ordinal6;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int poll_interval:16;
|
||||
unsigned int reserved3:16;
|
||||
} bitfields5;
|
||||
unsigned int ordinal7;
|
||||
};
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int gcr_cntl:18;
|
||||
unsigned int reserved4:14;
|
||||
} bitfields6;
|
||||
unsigned int ordinal8;
|
||||
};
|
||||
|
||||
|
||||
} PM4ACQUIRE_MEM_NV, *PPM4ACQUIRE_MEM_NV;
|
||||
|
||||
typedef struct PM4_MEC_RELEASE_MEM_NV {
|
||||
union {
|
||||
PM4_TYPE_3_HEADER header;
|
||||
unsigned int ordinal1;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int event_type:6;
|
||||
unsigned int reserved1:2;
|
||||
AI_MEC_RELEASE_MEM_event_index_enum event_index:4;
|
||||
unsigned int gcr_cntl:12;
|
||||
unsigned int reserved4:1;
|
||||
AI_MEC_RELEASE_MEM_cache_policy_enum cache_policy:2;
|
||||
unsigned int reserved5:1;
|
||||
AI_MEC_RELEASE_MEM_pq_exe_status_enum pq_exe_status:1;
|
||||
unsigned int reserved6:3;
|
||||
} bitfields2;
|
||||
unsigned int ordinal2;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int reserved7:16;
|
||||
AI_MEC_RELEASE_MEM_dst_sel_enum dst_sel:2;
|
||||
unsigned int reserved8:6;
|
||||
AI_MEC_RELEASE_MEM_int_sel_enum int_sel:3;
|
||||
unsigned int reserved9:2;
|
||||
AI_MEC_RELEASE_MEM_data_sel_enum data_sel:3;
|
||||
} bitfields3;
|
||||
unsigned int ordinal3;
|
||||
};
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int reserved10:2;
|
||||
unsigned int address_lo_32b:30;
|
||||
} bitfields4a;
|
||||
struct {
|
||||
unsigned int reserved11:3;
|
||||
unsigned int address_lo_64b:29;
|
||||
} bitfields4b;
|
||||
unsigned int reserved12;
|
||||
|
||||
unsigned int ordinal4;
|
||||
};
|
||||
|
||||
union {
|
||||
unsigned int address_hi;
|
||||
|
||||
unsigned int reserved13;
|
||||
|
||||
unsigned int ordinal5;
|
||||
};
|
||||
|
||||
union {
|
||||
unsigned int data_lo;
|
||||
|
||||
unsigned int cmp_data_lo;
|
||||
|
||||
struct {
|
||||
unsigned int dw_offset:16;
|
||||
unsigned int num_dwords:16;
|
||||
} bitfields6c;
|
||||
unsigned int reserved14;
|
||||
|
||||
unsigned int ordinal6;
|
||||
};
|
||||
|
||||
union {
|
||||
unsigned int data_hi;
|
||||
|
||||
unsigned int cmp_data_hi;
|
||||
|
||||
unsigned int reserved15;
|
||||
|
||||
unsigned int reserved16;
|
||||
|
||||
unsigned int ordinal7;
|
||||
};
|
||||
|
||||
unsigned int int_ctxid;
|
||||
} PM4MEC_RELEASE_MEM_NV, *PPM4MEC_RELEASE_MEM_NV;
|
||||
|
||||
|
||||
#endif // __PM4__PKT__STRUCT__NV__HPP__
|
||||
@@ -0,0 +1,443 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __SDMA_PKT_STRUCT_H__
|
||||
#define __SDMA_PKT_STRUCT_H__
|
||||
|
||||
|
||||
const unsigned int SDMA_OP_NOP = 0;
|
||||
const unsigned int SDMA_OP_COPY = 1;
|
||||
const unsigned int SDMA_OP_WRITE = 2;
|
||||
|
||||
const unsigned int SDMA_OP_FENCE = 5;
|
||||
const unsigned int SDMA_OP_TRAP = 6;
|
||||
const unsigned int SDMA_OP_POLL_REGMEM = 8;
|
||||
const unsigned int SDMA_OP_TIMESTAMP = 13;
|
||||
|
||||
const unsigned int SDMA_OP_CONST_FILL = 11;
|
||||
|
||||
const unsigned int SDMA_SUBOP_COPY_LINEAR = 0;
|
||||
|
||||
const unsigned int SDMA_SUBOP_WRITE_LINEAR = 0;
|
||||
|
||||
/*
|
||||
** Definitions for SDMA_PKT_COPY_LINEAR packet
|
||||
*/
|
||||
|
||||
typedef struct SDMA_PKT_COPY_LINEAR_TAG
|
||||
{
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int op:8;
|
||||
unsigned int sub_op:8;
|
||||
unsigned int reserved_0:11;
|
||||
unsigned int broadcast:1;
|
||||
unsigned int reserved_1:4;
|
||||
};
|
||||
unsigned int DW_0_DATA;
|
||||
} HEADER_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int count:22;
|
||||
unsigned int reserved_0:10;
|
||||
};
|
||||
unsigned int DW_1_DATA;
|
||||
} COUNT_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int reserved_0:16;
|
||||
unsigned int dst_sw:2;
|
||||
unsigned int reserved_1:4;
|
||||
unsigned int dst_ha:1;
|
||||
unsigned int reserved_2:1;
|
||||
unsigned int src_sw:2;
|
||||
unsigned int reserved_3:4;
|
||||
unsigned int src_ha:1;
|
||||
unsigned int reserved_4:1;
|
||||
};
|
||||
unsigned int DW_2_DATA;
|
||||
} PARAMETER_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int src_addr_31_0:32;
|
||||
};
|
||||
unsigned int DW_3_DATA;
|
||||
} SRC_ADDR_LO_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int src_addr_63_32:32;
|
||||
};
|
||||
unsigned int DW_4_DATA;
|
||||
} SRC_ADDR_HI_UNION;
|
||||
|
||||
struct
|
||||
{
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int dst_addr_31_0:32;
|
||||
};
|
||||
unsigned int DW_5_DATA;
|
||||
} DST_ADDR_LO_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int dst_addr_63_32:32;
|
||||
};
|
||||
unsigned int DW_6_DATA;
|
||||
} DST_ADDR_HI_UNION;
|
||||
} DST_ADDR[0];
|
||||
} SDMA_PKT_COPY_LINEAR, *PSDMA_PKT_COPY_LINEAR;
|
||||
|
||||
/*
|
||||
** Definitions for SDMA_PKT_WRITE_UNTILED packet
|
||||
*/
|
||||
|
||||
typedef struct SDMA_PKT_WRITE_UNTILED_TAG
|
||||
{
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int op:8;
|
||||
unsigned int sub_op:8;
|
||||
unsigned int reserved_0:16;
|
||||
};
|
||||
unsigned int DW_0_DATA;
|
||||
} HEADER_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int dst_addr_31_0:32;
|
||||
};
|
||||
unsigned int DW_1_DATA;
|
||||
} DST_ADDR_LO_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int dst_addr_63_32:32;
|
||||
};
|
||||
unsigned int DW_2_DATA;
|
||||
} DST_ADDR_HI_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int count:22;
|
||||
unsigned int reserved_0:2;
|
||||
unsigned int sw:2;
|
||||
unsigned int reserved_1:6;
|
||||
};
|
||||
unsigned int DW_3_DATA;
|
||||
} DW_3_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int data0:32;
|
||||
};
|
||||
unsigned int DW_4_DATA;
|
||||
} DATA0_UNION;
|
||||
} SDMA_PKT_WRITE_UNTILED, *PSDMA_PKT_WRITE_UNTILED;
|
||||
|
||||
/*
|
||||
** Definitions for SDMA_PKT_FENCE packet
|
||||
*/
|
||||
|
||||
typedef struct SDMA_PKT_FENCE_TAG
|
||||
{
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int op:8;
|
||||
unsigned int sub_op:8;
|
||||
unsigned int reserved_0:16;
|
||||
};
|
||||
unsigned int DW_0_DATA;
|
||||
} HEADER_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int addr_31_0:32;
|
||||
};
|
||||
unsigned int DW_1_DATA;
|
||||
} ADDR_LO_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int addr_63_32:32;
|
||||
};
|
||||
unsigned int DW_2_DATA;
|
||||
} ADDR_HI_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int data:32;
|
||||
};
|
||||
unsigned int DW_3_DATA;
|
||||
} DATA_UNION;
|
||||
} SDMA_PKT_FENCE, *PSDMA_PKT_FENCE;
|
||||
|
||||
/*
|
||||
** Definitions for SDMA_PKT_CONSTANT_FILL packet
|
||||
*/
|
||||
|
||||
typedef struct SDMA_PKT_CONSTANT_FILL_TAG
|
||||
{
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int op:8;
|
||||
unsigned int sub_op:8;
|
||||
unsigned int sw:2;
|
||||
unsigned int reserved_0:12;
|
||||
unsigned int fillsize:2;
|
||||
};
|
||||
unsigned int DW_0_DATA;
|
||||
} HEADER_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int dst_addr_31_0:32;
|
||||
};
|
||||
unsigned int DW_1_DATA;
|
||||
} DST_ADDR_LO_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int dst_addr_63_32:32;
|
||||
};
|
||||
unsigned int DW_2_DATA;
|
||||
} DST_ADDR_HI_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int src_data_31_0:32;
|
||||
};
|
||||
unsigned int DW_3_DATA;
|
||||
} DATA_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int count:22;
|
||||
unsigned int reserved_0:10;
|
||||
};
|
||||
unsigned int DW_4_DATA;
|
||||
} COUNT_UNION;
|
||||
} SDMA_PKT_CONSTANT_FILL, *PSDMA_PKT_CONSTANT_FILL;
|
||||
|
||||
/*
|
||||
** Definitions for SDMA_PKT_TRAP packet
|
||||
*/
|
||||
|
||||
typedef struct SDMA_PKT_TRAP_TAG
|
||||
{
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int op:8;
|
||||
unsigned int sub_op:8;
|
||||
unsigned int reserved_0:16;
|
||||
};
|
||||
unsigned int DW_0_DATA;
|
||||
} HEADER_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int int_context:28;
|
||||
unsigned int reserved_0:4;
|
||||
};
|
||||
unsigned int DW_1_DATA;
|
||||
} INT_CONTEXT_UNION;
|
||||
} SDMA_PKT_TRAP, *PSDMA_PKT_TRAP;
|
||||
|
||||
/*
|
||||
** Definitions for SDMA_PKT_POLL_REGMEM_TAG packet
|
||||
*/
|
||||
|
||||
typedef struct SDMA_PKT_POLL_REGMEM_TAG {
|
||||
union {
|
||||
struct {
|
||||
unsigned int op : 8;
|
||||
unsigned int sub_op : 8;
|
||||
unsigned int reserved_0 : 10;
|
||||
unsigned int hdp_flush : 1;
|
||||
unsigned int reserved_1 : 1;
|
||||
unsigned int func : 3;
|
||||
unsigned int mem_poll : 1;
|
||||
};
|
||||
unsigned int DW_0_DATA;
|
||||
} HEADER_UNION;
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int addr_31_0 : 32;
|
||||
};
|
||||
unsigned int DW_1_DATA;
|
||||
} ADDR_LO_UNION;
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int addr_63_32 : 32;
|
||||
};
|
||||
unsigned int DW_2_DATA;
|
||||
} ADDR_HI_UNION;
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int value : 32;
|
||||
};
|
||||
unsigned int DW_3_DATA;
|
||||
} VALUE_UNION;
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int mask : 32;
|
||||
};
|
||||
unsigned int DW_4_DATA;
|
||||
} MASK_UNION;
|
||||
|
||||
union {
|
||||
struct {
|
||||
unsigned int interval : 16;
|
||||
unsigned int retry_count : 12;
|
||||
unsigned int reserved_0 : 4;
|
||||
};
|
||||
unsigned int DW_5_DATA;
|
||||
} DW5_UNION;
|
||||
} SDMA_PKT_POLL_REGMEM, *PSDMA_PKT_POLL_REGMEM;
|
||||
|
||||
/*
|
||||
** Definitions for SDMA_PKT_TIMESTAMP packet
|
||||
*/
|
||||
|
||||
typedef struct SDMA_PKT_TIMESTAMP_TAG
|
||||
{
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int op:8;
|
||||
unsigned int sub_op:8;
|
||||
unsigned int reserved_0:16;
|
||||
};
|
||||
unsigned int DW_0_DATA;
|
||||
} HEADER_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int addr_31_0:32;
|
||||
};
|
||||
unsigned int DW_1_DATA;
|
||||
} ADDR_LO_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int addr_63_32:32;
|
||||
};
|
||||
unsigned int DW_2_DATA;
|
||||
} ADDR_HI_UNION;
|
||||
} SDMA_PKT_TIMESTAMP, *PSDMA_PKT_TIMESTAMP;
|
||||
|
||||
|
||||
/*
|
||||
** Definitions for SDMA_PKT_NOP packet
|
||||
*/
|
||||
|
||||
typedef struct SDMA_PKT_NOP_TAG
|
||||
{
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int op:8;
|
||||
unsigned int sub_op:8;
|
||||
unsigned int count:14;
|
||||
unsigned int reserved_0:2;
|
||||
};
|
||||
unsigned int DW_0_DATA;
|
||||
} HEADER_UNION;
|
||||
|
||||
union
|
||||
{
|
||||
struct
|
||||
{
|
||||
unsigned int data0:32;
|
||||
};
|
||||
unsigned int DW_1_DATA;
|
||||
} DATA0_UNION;
|
||||
} SDMA_PKT_NOP, *PSDMA_PKT_NOP;
|
||||
|
||||
#endif // __SDMA_PKT_STRUCT_H__
|
||||
@@ -0,0 +1,379 @@
|
||||
declare -A FILTER
|
||||
|
||||
# Power management tests
|
||||
FILTER[pm]=\
|
||||
"KFDPMTest.SuspendWithActiveProcess:"\
|
||||
"KFDPMTest.SuspendWithIdleQueue:"\
|
||||
"KFDPMTest.SuspendWithIdleQueueAfterWork"
|
||||
|
||||
|
||||
# Core tests, used in scenarios like bringup
|
||||
# Software scheduler mode, i. e. non HWS mode
|
||||
FILTER[core_sws]=\
|
||||
"KFDQMTest.CreateDestroyCpQueue:"\
|
||||
"KFDQMTest.SubmitNopCpQueue:"\
|
||||
"KFDQMTest.SubmitPacketCpQueue:"\
|
||||
"KFDQMTest.AllCpQueues:"\
|
||||
"KFDQMTest.CreateDestroySdmaQueue:"\
|
||||
"KFDQMTest.SubmitNopSdmaQueue:"\
|
||||
"KFDQMTest.SubmitPacketSdmaQueue:"\
|
||||
"KFDQMTest.AllSdmaQueues:"\
|
||||
"KFDQMTest.AllXgmiSdmaQueues:"\
|
||||
"KFDQMTest.AllQueues:"\
|
||||
"KFDLocalMemoryTest.AccessLocalMem:"\
|
||||
"KFDEventTest.SignalEvent"
|
||||
|
||||
# HWS mode
|
||||
FILTER[core]=\
|
||||
"${FILTER[core_sws]}:"\
|
||||
"KFDCWSRTest.BasicTest"
|
||||
|
||||
# Permanent exclusions
|
||||
# These tests are included for debugging, but are not executed in normal execution on any ASIC:
|
||||
# FILTER[pm] need human intervention, so put it here. Developers can run them
|
||||
# manually through "-p pm" option.
|
||||
#
|
||||
# CU Masking Linear are not working correctly due to how the HW distributes work over CUs.
|
||||
# They are available for testing but are not currently expected to pass on CI/VI/AI.
|
||||
#
|
||||
# CU Masking Even is added here due to some non-obvious baseline measurements. Though
|
||||
# using wallclock to measure performance is always risky, there are just too many ASICs
|
||||
# where this test is failing. Ideally we'll get better CU Masking coverage via rocrtst
|
||||
#
|
||||
# The CheckZeroInitializationVram test is no longer expected to pass as KFD no longer
|
||||
# clears memory at allocation time.
|
||||
PERMANENT_BLACKLIST_ALL_ASICS=\
|
||||
"-${FILTER[pm]}:"\
|
||||
"KFDQMTest.BasicCuMaskingLinear:"\
|
||||
"KFDQMTest.BasicCuMaskingEven:"\
|
||||
"RDMATest.GPUDirect:"\
|
||||
"KFDLocalMemoryTest.CheckZeroInitializationVram"
|
||||
|
||||
# This is the temporary blacklist for all ASICs. This is to be used when a test is failing consistently
|
||||
# on every ASIC (Kaveri, Carrizo, Hawaii, Tonga, Fiji, Polaris10, Polaris11 and Vega10 .
|
||||
# TODO means that a JIRA ticket needs to be created for this issue, as no documentation regarding
|
||||
# failures can be found
|
||||
# NOTE: If you update this alphabetical listing, add the corresponding JIRA ticket for reference
|
||||
#
|
||||
# KFDQMTest.GPUDoorbellWrite fails intermittently (KFD-318)
|
||||
# KFDQMTest.mGPUShareBO (KFD-334)
|
||||
# KFDHWSTest.* (SWDEV-193035)
|
||||
# KFDEvictTest.BurstyTest (ROCMOPS-464)
|
||||
# KFDEvictTest.BurstyTest (SWDEV-291256)
|
||||
# KFDEvictTest.BurstyTest (KFD-425)
|
||||
# KFDDBGTest.SuspendQueues (SWDEV-417850)
|
||||
# KFDDBGTest.HitAddressWatch (SWDEV-420281)
|
||||
TEMPORARY_BLACKLIST_ALL_ASICS=\
|
||||
"KFDQMTest.GPUDoorbellWrite:"\
|
||||
"KFDQMTest.mGPUShareBO:"\
|
||||
"KFDQMTest.SdmaEventInterrupt:"\
|
||||
"KFDMemoryTest.CacheInvalidateOnRemoteWrite:"\
|
||||
"KFDEvictTest.BurstyTest:"\
|
||||
"KFDHWSTest.*:"\
|
||||
"KFDSVMRangeTest.ReadOnlyRangeTest*:"\
|
||||
"KFDDBGTest.SuspendQueues:"\
|
||||
"KFDDBGTest.HitAddressWatch"
|
||||
|
||||
BLACKLIST_ALL_ASICS=\
|
||||
"$PERMANENT_BLACKLIST_ALL_ASICS:"\
|
||||
"$TEMPORARY_BLACKLIST_ALL_ASICS"
|
||||
|
||||
# SDMA-based tests (KFDIPCTest.BasicTest, KFDQM.*Sdma*, KFDMemoryTest.MMBench) are all
|
||||
# disabled on non-Hawaii due to SDMA instability - SWDEV-101666
|
||||
SDMA_BLACKLIST=\
|
||||
"KFDIPCTest.*:"\
|
||||
"KFDLocalMemoryTest.CheckZeroInitializationVram:"\
|
||||
"KFDMemoryTest.MemoryRegister:"\
|
||||
"KFDMemoryTest.MMBench:"\
|
||||
"KFDMemoryTest.SignalHandling:"\
|
||||
"KFDQMTest.AllQueues:"\
|
||||
"KFDQMTest.*Sdma*:"\
|
||||
"KFDQMTest.CreateQueueStressSingleThreaded:"\
|
||||
"KFDQMTest.GPUDoorbellWrite:"\
|
||||
"KFDQMTest.P2PTest:"\
|
||||
"KFDPerformanceTest.P2PBandWidthTest:"\
|
||||
"KFDPerformanceTest.P2POverheadTest"
|
||||
|
||||
# Anything involving CP queue creation is failing on Kaveri. Separate them here for convenience (KFD-336)
|
||||
KV_QUEUE_BLACKLIST=\
|
||||
"KFDExceptionTest.AddressFault:"\
|
||||
"KFDExceptionTest.PermissionFault:"\
|
||||
"KFDLocalMemoryTest.*:"\
|
||||
"KFDEventTest.Signal*Event*:"\
|
||||
"KFDQMTest.CreateQueueStressSingleThreaded:"\
|
||||
"KFDQMTest.*CpQueue*:"\
|
||||
"KFDQMTest.*Dispatch*:"\
|
||||
"KFDQMTest.Atomics:"\
|
||||
"KFDQMTest.GPUDoorbellWrite"
|
||||
|
||||
# KFDCWSRTest.BasicTest*: SWDEV-353206
|
||||
BLACKLIST_GFX10=\
|
||||
"KFDMemoryTest.DeviceHdpFlush:"\
|
||||
"KFDSVMEvictTest.*:"\
|
||||
"KFDCWSRTest.BasicTest*"
|
||||
|
||||
BLACKLIST_GFX10_NV2X=\
|
||||
"$BLACKLIST_GFX10:"\
|
||||
"KFDPerfCountersTest.*"
|
||||
|
||||
# KFDMemoryTest.FlatScratchAccess - SWDEV-329877
|
||||
# KFDGWSTest.*: GFX11 will no longer use global wave sync
|
||||
BLACKLIST_GFX11=\
|
||||
"KFDQMTest.CreateAqlCpQueue:"\
|
||||
"KFDCWSRTest.InterruptRestore:"\
|
||||
"KFDPerfCountersTest.*:"\
|
||||
"KFDMemoryTest.FlatScratchAccess:"\
|
||||
"KFDGWSTest.*"
|
||||
|
||||
BLACKLIST_GFX12=\
|
||||
"KFDQMTest.CreateAqlCpQueue:"\
|
||||
"KFDPerfCountersTest.*:"\
|
||||
"KFDMemoryTest.FlatScratchAccess:"\
|
||||
"KFDGWSTest.*"
|
||||
|
||||
# KFDQMTest.CpuWriteCoherence fails. 0 dwordsAvailable (KFD-338)
|
||||
# KFDMemoryTest.MemoryRegister fails on SDMA queue creation (KFD-337)
|
||||
FILTER[kaveri]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST:"\
|
||||
"$KV_QUEUE_BLACKLIST:"\
|
||||
"KFDMemoryTest.MemoryRegister:"\
|
||||
"KFDQMTest.CpuWriteCoherence"
|
||||
|
||||
# KFDLocalMemoryTest.BasicTest is failing intermittently (KFD-368)
|
||||
# KFDMemoryTest.BigSysBufferStressTest was failing intermittently on 4.9
|
||||
# and hangs when executed twice (KFD-312)
|
||||
# KFDQMTest.GPUDoorbellWrite fails on Hawaii. Could be HW-related (KFD-342)
|
||||
FILTER[hawaii]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDLocalMemoryTest.BasicTest:"\
|
||||
"KFDMemoryTest.BigSysBufferStressTest:"\
|
||||
"KFDQMTest.GPUDoorbellWrite"
|
||||
|
||||
FILTER[carrizo]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST:"\
|
||||
"KFDExceptionTest.PermissionFault"
|
||||
|
||||
# KFDPerfCountersTest.*Trace fail (KFD-339)
|
||||
# KFDMemoryTest.QueryPointerInfo/MemoryRegister* (KFD-341)
|
||||
# The remaining tests listed here fail on map memory to GPU with a VA conflict (KFD-340)
|
||||
FILTER[tonga]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST:"\
|
||||
"KFDCWSRTest.BasicTest:"\
|
||||
"KFDPerfCountersTest.*:"\
|
||||
"KFDQMTest.OverSubscribeCpQueues"
|
||||
|
||||
# Since Navi10 was merged, the PM4Event test takes 6min to run
|
||||
FILTER[fiji]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDQMTest.PM4EventInterrupt:"\
|
||||
"$SDMA_BLACKLIST"
|
||||
|
||||
FILTER[polaris10]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST"
|
||||
|
||||
FILTER[polaris11]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST"
|
||||
|
||||
FILTER[polaris12]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST"
|
||||
|
||||
# KFDIPCTest.BasicTest (ROCMOPS-459) .CMABasicTest (ROCMOPS-460) .CrossMemoryAttachTest (ROCMOPS-461)
|
||||
# KFDQMTest.AllSdmaQueues (ROCMOPS-463)
|
||||
FILTER[vega10]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDIPCTest.BasicTest:"\
|
||||
"KFDIPCTest.CMABasicTest:"\
|
||||
"KFDIPCTest.CrossMemoryAttachTest:"\
|
||||
"KFDQMTest.AllSdmaQueues"
|
||||
|
||||
FILTER[vega12]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST"\
|
||||
|
||||
FILTER[vega20]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST:"\
|
||||
"KFDQMTest.GPUDoorbellWrite"
|
||||
|
||||
FILTER[raven_dgpuFallback]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST:"\
|
||||
"KFDEvictTest.*:"\
|
||||
"KFDMemoryTest.MemoryRegister:"\
|
||||
"KFDSVMRangeTest.BasicSystemMemTest:"\
|
||||
"KFDSVMRangeTest.BasicVramTest:"\
|
||||
"KFDSVMRangeTest.EvictSystemRangeTest:"\
|
||||
"KFDSVMRangeTest.PartialUnmapSysMemTest:"\
|
||||
"KFDSVMRangeTest.MigrateTest:"\
|
||||
"KFDSVMRangeTest.MigratePolicyTest:"\
|
||||
"KFDSVMRangeTest.MigrateGranularityTest:"\
|
||||
"KFDSVMRangeTest.MigrateLargeBufTest:"\
|
||||
"KFDSVMRangeTest.MultiThreadMigrationTest:"\
|
||||
"KFDSVMRangeTest.MigrateAccessInPlaceTest:"\
|
||||
"KFDSVMEvictTest.QueueTest"
|
||||
|
||||
FILTER[raven]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$SDMA_BLACKLIST:"\
|
||||
"KFDEvictTest.*:"\
|
||||
"KFDSVMRangeTest.EvictSystemRangeTest:"\
|
||||
"KFDSVMRangeTest.PartialUnmapSysMemTest:"\
|
||||
"KFDSVMRangeTest.PrefetchTest:"\
|
||||
"KFDSVMRangeTest.MultiThreadMigrationTest:"\
|
||||
"KFDSVMEvictTest.QueueTest:"\
|
||||
"KFDQMTest.MultipleCpQueuesStressDispatch"
|
||||
|
||||
FILTER[renoir]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDEvictTest.*:"\
|
||||
"KFDMemoryTest.LargestSysBufferTest:"\
|
||||
"KFDMemoryTest.SignalHandling"
|
||||
|
||||
# KFDExceptionTest.* (KFD-435)
|
||||
FILTER[arcturus]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDExceptionTest.FaultStorm:"\
|
||||
"KFDNegativeTest.*"
|
||||
|
||||
FILTER[aldebaran]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDExceptionTest.FaultStorm:"\
|
||||
"KFDMemoryTest.PtraceAccess:"\
|
||||
"KFDMemoryTest.DeviceHdpFlush"
|
||||
|
||||
FILTER[navi10]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10:"\
|
||||
"KFDMemoryTest.MMBench"
|
||||
|
||||
# Need to verify the following failed tests on another machine:
|
||||
# Exceptions not being received during exception tests
|
||||
# PerfCounters return HSAKMT_STATUS_INVALID_PARAMETER
|
||||
# P2PBandwidth failing (wait times out) on node-to-multiple-nodes by [push, NONE]
|
||||
FILTER[navi12]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10:"\
|
||||
"KFDExceptionTest.*:"\
|
||||
"KFDPerfCountersTest.*:"\
|
||||
"KFDPerformanceTest.P2PBandWidthTest"
|
||||
|
||||
FILTER[navi14]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10"
|
||||
|
||||
FILTER[sienna_cichlid]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10_NV2X"
|
||||
|
||||
FILTER[navy_flounder]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10_NV2X"
|
||||
|
||||
FILTER[dimgrey_cavefish]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10_NV2X"
|
||||
|
||||
FILTER[beige_goby]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10_NV2X"
|
||||
|
||||
FILTER[yellow_carp]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10_NV2X"
|
||||
|
||||
FILTER[gfx1100]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11"
|
||||
|
||||
# SWDEV-384028
|
||||
FILTER[gfx1101]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11:"\
|
||||
"KFDExceptionTest.SdmaQueueException"
|
||||
|
||||
FILTER[gfx1102]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11"
|
||||
|
||||
FILTER[gfx1103]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11"
|
||||
|
||||
FILTER[gfx1150]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11"
|
||||
|
||||
FILTER[gfx1151]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11"
|
||||
|
||||
FILTER[gfx1152]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11"
|
||||
|
||||
FILTER[gfx1153]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11"
|
||||
|
||||
FILTER[gfx1036]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX10_NV2X"
|
||||
|
||||
FILTER[gfx940]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDMemoryTest.LargestSysBufferTest:"\
|
||||
"KFDMemoryTest.BigSysBufferStressTest:"\
|
||||
"KFDMemoryTest.FlatScratchAccess:"\
|
||||
"KFDIPCTest.BasicTest:"\
|
||||
"KFDQMTest.QueueLatency"
|
||||
|
||||
FILTER[gfx941]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDMemoryTest.LargestSysBufferTest:"\
|
||||
"KFDMemoryTest.BigSysBufferStressTest:"\
|
||||
"KFDMemoryTest.FlatScratchAccess:"\
|
||||
"KFDIPCTest.BasicTest:"\
|
||||
"KFDQMTest.QueueLatency"
|
||||
|
||||
FILTER[gfx942]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDMemoryTest.LargestSysBufferTest:"\
|
||||
"KFDMemoryTest.BigSysBufferStressTest:"\
|
||||
"KFDMemoryTest.FlatScratchAccess:"\
|
||||
"KFDIPCTest.BasicTest:"\
|
||||
"KFDQMTest.QueueLatency"
|
||||
|
||||
FILTER[gfx950]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"KFDMemoryTest.LargestSysBufferTest:"\
|
||||
"KFDMemoryTest.BigSysBufferStressTest:"\
|
||||
"KFDMemoryTest.FlatScratchAccess:"\
|
||||
"KFDIPCTest.BasicTest:"\
|
||||
"KFDQMTest.QueueLatency:"\
|
||||
"KFDEvictTest.*:"\
|
||||
"KFDSVMEvictTest.QueueTest*:"\
|
||||
"KFDGWSTest.Semaphore"
|
||||
|
||||
FILTER[gfx1200]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX12"
|
||||
|
||||
FILTER[gfx1201]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX12"
|
||||
|
||||
FILTER[RHEL9]=\
|
||||
"$BLACKLIST_ALL_ASICS:"\
|
||||
"$BLACKLIST_GFX11:"\
|
||||
"KFDQMTest.ExtendedCuMasking:"\
|
||||
"KFDEvictTest.QueueTest:"\
|
||||
"KFDPCSamplingTest.*"
|
||||
|
||||
FILTER[upstream]=\
|
||||
"KFDIPCTest.*"
|
||||
@@ -0,0 +1,325 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# Copyright (C) 2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
#
|
||||
# Permission is hereby granted, free of charge, to any person obtaining a
|
||||
# copy of this software and associated documentation files (the "Software"),
|
||||
# to deal in the Software without restriction, including without limitation
|
||||
# the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
# and/or sell copies of the Software, and to permit persons to whom the
|
||||
# Software is furnished to do so, subject to the following conditions:
|
||||
#
|
||||
# The above copyright notice and this permission notice shall be included in
|
||||
# all copies or substantial portions of the Software.
|
||||
#
|
||||
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
# THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
# OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
# ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
# OTHER DEALINGS IN THE SOFTWARE.
|
||||
#
|
||||
#
|
||||
|
||||
# See if we can find the SHARE/BIN dirs in their expected locations
|
||||
CWD="${BASH_SOURCE%/*}"
|
||||
while read candidate; do
|
||||
if [ -e "$candidate/kfdtest.exclude" ]; then
|
||||
source "$candidate/kfdtest.exclude"
|
||||
break
|
||||
fi
|
||||
done <<EOF
|
||||
$KFDTEST_SHARE_DIR
|
||||
$CWD
|
||||
$CWD/../share/kfdtest
|
||||
/opt/rocm/share/kfdtest
|
||||
EOF
|
||||
|
||||
# Keep these checks until automation starts using the package install
|
||||
if [ -z "${FILTER[core]}" ]; then
|
||||
if [ -e "$CWD/../bin/kfdtest/kfdtest.exclude" ]; then
|
||||
source "$CWD/../bin/kfdtest/kfdtest.exclude"
|
||||
elif [ -e "$CWD/../../share/kfdtest.exclude" ]; then
|
||||
source "$CWD/../../share/kfdtest.exclude"
|
||||
fi
|
||||
fi
|
||||
|
||||
# This filter will always exist if we sourced a valid kfdtest.exclude
|
||||
if [ -z "${FILTER[core]}" ]; then
|
||||
echo "Unable to locate kfdtest.exclude."
|
||||
echo "Please set KFDTEST_SHARE_DIR or ensure that kfdtest.exclude is present inside $CWD, $CWD/../share/kfdtest or /opt/rocm/share/kfdtest"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Using "which" produces different results in different
|
||||
# OSes so use command -v instead. It returns "" if the
|
||||
# command isn't in the PATH
|
||||
if [ -z "$(command -v kfdtest)" ]; then
|
||||
if [ -z "$BIN_DIR" ]; then
|
||||
if [ -e "${0%/*}/kfdtest" ]; then
|
||||
BIN_DIR="${0%/*}"
|
||||
else
|
||||
# The default location
|
||||
BIN_DIR="/opt/rocm/bin"
|
||||
fi
|
||||
fi
|
||||
if [ -e "$BIN_DIR/kfdtest" ]; then
|
||||
KFDTEST="$BIN_DIR/kfdtest"
|
||||
else
|
||||
echo "Unable to locate kfdtest."
|
||||
echo "Please set BIN_DIR, ensure that kfdtest is in $PATH, or ensure that kfdtest is present inside ${0%/*} or /opt/rocm/bin"
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
KFDTEST="kfdtest"
|
||||
fi
|
||||
|
||||
PLATFORM=""
|
||||
GDB=""
|
||||
NODE=""
|
||||
FORCE_HIGH=""
|
||||
RUN_IN_DOCKER=""
|
||||
ADDITIONAL_EXCLUDE=""
|
||||
|
||||
printUsage() {
|
||||
echo
|
||||
echo "Usage: $(basename $0) [options ...] [gtest arguments]"
|
||||
echo
|
||||
echo "Options:"
|
||||
echo " -p <platform> , --platform <platform> Only run tests that"\
|
||||
"pass on the specified platform. Usually you"\
|
||||
"don't need this option"
|
||||
echo " -g , --gdb Run in debugger"
|
||||
echo " -n <node(s)> , --node <node(s)> NodeId(s) to test. Takes a single integer, or a"\
|
||||
"quoted, space-separated string as an argument"\
|
||||
"(e.g. -n 1 OR -n \"1 2 3\")"\
|
||||
"NOTE: Node numbers come from /sys/class/kfd/kfd/topology/nodes/#"
|
||||
echo " -l , --list List available nodes"
|
||||
echo " --high Force clocks to high for test execution"
|
||||
echo " -d , --docker Run in docker container"
|
||||
echo " -e <list> , --exclude <list> Additional tests to exclude, in addition to kfdtest.exclude."\
|
||||
"Takes a colon-separated string as an argument"\
|
||||
"(e.g. -e KFDEvictTest.*:KFDSVMEvictTest.*)"
|
||||
echo " -h , --help Prints this help"
|
||||
echo
|
||||
echo "Gtest arguments will be forwarded to the app"
|
||||
echo
|
||||
echo "Valid platform options: core_sws, core, polaris10, vega10, vega20, pm, all, and so on"
|
||||
echo "'all' option runs all tests"
|
||||
|
||||
return 0
|
||||
}
|
||||
# Print gtest_filter for the given Platform
|
||||
# param - Platform.
|
||||
getFilter() {
|
||||
# For regular platforms such as vega10, this will automatically generate
|
||||
# the valid variable BLACKLIST based on the variable platform.
|
||||
local platform=$1;
|
||||
|
||||
case "$platform" in
|
||||
all ) gtestFilter="" ;;
|
||||
* )
|
||||
if [ -z "${FILTER[$platform]}" ]; then
|
||||
echo "Unsupported platform $platform. Exiting"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
gtestFilter="--gtest_filter=${FILTER[$platform]}"
|
||||
;;
|
||||
esac
|
||||
|
||||
# Check if the loaded driver is upstream (in-box) or DKMS
|
||||
rdma_get_pages_func=$(cat /proc/kallsyms | grep rdma_get_pages)
|
||||
if [ -z "$rdma_get_pages_func" ]; then
|
||||
gtestFilter="$gtestFilter:${FILTER[upstream]}"
|
||||
fi
|
||||
|
||||
if [ -n "$ADDITIONAL_EXCLUDE" ]; then
|
||||
gtestFilter="$gtestFilter:$ADDITIONAL_EXCLUDE"
|
||||
fi
|
||||
}
|
||||
|
||||
TOPOLOGY_SYSFS_DIR=/sys/devices/virtual/kfd/kfd/topology/nodes
|
||||
|
||||
# Prints list of HSA Nodes. HSA Nodes are identified from sysfs KFD topology. The nodes
|
||||
# should have valid SIMD count
|
||||
getHsaNodes() {
|
||||
for i in $(find $TOPOLOGY_SYSFS_DIR -maxdepth 1 -mindepth 1 -type d); do
|
||||
simdcount=$(cat $i/properties | grep simd_count | awk '{print $2}')
|
||||
if [ $simdcount != 0 ]; then
|
||||
hsaNodeList+="$(basename $i) "
|
||||
fi
|
||||
done
|
||||
echo "$hsaNodeList"
|
||||
}
|
||||
|
||||
|
||||
# Prints GPU Name for the given Node ID. If transitioned to IP discovery,
|
||||
# use target gfx version
|
||||
# param - Node ID
|
||||
getNodeName() {
|
||||
local nodeId=$1; shift;
|
||||
local gpuName=$(cat $TOPOLOGY_SYSFS_DIR/$nodeId/name)
|
||||
if [ "$gpuName" == "raven" ]; then
|
||||
local CpuCoresCount=$(cat $TOPOLOGY_SYSFS_DIR/$nodeId/properties | grep cpu_cores_count | awk '{print $2}')
|
||||
local SimdCount=$(cat $TOPOLOGY_SYSFS_DIR/$nodeId/properties | grep simd_count | awk '{print $2}')
|
||||
if [ "$CpuCoresCount" -eq 0 ] && [ "$SimdCount" -gt 0 ]; then
|
||||
gpuName="raven_dgpuFallback"
|
||||
fi
|
||||
elif [ "$gpuName" == "ip discovery" ]; then
|
||||
if [ -n "$HSA_OVERRIDE_GFX_VERSION" ]; then
|
||||
gpuName="gfx$(echo "$HSA_OVERRIDE_GFX_VERSION" | awk 'BEGIN {FS="."; RS=""} {printf "%d%x%x", $1, $2, $3 }')"
|
||||
else
|
||||
local GfxVersionDec=$(cat $TOPOLOGY_SYSFS_DIR/$nodeId/properties | grep gfx_target_version | awk '{print $2}')
|
||||
if [[ ${#GfxVersionDec} = 5 ]]; then
|
||||
GfxVersionDec="0${GfxVersionDec}"
|
||||
fi
|
||||
gpuName="gfx$(printf "$GfxVersionDec" | fold -w2 | awk 'BEGIN {FS="\n"; RS=""} {printf "%d%x%x", $1, $2, $3}')"
|
||||
fi
|
||||
fi
|
||||
echo "$gpuName"
|
||||
}
|
||||
|
||||
# Run KfdTest independently. Two global variables set by command-line
|
||||
# will influence the tests as indicated below
|
||||
# PLATFORM - If set all tests will run with this platform filter
|
||||
# NODE - If set tests will be run only on this NODE, else it will be
|
||||
# run on all available HSA Nodes
|
||||
runKfdTest() {
|
||||
if [ "$RUN_IN_DOCKER" == "true" ]; then
|
||||
if [ `sudo systemctl is-active docker` != "active" ]; then
|
||||
echo "docker isn't active, install and setup docker first!!!!"
|
||||
exit 0
|
||||
fi
|
||||
PKG_ROOT="$(getPackageRoot)"
|
||||
fi
|
||||
|
||||
if [ -n "$GTEST_ARGS" ] && [ -n "$ADDITIONAL_EXCLUDE" ]; then
|
||||
echo "Cannot use -e and --gtest_filter flags together"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
if [ "$NODE" == "" ]; then
|
||||
hsaNodes=$(getHsaNodes)
|
||||
|
||||
if [ "$hsaNodes" == "" ]; then
|
||||
echo "No GPU found in the system."
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
hsaNodes=$NODE
|
||||
fi
|
||||
|
||||
for hsaNode in $hsaNodes; do
|
||||
nodeName=$(getNodeName $hsaNode)
|
||||
if [ "$PLATFORM" != "" ] && [ "$PLATFORM" != "$nodeName" ]; then
|
||||
echo "WARNING: Actual ASIC $nodeName treated as $PLATFORM"
|
||||
nodeName="$PLATFORM"
|
||||
fi
|
||||
|
||||
getFilter $nodeName
|
||||
|
||||
if [ "$RUN_IN_DOCKER" == "true" ]; then
|
||||
if [ "$NODE" == "" ]; then
|
||||
DEVICE_NODE="/dev/dri"
|
||||
else
|
||||
RENDER_NODE=$(($hsaNode + 127))
|
||||
DEVICE_NODE="/dev/dri/renderD${RENDER_NODE}"
|
||||
fi
|
||||
|
||||
echo "Starting testing node $hsaNode ($nodeName) in docker container"
|
||||
sudo docker run -it --name kfdtest_docker --user="jenkins" --network=host \
|
||||
--device=/dev/kfd --device=${DEVICE_NODE} --group-add video --cap-add=SYS_PTRACE \
|
||||
--security-opt seccomp=unconfined -v $PKG_ROOT:/home/jenkins/rocm \
|
||||
compute-artifactory.amd.com:5000/yuho/tianli-ubuntu1604-kfdtest:01 \
|
||||
/home/jenkins/rocm/utils/run_kfdtest.sh -n $hsaNode $gtestFilter $GTEST_ARGS
|
||||
if [ "$?" = "0" ]; then
|
||||
echo "Finished node $hsaNode ($nodeName) successfully in docker container"
|
||||
else
|
||||
echo "Testing failed for node $hsaNode ($nodeName) in docker container"
|
||||
fi
|
||||
sudo docker rm kfdtest_docker
|
||||
else
|
||||
if [ "$HSA_TEST_GPUS_NUM" != "" ]; then
|
||||
echo "++++ Starting parallel testing on $HSA_TEST_GPUS_NUM gpu(s) ++++"
|
||||
$GDB $KFDTEST $gtestFilter $GTEST_ARGS
|
||||
echo "++++ Finished parallel testing on $HSA_TEST_GPUS_NUM gpu(s) ++++"
|
||||
exit 0;
|
||||
else
|
||||
echo ""
|
||||
echo "++++ Starting testing node $hsaNode ($nodeName) ++++"
|
||||
$GDB $KFDTEST "--node=$hsaNode" $gtestFilter $GTEST_ARGS
|
||||
echo "---- Finished testing node $hsaNode ($nodeName) ----"
|
||||
fi
|
||||
|
||||
fi
|
||||
|
||||
|
||||
done
|
||||
|
||||
}
|
||||
|
||||
# Prints number of GPUs present in the system
|
||||
getGPUCount() {
|
||||
gNodes=$(getHsaNodes)
|
||||
gNodes=( $gNodes )
|
||||
gpuCount=${#gNodes[@]}
|
||||
echo "$gpuCount"
|
||||
}
|
||||
|
||||
while [ "$1" != "" ]; do
|
||||
case "$1" in
|
||||
-p | --platform )
|
||||
shift 1; PLATFORM=$1 ;;
|
||||
-g | --gdb )
|
||||
GDB="gdb --args" ;;
|
||||
-l | --list )
|
||||
printGpuNodelist; exit 0 ;;
|
||||
-n | --node )
|
||||
shift 1; NODE=$1 ;;
|
||||
--high)
|
||||
FORCE_HIGH="true" ;;
|
||||
-d | --docker )
|
||||
RUN_IN_DOCKER="true" ;;
|
||||
-e | --exclude )
|
||||
shift 1; ADDITIONAL_EXCLUDE="$1" ;;
|
||||
-h | --help )
|
||||
printUsage; exit 0 ;;
|
||||
*)
|
||||
GTEST_ARGS=$@; break;;
|
||||
esac
|
||||
shift 1
|
||||
done
|
||||
|
||||
# If the SMI is missing, try to find it
|
||||
SMI="$(find /opt/rocm* -type l -name rocm-smi 2>/dev/null | tail -1)"
|
||||
if [ -z ${SMI} ]; then
|
||||
if [ -x ${BIN_DIR}/rocm-smi ]; then
|
||||
SMI=${BIN_DIR}/rocm-smi
|
||||
else
|
||||
SMI=`which rocm-smi`
|
||||
fi
|
||||
fi
|
||||
# If the SMI is still missing, just report and continue
|
||||
if [ "$FORCE_HIGH" == "true" ]; then
|
||||
if [ -e "$SMI" ]; then
|
||||
OLDPERF=$($SMI -p | awk '/Performance Level:/ {print $NF; exit}')
|
||||
$($SMI --setperflevel high &> /dev/null)
|
||||
if [ $? != 0 ]; then
|
||||
echo "SMI failed to set perf level"
|
||||
OLDPERF=""
|
||||
fi
|
||||
else
|
||||
echo "Unable to set clocks to high, cannot find rocm-smi"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Set HSA_DEBUG env to run KFDMemoryTest.PtraceAccessInvisibleVram
|
||||
export HSA_DEBUG=1
|
||||
runKfdTest
|
||||
|
||||
# OLDPERF is only set if FORCE_HIGH and SMI both exist
|
||||
if [ -n "$OLDPERF" ]; then
|
||||
$SMI --setperflevel $OLDPERF &> /dev/null
|
||||
fi
|
||||
@@ -0,0 +1,52 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#include "AqlQueue.hpp"
|
||||
#include "GoogleTestExtension.hpp"
|
||||
|
||||
|
||||
AqlQueue::AqlQueue(void) {
|
||||
}
|
||||
|
||||
|
||||
AqlQueue::~AqlQueue(void) {
|
||||
}
|
||||
|
||||
unsigned int AqlQueue::Wptr() {
|
||||
return *m_Resources.Queue_write_ptr;
|
||||
}
|
||||
|
||||
unsigned int AqlQueue::Rptr() {
|
||||
return *m_Resources.Queue_read_ptr;
|
||||
}
|
||||
|
||||
unsigned int AqlQueue::RptrWhenConsumed() {
|
||||
return Wptr();
|
||||
}
|
||||
|
||||
void AqlQueue::SubmitPacket() {
|
||||
// m_pending Wptr is in dwords
|
||||
*m_Resources.Queue_write_ptr = m_pendingWptr;
|
||||
*(m_Resources.Queue_DoorBell) = Wptr();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __KFD_AQL_QUEUE__H__
|
||||
#define __KFD_AQL_QUEUE__H__
|
||||
|
||||
#include "BaseQueue.hpp"
|
||||
|
||||
class AqlQueue : public BaseQueue {
|
||||
public:
|
||||
AqlQueue();
|
||||
virtual ~AqlQueue();
|
||||
|
||||
// @brief Updates queue write pointer and sets the queue doorbell to the queue write pointer
|
||||
virtual void SubmitPacket();
|
||||
|
||||
// @return Read pointer in dwords
|
||||
virtual unsigned int Rptr();
|
||||
// @return Write pointer in dwords
|
||||
virtual unsigned int Wptr();
|
||||
// @return Expected m_Resources.Queue_read_ptr when all packets are consumed
|
||||
virtual unsigned int RptrWhenConsumed();
|
||||
|
||||
protected:
|
||||
virtual PACKETTYPE PacketTypeSupported() { return PACKETTYPE_AQL; }
|
||||
|
||||
virtual _HSA_QUEUE_TYPE GetQueueType() { return HSA_QUEUE_COMPUTE_AQL; }
|
||||
};
|
||||
|
||||
#endif // __KFD_AQL_QUEUE__H__
|
||||
@@ -0,0 +1,407 @@
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
//
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2022, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
// AMD Research and AMD HSA Software Development
|
||||
//
|
||||
// Advanced Micro Devices, Inc.
|
||||
//
|
||||
// www.amd.com
|
||||
//
|
||||
// Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
// of this software and associated documentation files (the "Software"), to
|
||||
// deal with the Software without restriction, including without limitation
|
||||
// the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
// and/or sell copies of the Software, and to permit persons to whom the
|
||||
// Software is furnished to do so, subject to the following conditions:
|
||||
//
|
||||
// - Redistributions of source code must retain the above copyright notice,
|
||||
// this list of conditions and the following disclaimers.
|
||||
// - Redistributions in binary form must reproduce the above copyright
|
||||
// notice, this list of conditions and the following disclaimers in
|
||||
// the documentation and/or other materials provided with the distribution.
|
||||
// - Neither the names of Advanced Micro Devices, Inc,
|
||||
// nor the names of its contributors may be used to endorse or promote
|
||||
// products derived from this Software without specific prior written
|
||||
// permission.
|
||||
//
|
||||
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
// THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
// OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
// ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
// DEALINGS WITH THE SOFTWARE.
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
/**
|
||||
* Self-contained assembler that uses the LLVM MC API to assemble AMDGCN
|
||||
* instructions
|
||||
*/
|
||||
|
||||
#include <llvm/Config/llvm-config.h>
|
||||
#include <llvm/MC/MCAsmBackend.h>
|
||||
#include <llvm/MC/MCAsmInfo.h>
|
||||
#include <llvm/MC/MCCodeEmitter.h>
|
||||
#include <llvm/MC/MCContext.h>
|
||||
#include <llvm/MC/MCInstPrinter.h>
|
||||
#include <llvm/MC/MCInstrInfo.h>
|
||||
#include <llvm/MC/MCObjectFileInfo.h>
|
||||
#include <llvm/MC/MCObjectWriter.h>
|
||||
#include <llvm/MC/MCParser/AsmLexer.h>
|
||||
#include <llvm/MC/MCParser/MCTargetAsmParser.h>
|
||||
#include <llvm/MC/MCRegisterInfo.h>
|
||||
#include <llvm/MC/MCStreamer.h>
|
||||
#include <llvm/MC/MCSubtargetInfo.h>
|
||||
#include <llvm/Support/CommandLine.h>
|
||||
#include <llvm/Support/InitLLVM.h>
|
||||
#include <llvm/Support/MemoryBuffer.h>
|
||||
#include <llvm/Support/SourceMgr.h>
|
||||
#include <llvm/Support/TargetSelect.h>
|
||||
#if LLVM_VERSION_MAJOR > 13
|
||||
#include <llvm/MC/TargetRegistry.h>
|
||||
#else
|
||||
#include <llvm/Support/TargetRegistry.h>
|
||||
#endif
|
||||
#if LLVM_VERSION_MAJOR > 18
|
||||
#include "llvm/Support/ManagedStatic.h"
|
||||
#endif
|
||||
|
||||
#include <linux/elf.h>
|
||||
#include "OSWrapper.hpp"
|
||||
#include "Assemble.hpp"
|
||||
|
||||
using namespace llvm;
|
||||
|
||||
/* Assembler implementation is not multi-thread safe and is
|
||||
* asic type dependent. Instantiate it per thread/gpu use case,
|
||||
* delete each assembler after assembling
|
||||
*/
|
||||
|
||||
void Init_LLVM() {
|
||||
LLVMInitializeAMDGPUTargetInfo();
|
||||
LLVMInitializeAMDGPUTargetMC();
|
||||
LLVMInitializeAMDGPUAsmParser();
|
||||
}
|
||||
|
||||
void Shutdown_LLVM() {
|
||||
llvm_shutdown();
|
||||
}
|
||||
|
||||
Assembler::Assembler(const uint32_t Gfxv) {
|
||||
SetTargetAsic(Gfxv);
|
||||
TextData = nullptr;
|
||||
TextSize = 0;
|
||||
}
|
||||
|
||||
Assembler::~Assembler() {
|
||||
FlushText();
|
||||
}
|
||||
|
||||
const char* Assembler::GetInstrStream() {
|
||||
return TextData;
|
||||
}
|
||||
|
||||
const size_t Assembler::GetInstrStreamSize() {
|
||||
return TextSize;
|
||||
}
|
||||
|
||||
int Assembler::CopyInstrStream(char* OutBuf, const size_t BufSize) {
|
||||
if (TextSize > BufSize)
|
||||
return -2;
|
||||
|
||||
std::copy(TextData, TextData + TextSize, OutBuf);
|
||||
return 0;
|
||||
}
|
||||
|
||||
const char* Assembler::GetTargetAsic() {
|
||||
return MCPU;
|
||||
}
|
||||
|
||||
/**
|
||||
* Set MCPU via GFX Version from Thunk
|
||||
* LLVM Target IDs use decimal for Maj/Min, hex for Step
|
||||
*/
|
||||
void Assembler::SetTargetAsic(const uint32_t Gfxv) {
|
||||
const uint8_t Major = (Gfxv >> 16) & 0xff;
|
||||
const uint8_t Minor = (Gfxv >> 8) & 0xff;
|
||||
const uint8_t Step = Gfxv & 0xff;
|
||||
|
||||
snprintf(MCPU, ASM_MCPU_LEN, "gfx%d%d%x", Major, Minor, Step);
|
||||
}
|
||||
|
||||
/**
|
||||
* Flush/reset TextData and TextSize to initial state
|
||||
*/
|
||||
void Assembler::FlushText() {
|
||||
if (TextData)
|
||||
delete[] TextData;
|
||||
TextData = nullptr;
|
||||
TextSize = 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Print hex of ELF object to stdout (debug)
|
||||
*/
|
||||
void Assembler::PrintELFHex(const std::string Data) {
|
||||
outs() << "ASM Info: assembled ELF hex data (length " << Data.length() << "):\n";
|
||||
outs() << "0x00:\t";
|
||||
for (size_t i = 0; i < Data.length(); ++i) {
|
||||
char c = Data[i];
|
||||
outs() << format_hex(static_cast<uint8_t>(c), 4);
|
||||
if ((i+1) % 16 == 0)
|
||||
outs() << "\n" << format_hex(i+1, 4) << ":\t";
|
||||
else
|
||||
outs() << " ";
|
||||
}
|
||||
outs() << "\n";
|
||||
}
|
||||
|
||||
/**
|
||||
* Print hex of raw instruction stream to stdout (debug)
|
||||
*/
|
||||
void Assembler::PrintTextHex() {
|
||||
outs() << "ASM Info: assembled .text hex data (length " << TextSize << "):\n";
|
||||
outs() << "0x00:\t";
|
||||
for (size_t i = 0; i < TextSize; i++) {
|
||||
outs() << format_hex(static_cast<uint8_t>(TextData[i]), 4);
|
||||
if ((i+1) % 16 == 0)
|
||||
outs() << "\n" << format_hex(i+1, 4) << ":\t";
|
||||
else
|
||||
outs() << " ";
|
||||
}
|
||||
outs() << "\n";
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract raw instruction stream from .text section in ELF object
|
||||
*
|
||||
* @param RawData Raw C string of ELF object
|
||||
* @return 0 on success
|
||||
*/
|
||||
int Assembler::ExtractELFText(const char* RawData) {
|
||||
const Elf64_Ehdr* ElfHeader;
|
||||
const Elf64_Shdr* SectHeader;
|
||||
const Elf64_Shdr* SectStrTable;
|
||||
const char* SectStrAddr;
|
||||
unsigned NumSects, SectIdx;
|
||||
|
||||
if (!(ElfHeader = reinterpret_cast<const Elf64_Ehdr*>(RawData))) {
|
||||
outs() << "ASM Error: elf data is invalid or corrupted\n";
|
||||
return -1;
|
||||
}
|
||||
if (ElfHeader->e_ident[EI_CLASS] != ELFCLASS64) {
|
||||
outs() << "ASM Error: elf object must be of 64-bit type\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
SectHeader = reinterpret_cast<const Elf64_Shdr*>(RawData + ElfHeader->e_shoff);
|
||||
SectStrTable = &SectHeader[ElfHeader->e_shstrndx];
|
||||
SectStrAddr = static_cast<const char*>(RawData + SectStrTable->sh_offset);
|
||||
|
||||
// Loop through sections, break on .text
|
||||
NumSects = ElfHeader->e_shnum;
|
||||
for (SectIdx = 0; SectIdx < NumSects; SectIdx++) {
|
||||
std::string SectName = std::string(SectStrAddr + SectHeader[SectIdx].sh_name);
|
||||
if (SectName == std::string(".text")) {
|
||||
TextSize = SectHeader[SectIdx].sh_size;
|
||||
TextData = new char[TextSize];
|
||||
memcpy(TextData, RawData + SectHeader[SectIdx].sh_offset, TextSize);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (SectIdx >= NumSects) {
|
||||
outs() << "ASM Error: couldn't locate .text section\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Assemble shader, fill member vars, and copy to output buffer
|
||||
*
|
||||
* @param AssemblySource Shader source represented as a raw C string
|
||||
* @param OutBuf Raw instruction stream output buffer
|
||||
* @param BufSize Size of OutBuf (defaults to PAGE_SIZE)
|
||||
* @param Gfxv Optional overload to temporarily set target ASIC
|
||||
* @return Value of RunAssemble() (0 on success)
|
||||
*/
|
||||
int Assembler::RunAssembleBuf(const char* const AssemblySource, char* OutBuf,
|
||||
const size_t BufSize) {
|
||||
int ret = RunAssemble(AssemblySource);
|
||||
return ret ? ret : CopyInstrStream(OutBuf, BufSize);
|
||||
}
|
||||
int Assembler::RunAssembleBuf(const char* const AssemblySource, char* OutBuf,
|
||||
const size_t BufSize, const uint32_t Gfxv) {
|
||||
const char* defaultMCPU = GetTargetAsic();
|
||||
SetTargetAsic(Gfxv);
|
||||
int ret = RunAssemble(AssemblySource);
|
||||
strncpy(MCPU, defaultMCPU, ASM_MCPU_LEN);
|
||||
return ret ? ret : CopyInstrStream(OutBuf, BufSize);
|
||||
}
|
||||
|
||||
/**
|
||||
* Assemble shader and fill member vars
|
||||
*
|
||||
* @param AssemblySource Shader source represented as a raw C string
|
||||
* @return 0 on success
|
||||
*/
|
||||
int Assembler::RunAssemble(const char* const AssemblySource) {
|
||||
// Ensure target ASIC has been set
|
||||
if (!*MCPU) {
|
||||
outs() << "ASM Error: target asic is uninitialized\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
// Delete TextData for any previous runs
|
||||
FlushText();
|
||||
|
||||
#if 0
|
||||
outs() << "ASM Info: running assembly for target: " << MCPU << "\n";
|
||||
outs() << "ASM Info: source:\n";
|
||||
outs() << AssemblySource << "\n";
|
||||
#endif
|
||||
|
||||
// Initialize MCOptions and target triple
|
||||
const MCTargetOptions MCOptions;
|
||||
Triple TheTriple;
|
||||
|
||||
const Target* TheTarget =
|
||||
TargetRegistry::lookupTarget(ArchName, TheTriple, Error);
|
||||
if (!TheTarget) {
|
||||
outs() << Error;
|
||||
return -1;
|
||||
}
|
||||
|
||||
TheTriple.setArchName(ArchName);
|
||||
TheTriple.setVendorName(VendorName);
|
||||
TheTriple.setOSName(OSName);
|
||||
|
||||
TripleName = TheTriple.getTriple();
|
||||
TheTriple.setTriple(Triple::normalize(TripleName));
|
||||
|
||||
// Create MemoryBuffer for assembly source
|
||||
StringRef AssemblyRef(AssemblySource);
|
||||
std::unique_ptr<MemoryBuffer> BufferPtr =
|
||||
MemoryBuffer::getMemBuffer(AssemblyRef, "", false);
|
||||
if (!BufferPtr->getBufferSize()) {
|
||||
outs() << "ASM Error: assembly source is empty\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
// Instantiate SrcMgr and transfer BufferPtr ownership
|
||||
SourceMgr SrcMgr;
|
||||
SrcMgr.AddNewSourceBuffer(std::move(BufferPtr), SMLoc());
|
||||
|
||||
// Initialize MC interfaces and base class objects
|
||||
std::unique_ptr<const MCRegisterInfo> MRI(
|
||||
TheTarget->createMCRegInfo(TripleName));
|
||||
if (!MRI) {
|
||||
outs() << "ASM Error: no register info for target " << MCPU << "\n";
|
||||
return -1;
|
||||
}
|
||||
#if LLVM_VERSION_MAJOR > 9
|
||||
std::unique_ptr<const MCAsmInfo> MAI(
|
||||
TheTarget->createMCAsmInfo(*MRI, TripleName, MCOptions));
|
||||
#else
|
||||
std::unique_ptr<const MCAsmInfo> MAI(
|
||||
TheTarget->createMCAsmInfo(*MRI, TripleName));
|
||||
#endif
|
||||
if (!MAI) {
|
||||
outs() << "ASM Error: no assembly info for target " << MCPU << "\n";
|
||||
return -1;
|
||||
}
|
||||
std::unique_ptr<MCInstrInfo> MCII(
|
||||
TheTarget->createMCInstrInfo());
|
||||
if (!MCII) {
|
||||
outs() << "ASM Error: no instruction info for target " << MCPU << "\n";
|
||||
return -1;
|
||||
}
|
||||
std::unique_ptr<MCSubtargetInfo> STI(
|
||||
TheTarget->createMCSubtargetInfo(TripleName, MCPU, std::string()));
|
||||
if (!STI || !STI->isCPUStringValid(MCPU)) {
|
||||
outs() << "ASM Error: no subtarget info for target " << MCPU << "\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
// Set up the MCContext for creating symbols and MCExpr's
|
||||
#if LLVM_VERSION_MAJOR > 12
|
||||
MCContext Ctx(TheTriple, MAI.get(), MRI.get(), STI.get(), &SrcMgr, &MCOptions);
|
||||
#else
|
||||
MCObjectFileInfo MOFI;
|
||||
MCContext Ctx(MAI.get(), MRI.get(), &MOFI, &SrcMgr, &MCOptions);
|
||||
MOFI.InitMCObjectFileInfo(TheTriple, true, Ctx);
|
||||
#endif
|
||||
|
||||
// Finalize setup for output object code stream
|
||||
std::string Data;
|
||||
std::unique_ptr<raw_string_ostream> DataStream(std::make_unique<raw_string_ostream>(Data));
|
||||
std::unique_ptr<buffer_ostream> BOS(std::make_unique<buffer_ostream>(*DataStream));
|
||||
raw_pwrite_stream* OS = BOS.get();
|
||||
|
||||
#if LLVM_VERSION_MAJOR > 14
|
||||
MCCodeEmitter* CE = TheTarget->createMCCodeEmitter(*MCII, Ctx);
|
||||
#else
|
||||
MCCodeEmitter* CE = TheTarget->createMCCodeEmitter(*MCII, *MRI, Ctx);
|
||||
#endif
|
||||
MCAsmBackend* MAB = TheTarget->createMCAsmBackend(*STI, *MRI, MCOptions);
|
||||
|
||||
if (!MAB) {
|
||||
outs() << "ASM Error: Unable to create MCA Backend\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
#if LLVM_VERSION_MAJOR > 20
|
||||
std::unique_ptr<MCStreamer> Streamer(TheTarget->createMCObjectStreamer(
|
||||
TheTriple, Ctx,
|
||||
std::unique_ptr<MCAsmBackend>(MAB), MAB->createObjectWriter(*OS),
|
||||
std::unique_ptr<MCCodeEmitter>(CE), *STI));
|
||||
#else
|
||||
std::unique_ptr<MCStreamer> Streamer(TheTarget->createMCObjectStreamer(
|
||||
TheTriple, Ctx,
|
||||
std::unique_ptr<MCAsmBackend>(MAB), MAB->createObjectWriter(*OS),
|
||||
std::unique_ptr<MCCodeEmitter>(CE), *STI, MCOptions.MCRelaxAll,
|
||||
MCOptions.MCIncrementalLinkerCompatible, /*DWARFMustBeAtTheEnd*/ false));
|
||||
#endif
|
||||
|
||||
std::unique_ptr<MCAsmParser> Parser(
|
||||
createMCAsmParser(SrcMgr, Ctx, *Streamer, *MAI));
|
||||
|
||||
// Set parser to target parser and run
|
||||
std::unique_ptr<MCTargetAsmParser> TAP(
|
||||
TheTarget->createMCAsmParser(*STI, *Parser, *MCII, MCOptions));
|
||||
if (!TAP) {
|
||||
outs() << "ASM Error: no assembly parsing support for target " << MCPU << "\n";
|
||||
return -1;
|
||||
}
|
||||
Parser->setTargetParser(*TAP);
|
||||
|
||||
if (Parser->Run(true)) {
|
||||
outs() << "ASM Error: assembly parser failed\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
BOS.reset();
|
||||
DataStream->flush();
|
||||
|
||||
int ret = ExtractELFText(Data.data());
|
||||
if (ret < 0 || !TextData) {
|
||||
outs() << "ASM Error: .text extraction failed\n";
|
||||
return ret;
|
||||
}
|
||||
|
||||
#if 0
|
||||
PrintELFHex(Data);
|
||||
PrintTextHex();
|
||||
#endif
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,92 @@
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
//
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2022, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
// AMD Research and AMD HSA Software Development
|
||||
//
|
||||
// Advanced Micro Devices, Inc.
|
||||
//
|
||||
// www.amd.com
|
||||
//
|
||||
// Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
// of this software and associated documentation files (the "Software"), to
|
||||
// deal with the Software without restriction, including without limitation
|
||||
// the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
// and/or sell copies of the Software, and to permit persons to whom the
|
||||
// Software is furnished to do so, subject to the following conditions:
|
||||
//
|
||||
// - Redistributions of source code must retain the above copyright notice,
|
||||
// this list of conditions and the following disclaimers.
|
||||
// - Redistributions in binary form must reproduce the above copyright
|
||||
// notice, this list of conditions and the following disclaimers in
|
||||
// the documentation and/or other materials provided with the distribution.
|
||||
// - Neither the names of Advanced Micro Devices, Inc,
|
||||
// nor the names of its contributors may be used to endorse or promote
|
||||
// products derived from this Software without specific prior written
|
||||
// permission.
|
||||
//
|
||||
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
// THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
// OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
// ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
// DEALINGS WITH THE SOFTWARE.
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#ifndef _ASSEMBLE_H_
|
||||
#define _ASSEMBLE_H_
|
||||
|
||||
#include "OSWrapper.hpp"
|
||||
|
||||
#define ASM_MCPU_LEN 16
|
||||
|
||||
/* initialize LLVM targets and assembly printers/parsers */
|
||||
void Init_LLVM();
|
||||
/* shutdown LLVM */
|
||||
void Shutdown_LLVM();
|
||||
|
||||
class Assembler {
|
||||
private:
|
||||
const char* ArchName = "amdgcn";
|
||||
const char* VendorName = "amd";
|
||||
const char* OSName = "amdhsa";
|
||||
char MCPU[ASM_MCPU_LEN];
|
||||
|
||||
std::string TripleName;
|
||||
std::string Error;
|
||||
|
||||
char* TextData;
|
||||
size_t TextSize;
|
||||
|
||||
void SetTargetAsic(const uint32_t Gfxv);
|
||||
|
||||
void FlushText();
|
||||
void PrintELFHex(const std::string Data);
|
||||
int ExtractELFText(const char* RawData);
|
||||
|
||||
public:
|
||||
Assembler(const uint32_t Gfxv);
|
||||
~Assembler();
|
||||
|
||||
void PrintTextHex();
|
||||
const char* GetTargetAsic();
|
||||
|
||||
const char* GetInstrStream();
|
||||
const size_t GetInstrStreamSize();
|
||||
int CopyInstrStream(char* OutBuf, const size_t BufSize = PAGE_SIZE);
|
||||
|
||||
int RunAssemble(const char* const AssemblySource);
|
||||
int RunAssembleBuf(const char* const AssemblySource, char* OutBuf,
|
||||
const size_t BufSize = PAGE_SIZE);
|
||||
int RunAssembleBuf(const char* const AssemblySource, char* OutBuf,
|
||||
const size_t BufSize, const uint32_t Gfxv);
|
||||
};
|
||||
|
||||
#endif // _ASSEMBLE_H_
|
||||
@@ -0,0 +1,311 @@
|
||||
/*
|
||||
* Copyright (C) 2023 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#include "BaseDebug.hpp"
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <sys/types.h>
|
||||
#include <sys/stat.h>
|
||||
#include <hsakmt/linux/kfd_ioctl.h>
|
||||
#include <fcntl.h>
|
||||
#include "unistd.h"
|
||||
|
||||
BaseDebug::BaseDebug(void) {
|
||||
}
|
||||
|
||||
BaseDebug::~BaseDebug(void) {
|
||||
/*
|
||||
* If the process is still attached, close and destroy the polling file
|
||||
* descriptor. Note that on process termination, the KFD automatically
|
||||
* disables processes that are still runtime enabled and debug enabled
|
||||
* so we don't do it here.
|
||||
*/
|
||||
if (m_Pid) {
|
||||
close(m_Fd.fd);
|
||||
unlink(m_Fd_Name);
|
||||
}
|
||||
}
|
||||
|
||||
// Creates temp file descriptor and debug attaches.
|
||||
HSAKMT_STATUS BaseDebug::Attach(struct kfd_runtime_info *rInfo,
|
||||
int rInfoSize,
|
||||
unsigned int pid,
|
||||
uint64_t exceptionEnable) {
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
char fd_name[32];
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
mkfifo(m_Fd_Name, 0666);
|
||||
m_Fd.fd = open(m_Fd_Name, O_CLOEXEC | O_NONBLOCK | O_RDWR);
|
||||
m_Fd.events = POLLIN | POLLRDNORM;
|
||||
|
||||
args.pid = pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_ENABLE;
|
||||
args.enable.rinfo_ptr = (uint64_t)rInfo;
|
||||
args.enable.rinfo_size = rInfoSize;
|
||||
args.enable.dbg_fd = m_Fd.fd;
|
||||
args.enable.exception_mask = exceptionEnable;
|
||||
|
||||
if (hsaKmtDebugTrapIoctl(&args, NULL, NULL)) {
|
||||
close(m_Fd.fd);
|
||||
unlink(m_Fd_Name);
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
m_Pid = pid;
|
||||
|
||||
return HSAKMT_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
|
||||
void BaseDebug::Detach(void) {
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_DISABLE;
|
||||
|
||||
hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
|
||||
close(m_Fd.fd);
|
||||
unlink(m_Fd_Name);
|
||||
|
||||
m_Pid = 0;
|
||||
m_Fd.fd = 0;
|
||||
m_Fd.events = 0;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::SendRuntimeEvent(uint64_t exceptions, int gpuId, int queueId)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_SEND_RUNTIME_EVENT;
|
||||
args.send_runtime_event.exception_mask = exceptions;
|
||||
args.send_runtime_event.gpu_id = gpuId;
|
||||
args.send_runtime_event.queue_id = queueId;
|
||||
|
||||
return hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::QueryDebugEvent(uint64_t *exceptions,
|
||||
uint32_t *gpuId, uint32_t *queueId,
|
||||
int timeoutMsec)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
HSAKMT_STATUS result;
|
||||
int r = poll(&m_Fd, 1, timeoutMsec);
|
||||
|
||||
if (r > 0) {
|
||||
char tmp[r];
|
||||
|
||||
read(m_Fd.fd, tmp, sizeof(tmp));
|
||||
} else {
|
||||
return HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_QUERY_DEBUG_EVENT;
|
||||
args.query_debug_event.exception_mask = *exceptions;
|
||||
|
||||
result = hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
|
||||
*exceptions = args.query_debug_event.exception_mask;
|
||||
|
||||
if (gpuId)
|
||||
*gpuId = args.query_debug_event.gpu_id;
|
||||
|
||||
if (queueId)
|
||||
*queueId = args.query_debug_event.queue_id;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
void BaseDebug::SetExceptionsEnabled(uint64_t exceptions)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_SET_EXCEPTIONS_ENABLED;
|
||||
args.set_exceptions_enabled.exception_mask = exceptions;
|
||||
|
||||
hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::SuspendQueues(unsigned int *numQueues,
|
||||
HSA_QUEUEID *queues,
|
||||
uint32_t *queueIds,
|
||||
uint64_t exceptionsToClear)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_SUSPEND_QUEUES;
|
||||
args.suspend_queues.num_queues = *numQueues;
|
||||
args.suspend_queues.queue_array_ptr = (uint64_t)queueIds;
|
||||
args.suspend_queues.exception_mask = exceptionsToClear;
|
||||
|
||||
return hsaKmtDebugTrapIoctl(&args, queues, (HSAuint64 *)numQueues);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::ResumeQueues(unsigned int *numQueues,
|
||||
HSA_QUEUEID *queues,
|
||||
uint32_t *queueIds)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_RESUME_QUEUES;
|
||||
args.resume_queues.num_queues = *numQueues;
|
||||
args.resume_queues.queue_array_ptr = (uint64_t)queueIds;
|
||||
|
||||
return hsaKmtDebugTrapIoctl(&args, queues, (HSAuint64 *)numQueues);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::QueueSnapshot(uint64_t exceptionsToClear,
|
||||
uint64_t snapshotBufAddr,
|
||||
uint32_t *numSnapshots)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
HSAKMT_STATUS result;
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_GET_QUEUE_SNAPSHOT;
|
||||
args.queue_snapshot.exception_mask = exceptionsToClear;
|
||||
args.queue_snapshot.snapshot_buf_ptr = snapshotBufAddr;
|
||||
args.queue_snapshot.num_queues = *numSnapshots;
|
||||
args.queue_snapshot.entry_size = sizeof(struct kfd_queue_snapshot_entry);
|
||||
|
||||
result = hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
|
||||
*numSnapshots = args.queue_snapshot.num_queues;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::DeviceSnapshot(uint64_t exceptionsToClear,
|
||||
uint64_t snapshotBufAddr,
|
||||
uint32_t *numSnapshots)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
HSAKMT_STATUS result;
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_GET_DEVICE_SNAPSHOT;
|
||||
args.device_snapshot.exception_mask = exceptionsToClear;
|
||||
args.device_snapshot.snapshot_buf_ptr = snapshotBufAddr;
|
||||
args.device_snapshot.num_devices = *numSnapshots;
|
||||
args.device_snapshot.entry_size = sizeof(struct kfd_dbg_device_info_entry);
|
||||
|
||||
result = hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
|
||||
*numSnapshots = args.device_snapshot.num_devices;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::SetWaveLaunchOverride(int mode,
|
||||
uint32_t *enableMask,
|
||||
uint32_t *supportMask)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {0};
|
||||
HSAKMT_STATUS Result;
|
||||
|
||||
memset(&args, 0x00, sizeof(args));
|
||||
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_SET_WAVE_LAUNCH_OVERRIDE;
|
||||
args.launch_override.override_mode = mode;
|
||||
args.launch_override.enable_mask = *enableMask;
|
||||
args.launch_override.support_request_mask = *supportMask;
|
||||
|
||||
Result = hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
|
||||
*enableMask = args.launch_override.enable_mask;
|
||||
*supportMask = args.launch_override.support_request_mask;
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::SetAddressWatch(uint64_t address,
|
||||
int mode,
|
||||
uint64_t mask,
|
||||
uint32_t gpuId,
|
||||
uint32_t *id)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {};
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_SET_NODE_ADDRESS_WATCH;
|
||||
args.set_node_address_watch.address = address;
|
||||
args.set_node_address_watch.mode = mode;
|
||||
args.set_node_address_watch.mask = mask;
|
||||
args.set_node_address_watch.gpu_id = gpuId;
|
||||
|
||||
HSAKMT_STATUS result = hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
|
||||
*id = args.set_node_address_watch.id;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::ClearAddressWatch(uint32_t gpuId,
|
||||
uint32_t id)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {};
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_CLEAR_NODE_ADDRESS_WATCH;
|
||||
args.clear_node_address_watch.gpu_id = gpuId;
|
||||
args.clear_node_address_watch.id = id;
|
||||
|
||||
return hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseDebug::SetFlags(uint32_t *flags)
|
||||
{
|
||||
struct kfd_ioctl_dbg_trap_args args = {};
|
||||
args.pid = m_Pid;
|
||||
args.op = KFD_IOC_DBG_TRAP_SET_FLAGS;
|
||||
args.set_flags.flags = *flags;
|
||||
|
||||
HSAKMT_STATUS result = hsaKmtDebugTrapIoctl(&args, NULL, NULL);
|
||||
|
||||
*flags = args.set_flags.flags;
|
||||
|
||||
return result;
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
/*
|
||||
* Copyright (C) 2023 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __KFD_BASE_DEBUG__H__
|
||||
#define __KFD_BASE_DEBUG__H__
|
||||
|
||||
#include "hsakmt/hsakmt.h"
|
||||
#include <poll.h>
|
||||
#include <stdlib.h>
|
||||
|
||||
// @class BaseDebug
|
||||
class BaseDebug {
|
||||
public:
|
||||
BaseDebug(void);
|
||||
virtual ~BaseDebug(void);
|
||||
|
||||
HSAKMT_STATUS Attach(struct kfd_runtime_info *rInfo,
|
||||
int rInfoSize,
|
||||
unsigned int pid,
|
||||
uint64_t exceptionEnable);
|
||||
|
||||
void Detach(void);
|
||||
HSAKMT_STATUS SendRuntimeEvent(uint64_t exceptions, int gpuId, int queueId);
|
||||
HSAKMT_STATUS QueryDebugEvent(uint64_t *exceptions,
|
||||
uint32_t *gpuId, uint32_t *queueId,
|
||||
int timeoutMsec);
|
||||
void SetExceptionsEnabled(uint64_t exceptions);
|
||||
HSAKMT_STATUS SuspendQueues(unsigned int *numQueues, HSA_QUEUEID *queues, uint32_t *queueIds,
|
||||
uint64_t exceptionsToClear);
|
||||
HSAKMT_STATUS ResumeQueues(unsigned int *numQueues, HSA_QUEUEID *queues, uint32_t *queueIds);
|
||||
HSAKMT_STATUS QueueSnapshot(uint64_t exceptionsToClear, uint64_t snapshotBufAddr,
|
||||
uint32_t *numSnapshots);
|
||||
HSAKMT_STATUS DeviceSnapshot(uint64_t exceptionsToClear, uint64_t snapshotBuffAddr,
|
||||
uint32_t *numSnapshots);
|
||||
HSAKMT_STATUS SetWaveLaunchOverride(int mode, uint32_t *enableMask, uint32_t *supportMask);
|
||||
HSAKMT_STATUS SetAddressWatch(uint64_t address, int mode, uint64_t mask, uint32_t gpuId, uint32_t *id);
|
||||
HSAKMT_STATUS ClearAddressWatch(uint32_t gpuId, uint32_t id);
|
||||
HSAKMT_STATUS SetFlags(uint32_t *flags);
|
||||
|
||||
private:
|
||||
unsigned int m_Pid;
|
||||
struct pollfd m_Fd;
|
||||
const char *m_Fd_Name = "/tmp/dbg_fifo";
|
||||
};
|
||||
|
||||
#endif // __KFD_BASE_DEBUG__H__
|
||||
@@ -0,0 +1,60 @@
|
||||
/*
|
||||
* Copyright (C) 2017-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#include "BasePacket.hpp"
|
||||
#include "KFDTestUtil.hpp"
|
||||
#include "KFDBaseComponentTest.hpp"
|
||||
|
||||
BasePacket::BasePacket(void): m_packetAllocation(NULL) {
|
||||
m_FamilyId = g_baseTest->GetFamilyIdFromDefaultNode();
|
||||
}
|
||||
|
||||
BasePacket::~BasePacket(void) {
|
||||
if (m_packetAllocation)
|
||||
free(m_packetAllocation);
|
||||
}
|
||||
|
||||
void BasePacket::Dump() const {
|
||||
unsigned int size = SizeInDWords();
|
||||
const HSAuint32 *packet = (const HSAuint32 *)GetPacket();
|
||||
std::ostream &log = LOG();
|
||||
unsigned int i;
|
||||
|
||||
log << "Packet dump:" << std::hex;
|
||||
for (i = 0; i < size; i++)
|
||||
log << " " << std::setw(8) << std::setfill('0') << packet[i];
|
||||
log << std::endl;
|
||||
}
|
||||
|
||||
void *BasePacket::AllocPacket(void) {
|
||||
unsigned int size = SizeInBytes();
|
||||
|
||||
EXPECT_NE(0, size);
|
||||
if (!size)
|
||||
return NULL;
|
||||
|
||||
m_packetAllocation = calloc(1, size);
|
||||
EXPECT_NOTNULL(m_packetAllocation);
|
||||
|
||||
return m_packetAllocation;
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __KFD_BASE_PACKET__H__
|
||||
#define __KFD_BASE_PACKET__H__
|
||||
|
||||
/**
|
||||
* All packets profiles must be defined here
|
||||
* Every type defined here has sub-types
|
||||
*/
|
||||
enum PACKETTYPE {
|
||||
PACKETTYPE_PM4,
|
||||
PACKETTYPE_SDMA,
|
||||
PACKETTYPE_AQL
|
||||
};
|
||||
|
||||
// @class BasePacket
|
||||
class BasePacket {
|
||||
public:
|
||||
BasePacket(void);
|
||||
virtual ~BasePacket(void);
|
||||
|
||||
// @returns Packet type
|
||||
virtual PACKETTYPE PacketType() const = 0;
|
||||
// @returns Pointer to the packet
|
||||
virtual const void *GetPacket() const = 0;
|
||||
// @returns Packet size in bytes
|
||||
virtual unsigned int SizeInBytes() const = 0;
|
||||
// @returns Packet size in dwordS
|
||||
unsigned int SizeInDWords() const { return SizeInBytes()/sizeof(unsigned int); }
|
||||
|
||||
void Dump() const;
|
||||
|
||||
protected:
|
||||
unsigned int m_FamilyId;
|
||||
void *m_packetAllocation;
|
||||
|
||||
void *AllocPacket(void);
|
||||
};
|
||||
|
||||
#endif // __KFD_BASE_PACKET__H__
|
||||
@@ -0,0 +1,217 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#include "BaseQueue.hpp"
|
||||
#include "SDMAQueue.hpp"
|
||||
#include "PM4Queue.hpp"
|
||||
#include "AqlQueue.hpp"
|
||||
#include "hsakmt/hsakmt.h"
|
||||
#include "KFDBaseComponentTest.hpp"
|
||||
|
||||
BaseQueue::BaseQueue()
|
||||
:m_QueueBuf(NULL),
|
||||
m_SkipWaitConsumption(true) {
|
||||
}
|
||||
|
||||
BaseQueue::~BaseQueue(void) {
|
||||
Destroy();
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseQueue::Create(unsigned int NodeId, unsigned int size, HSAuint64 *pointers) {
|
||||
HSAKMT_STATUS status;
|
||||
HSA_QUEUE_TYPE type = GetQueueType();
|
||||
|
||||
if (m_QueueBuf != NULL) {
|
||||
// Queue already exists, one queue per object
|
||||
Destroy();
|
||||
}
|
||||
|
||||
memset(&m_Resources, 0, sizeof(m_Resources));
|
||||
|
||||
m_QueueBuf = new HsaMemoryBuffer(size, NodeId, true/*zero*/, false/*local*/, true/*exec*/,
|
||||
/*isScratch */ false, /* isReadOnly */false, /* isUncached */true);
|
||||
|
||||
if (type == HSA_QUEUE_COMPUTE_AQL) {
|
||||
m_Resources.Queue_read_ptr_aql = &pointers[0];
|
||||
m_Resources.Queue_write_ptr_aql = &pointers[1];
|
||||
}
|
||||
|
||||
if (type == HSA_QUEUE_SDMA_BY_ENG_ID)
|
||||
status = hsaKmtCreateQueueExt(NodeId,
|
||||
type,
|
||||
DEFAULT_QUEUE_PERCENTAGE,
|
||||
DEFAULT_PRIORITY,
|
||||
m_SdmaEngineId,
|
||||
m_QueueBuf->As<unsigned int*>(),
|
||||
m_QueueBuf->Size(),
|
||||
NULL,
|
||||
&m_Resources);
|
||||
else
|
||||
status = hsaKmtCreateQueue(NodeId,
|
||||
type,
|
||||
DEFAULT_QUEUE_PERCENTAGE,
|
||||
DEFAULT_PRIORITY,
|
||||
m_QueueBuf->As<unsigned int*>(),
|
||||
m_QueueBuf->Size(),
|
||||
NULL,
|
||||
&m_Resources);
|
||||
|
||||
if (status != HSAKMT_STATUS_SUCCESS) {
|
||||
return status;
|
||||
}
|
||||
|
||||
if (m_Resources.Queue_read_ptr == NULL) {
|
||||
WARN() << "CreateQueue: read pointer value should be 0" << std::endl;
|
||||
status = HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
if (m_Resources.Queue_write_ptr == NULL) {
|
||||
WARN() << "CreateQueue: write pointer value should be 0" << std::endl;
|
||||
status = HSAKMT_STATUS_ERROR;
|
||||
}
|
||||
|
||||
// Needs to match the queue write ptr
|
||||
m_pendingWptr = 0;
|
||||
m_pendingWptr64 = 0;
|
||||
m_Node = NodeId;
|
||||
m_FamilyId = g_baseTest->GetFamilyIdFromNodeId(NodeId);
|
||||
return status;
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseQueue::Update(unsigned int percent, HSA_QUEUE_PRIORITY priority, bool nullifyBuffer) {
|
||||
void* pNewBuffer = (nullifyBuffer ? NULL : m_QueueBuf->As<void*>());
|
||||
HSAuint64 newSize = (nullifyBuffer ? 0 : m_QueueBuf->Size());
|
||||
|
||||
return hsaKmtUpdateQueue(m_Resources.QueueId, percent, priority, pNewBuffer, newSize, NULL);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseQueue::SetCUMask(unsigned int *mask, unsigned int mask_count) {
|
||||
return hsaKmtSetQueueCUMask(m_Resources.QueueId, mask_count, mask);
|
||||
}
|
||||
|
||||
HSAKMT_STATUS BaseQueue::Destroy() {
|
||||
HSAKMT_STATUS status = HSAKMT_STATUS_SUCCESS;
|
||||
|
||||
if (m_QueueBuf != NULL) {
|
||||
status = hsaKmtDestroyQueue(m_Resources.QueueId);
|
||||
|
||||
if (status == HSAKMT_STATUS_SUCCESS) {
|
||||
delete m_QueueBuf;
|
||||
m_QueueBuf = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
return status;
|
||||
}
|
||||
|
||||
void BaseQueue::PlaceAndSubmitPacket(const BasePacket &packet) {
|
||||
PlacePacket(packet);
|
||||
SubmitPacket();
|
||||
}
|
||||
|
||||
void BaseQueue::Wait4PacketConsumption(HsaEvent *event, unsigned int timeOut) {
|
||||
ASSERT_TRUE(!event) << "Not supported!" << std::endl;
|
||||
ASSERT_TRUE(WaitOnValue(m_Resources.Queue_read_ptr, RptrWhenConsumed(), timeOut));
|
||||
}
|
||||
|
||||
bool BaseQueue::AllPacketsSubmitted() {
|
||||
return Wptr() == Rptr();
|
||||
}
|
||||
|
||||
void BaseQueue::PlacePacket(const BasePacket &packet) {
|
||||
ASSERT_EQ(packet.PacketType(), PacketTypeSupported())
|
||||
<< "Cannot add a packet since packet type doesn't match queue";
|
||||
|
||||
unsigned int readPtr = Rptr();
|
||||
unsigned int writePtr = m_pendingWptr;
|
||||
HSAuint64 writePtr64 = m_pendingWptr64;
|
||||
|
||||
unsigned int packetSizeInDwords = packet.SizeInDWords();
|
||||
unsigned int dwordsRequired = packetSizeInDwords;
|
||||
unsigned int queueSizeInDWord = m_QueueBuf->Size() / sizeof(uint32_t);
|
||||
|
||||
if (writePtr + packetSizeInDwords > queueSizeInDWord) {
|
||||
// Wraparound expected. We need enough room to also place NOPs to avoid crossing the buffer end.
|
||||
dwordsRequired += queueSizeInDWord - writePtr;
|
||||
}
|
||||
|
||||
unsigned int dwordsAvailable = (readPtr - 1 - writePtr + queueSizeInDWord) % queueSizeInDWord;
|
||||
ASSERT_GE(dwordsAvailable, dwordsRequired) << "Cannot add a packet, buffer overrun";
|
||||
|
||||
ASSERT_GE(queueSizeInDWord, packetSizeInDwords) << "Cannot add a packet, packet size too large";
|
||||
|
||||
if (writePtr + packetSizeInDwords >= queueSizeInDWord) {
|
||||
// Wraparound
|
||||
while (writePtr + packetSizeInDwords > queueSizeInDWord) {
|
||||
m_QueueBuf->As<unsigned int *>()[writePtr] = CMD_NOP;
|
||||
writePtr = (writePtr + 1) % queueSizeInDWord;
|
||||
writePtr64++;
|
||||
}
|
||||
|
||||
// Not updating Wptr since we might want to place the packet without submission
|
||||
m_pendingWptr = (writePtr % queueSizeInDWord);
|
||||
m_pendingWptr64 = writePtr64;
|
||||
}
|
||||
|
||||
memcpy(m_pendingWptr + m_QueueBuf->As<unsigned int*>(), packet.GetPacket(), packetSizeInDwords * 4);
|
||||
|
||||
m_pendingWptr = (m_pendingWptr + packetSizeInDwords) % queueSizeInDWord;
|
||||
m_pendingWptr64 += packetSizeInDwords;
|
||||
}
|
||||
|
||||
BaseQueue* QueueArray::GetQueue(unsigned int Node) {
|
||||
// If a queue exists for that node then return, else create one
|
||||
for (unsigned int i = 0; i < m_QueueList.size(); i++) {
|
||||
if (Node == m_QueueList.at(i)->GetNodeId())
|
||||
return m_QueueList.at(i);
|
||||
}
|
||||
|
||||
BaseQueue *pQueue = NULL;
|
||||
|
||||
switch (m_QueueType) {
|
||||
case HSA_QUEUE_COMPUTE:
|
||||
pQueue = new PM4Queue();
|
||||
break;
|
||||
case HSA_QUEUE_SDMA:
|
||||
pQueue = new SDMAQueue();
|
||||
break;
|
||||
case HSA_QUEUE_COMPUTE_AQL:
|
||||
pQueue = new AqlQueue();
|
||||
break;
|
||||
default:
|
||||
return NULL;
|
||||
}
|
||||
|
||||
if (pQueue) {
|
||||
pQueue->Create(Node);
|
||||
m_QueueList.push_back(pQueue);
|
||||
}
|
||||
return pQueue;
|
||||
}
|
||||
|
||||
void QueueArray::Destroy() {
|
||||
for (unsigned int i = 0; i < m_QueueList.size(); i++)
|
||||
delete m_QueueList.at(i);
|
||||
|
||||
m_QueueList.clear();
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __KFD_BASE_QUEUE__H__
|
||||
#define __KFD_BASE_QUEUE__H__
|
||||
|
||||
#include <vector>
|
||||
#include "KFDTestUtil.hpp"
|
||||
#include "BasePacket.hpp"
|
||||
|
||||
// @class BasePacket
|
||||
class BaseQueue {
|
||||
public:
|
||||
static const unsigned int DEFAULT_QUEUE_SIZE = PAGE_SIZE;
|
||||
static const HSA_QUEUE_PRIORITY DEFAULT_PRIORITY = HSA_QUEUE_PRIORITY_NORMAL;
|
||||
static const unsigned int DEFAULT_QUEUE_PERCENTAGE = 100;
|
||||
static const unsigned int ZERO_QUEUE_PERCENTAGE = 0;
|
||||
static const unsigned int FLUSH_GPU_CACHES_TO = 1000;
|
||||
|
||||
BaseQueue(void);
|
||||
virtual ~BaseQueue(void);
|
||||
|
||||
/** Create the queue.
|
||||
* @see hsaKmtCreateQueue
|
||||
* @param pointers is used only for creating AQL queues. Otherwise it is omitted.
|
||||
*/
|
||||
virtual HSAKMT_STATUS Create(unsigned int NodeId, unsigned int size = DEFAULT_QUEUE_SIZE,
|
||||
HSAuint64 *pointers = NULL);
|
||||
/** Update the queue.
|
||||
* @see hsaKmtUpdateQueue
|
||||
* @param percent New queue percentage
|
||||
* @param priority New queue priority
|
||||
* @param nullifyBuffer
|
||||
* If 'true', set the new buffer address to NULL and the size to 0. Otherwise
|
||||
* don't change the queue buffer address/size.
|
||||
*/
|
||||
virtual HSAKMT_STATUS Update(unsigned int percent, HSA_QUEUE_PRIORITY priority, bool nullifyBuffer);
|
||||
virtual HSAKMT_STATUS SetCUMask(unsigned int *mask, unsigned int mask_count);
|
||||
/** Destroy the queue.
|
||||
* @see hsaKmtDestroyQueue
|
||||
*/
|
||||
virtual HSAKMT_STATUS Destroy();
|
||||
/** Wait for all the packets submitted to the queue to be consumed. (i.e. wait until RPTR=WPTR).
|
||||
* Note that all packets being consumed is not the same as all packets being processed.
|
||||
*/
|
||||
virtual void Wait4PacketConsumption(HsaEvent *event = NULL, unsigned int timeOut = g_TestTimeOut);
|
||||
/** @brief Place packet and submit it in one function
|
||||
*/
|
||||
virtual void PlaceAndSubmitPacket(const BasePacket &packet);
|
||||
/** @brief Copy packet to queue and update write pointer
|
||||
*/
|
||||
virtual void PlacePacket(const BasePacket &packet);
|
||||
/** @brief Update queue write pointer and set the queue doorbell to the queue write pointer
|
||||
*/
|
||||
virtual void SubmitPacket() = 0;
|
||||
/** @brief Check if all packets in queue are already processed
|
||||
* Compare queue read and write pointers
|
||||
*/
|
||||
bool AllPacketsSubmitted();
|
||||
|
||||
void SetSkipWaitConsump(int val) { m_SkipWaitConsumption = val; }
|
||||
int GetSkipWaitConsump() { return m_SkipWaitConsumption; }
|
||||
int Size() { return m_QueueBuf->Size(); }
|
||||
|
||||
HsaQueueResource *GetResource() { return &m_Resources; }
|
||||
unsigned int GetPendingWptr() { return m_pendingWptr; }
|
||||
HSAuint64 GetPendingWptr64() { return m_pendingWptr64; }
|
||||
virtual _HSA_QUEUE_TYPE GetQueueType() = 0;
|
||||
unsigned int GetNodeId() { return m_Node; }
|
||||
unsigned int GetFamilyId() { return m_FamilyId; }
|
||||
int GetSDMAEngineId() { return m_SdmaEngineId; }
|
||||
|
||||
protected:
|
||||
static const unsigned int CMD_NOP_TYPE_2 = 0x80000000;
|
||||
static const unsigned int CMD_NOP_TYPE_3 = 0xFFFF1002;
|
||||
|
||||
unsigned int CMD_NOP;
|
||||
unsigned int m_pendingWptr;
|
||||
HSAuint64 m_pendingWptr64;
|
||||
HsaQueueResource m_Resources;
|
||||
HsaMemoryBuffer *m_QueueBuf;
|
||||
unsigned int m_Node;
|
||||
unsigned int m_FamilyId;
|
||||
int m_SdmaEngineId;
|
||||
|
||||
// @return Write pointer modulo queue size in dwords
|
||||
virtual unsigned int Wptr() = 0;
|
||||
// @return Read pointer modulo queue size in dwords
|
||||
virtual unsigned int Rptr() = 0;
|
||||
// @return Expected m_Resources.Queue_read_ptr when all packets consumed
|
||||
virtual unsigned int RptrWhenConsumed() = 0;
|
||||
virtual PACKETTYPE PacketTypeSupported() = 0;
|
||||
|
||||
private:
|
||||
// Some tests(such as exception) may not need wait pm4 packet consumption on CZ.
|
||||
int m_SkipWaitConsumption;
|
||||
};
|
||||
|
||||
|
||||
// @class QueueArray
|
||||
// Managed QueueArray for different GPU Nodes
|
||||
class QueueArray {
|
||||
// List of Queues. One for each GPU
|
||||
std::vector<BaseQueue*> m_QueueList;
|
||||
_HSA_QUEUE_TYPE m_QueueType;
|
||||
|
||||
public:
|
||||
QueueArray(_HSA_QUEUE_TYPE type): m_QueueType(type) {}
|
||||
~QueueArray() {
|
||||
Destroy();
|
||||
}
|
||||
|
||||
BaseQueue* GetQueue(unsigned int Node);
|
||||
void Destroy();
|
||||
};
|
||||
|
||||
#endif // __KFD_BASE_QUEUE__H__
|
||||
@@ -0,0 +1,272 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#include "Dispatch.hpp"
|
||||
|
||||
#include "PM4Packet.hpp"
|
||||
|
||||
#include "asic_reg/gfx_7_2_d.h"
|
||||
#include "asic_reg/gfx_7_2_sh_mask.h"
|
||||
|
||||
#include "KFDBaseComponentTest.hpp"
|
||||
|
||||
#define mmCOMPUTE_PGM_RSRC3 0x2e2d
|
||||
|
||||
Dispatch::Dispatch(const HsaMemoryBuffer& isaBuf, const bool eventAutoReset)
|
||||
:m_IsaBuf(isaBuf), m_IndirectBuf(PACKETTYPE_PM4, PAGE_SIZE / sizeof(unsigned int), isaBuf.Node()),
|
||||
m_DimX(1), m_DimY(1), m_DimZ(1), m_pArg1(NULL), m_pArg2(NULL), m_pEop(NULL), m_ScratchEn(false),
|
||||
m_ComputeTmpringSize(0), m_scratch_base(0ll), m_SpiPriority(0) {
|
||||
HsaEventDescriptor eventDesc;
|
||||
eventDesc.EventType = HSA_EVENTTYPE_SIGNAL;
|
||||
eventDesc.NodeId = isaBuf.Node();
|
||||
eventDesc.SyncVar.SyncVar.UserData = NULL;
|
||||
eventDesc.SyncVar.SyncVarSize = 0;
|
||||
|
||||
hsaKmtCreateEvent(&eventDesc, !eventAutoReset, false, &m_pEop);
|
||||
|
||||
m_FamilyId = g_baseTest->GetFamilyIdFromNodeId(isaBuf.Node());
|
||||
m_NeedCwsrWA = g_baseTest->NeedCwsrWA(isaBuf.Node());
|
||||
}
|
||||
|
||||
Dispatch::~Dispatch() {
|
||||
if (m_pEop != NULL)
|
||||
hsaKmtDestroyEvent(m_pEop);
|
||||
}
|
||||
|
||||
void Dispatch::SetArgs(void* pArg1, void* pArg2) {
|
||||
m_pArg1 = pArg1;
|
||||
m_pArg2 = pArg2;
|
||||
}
|
||||
|
||||
void Dispatch::SetDim(unsigned int x, unsigned int y, unsigned int z) {
|
||||
m_DimX = x;
|
||||
m_DimY = y;
|
||||
m_DimZ = z;
|
||||
}
|
||||
|
||||
void Dispatch::SetScratch(int numWaves, int waveSize, HSAuint64 scratch_base) {
|
||||
m_ComputeTmpringSize = ((waveSize << 12) | (numWaves));
|
||||
m_ScratchEn = true;
|
||||
m_scratch_base = scratch_base;
|
||||
}
|
||||
|
||||
void Dispatch::SetSpiPriority(unsigned int priority) {
|
||||
m_SpiPriority = priority;
|
||||
}
|
||||
|
||||
void Dispatch::SetPriv(bool priv) {
|
||||
m_NeedCwsrWA = priv;
|
||||
}
|
||||
|
||||
void Dispatch::Submit(BaseQueue& queue) {
|
||||
ASSERT_NE(m_pEop, (void*)0);
|
||||
EXPECT_EQ(m_FamilyId, queue.GetFamilyId());
|
||||
|
||||
BuildIb();
|
||||
|
||||
queue.PlaceAndSubmitPacket(PM4IndirectBufPacket(&m_IndirectBuf));
|
||||
|
||||
// Write data to SyncVar for synchronization purpose
|
||||
if (m_pEop->EventData.EventData.SyncVar.SyncVar.UserData != NULL) {
|
||||
queue.PlaceAndSubmitPacket(PM4WriteDataPacket((unsigned int*)m_pEop->
|
||||
EventData.EventData.SyncVar.SyncVar.UserData, m_pEop->EventId));
|
||||
}
|
||||
|
||||
queue.PlaceAndSubmitPacket(PM4ReleaseMemoryPacket(m_FamilyId, false, m_pEop->EventData.HWData2, m_pEop->EventId));
|
||||
|
||||
if (!queue.GetSkipWaitConsump())
|
||||
queue.Wait4PacketConsumption();
|
||||
}
|
||||
|
||||
void Dispatch::Sync(unsigned int timeout) {
|
||||
ASSERT_SUCCESS(hsaKmtWaitOnEvent(m_pEop, timeout));
|
||||
}
|
||||
|
||||
// Returning with status in order to allow actions to be performed before process termination
|
||||
int Dispatch::SyncWithStatus(unsigned int timeout) {
|
||||
int stat;
|
||||
|
||||
return ((stat = hsaKmtWaitOnEvent(m_pEop, timeout)) != HSAKMT_STATUS_SUCCESS);
|
||||
}
|
||||
|
||||
void Dispatch::BuildIb() {
|
||||
HSAuint64 shiftedIsaAddr = m_IsaBuf.As<uint64_t>() >> 8;
|
||||
unsigned int arg0, arg1, arg2, arg3;
|
||||
SplitU64(reinterpret_cast<uint64_t>(m_pArg1), arg0, arg1);
|
||||
SplitU64(reinterpret_cast<uint64_t>(m_pArg2), arg2, arg3);
|
||||
|
||||
// Starts at COMPUTE_START_X
|
||||
const unsigned int COMPUTE_DISPATCH_DIMS_VALUES[] = {
|
||||
0, // START_X
|
||||
0, // START_Y
|
||||
0, // START_Z
|
||||
1, // NUM_THREADS_X - this is actually the number of threads in a thread group
|
||||
1, // NUM_THREADS_Y
|
||||
1, // NUM_THREADS_Z
|
||||
0, // COMPUTE_PIPELINESTAT_ENABLE
|
||||
0, // COMPUTE_PERFCOUNT_ENABLE
|
||||
};
|
||||
|
||||
/*
|
||||
* For some special asics in the list of DEGFX11_12113
|
||||
* COMPUTE_PGM_RSRC needs priv=1 to prevent hardware traps
|
||||
*/
|
||||
const bool priv = m_NeedCwsrWA;
|
||||
|
||||
unsigned int pgmRsrc1 =
|
||||
(0xc0 << COMPUTE_PGM_RSRC1__FLOAT_MODE__SHIFT) |
|
||||
((m_SpiPriority & 3) << COMPUTE_PGM_RSRC1__PRIORITY__SHIFT) |
|
||||
(priv << COMPUTE_PGM_RSRC1__PRIV__SHIFT) |
|
||||
((m_FamilyId < FAMILY_GFX12) ? (0x2 << COMPUTE_PGM_RSRC1__SGPRS__SHIFT) : 0) |
|
||||
(0x4 << COMPUTE_PGM_RSRC1__VGPRS__SHIFT); // 4 * 8 = 32 VGPRs
|
||||
|
||||
unsigned int pgmRsrc2 = 0;
|
||||
pgmRsrc2 |= (m_ScratchEn << COMPUTE_PGM_RSRC2__SCRATCH_EN__SHIFT)
|
||||
& COMPUTE_PGM_RSRC2__SCRATCH_EN_MASK;
|
||||
pgmRsrc2 |= ((m_scratch_base ? 6 : 4) << COMPUTE_PGM_RSRC2__USER_SGPR__SHIFT)
|
||||
& COMPUTE_PGM_RSRC2__USER_SGPR_MASK;
|
||||
|
||||
if (m_FamilyId < FAMILY_GFX12) {
|
||||
pgmRsrc2 |= (1 << COMPUTE_PGM_RSRC2__TRAP_PRESENT__SHIFT)
|
||||
& COMPUTE_PGM_RSRC2__TRAP_PRESENT_MASK;
|
||||
}
|
||||
|
||||
pgmRsrc2 |= (1 << COMPUTE_PGM_RSRC2__TGID_X_EN__SHIFT)
|
||||
& COMPUTE_PGM_RSRC2__TGID_X_EN_MASK;
|
||||
pgmRsrc2 |= (1 << COMPUTE_PGM_RSRC2__TIDIG_COMP_CNT__SHIFT)
|
||||
& COMPUTE_PGM_RSRC2__TIDIG_COMP_CNT_MASK;
|
||||
pgmRsrc2 |= (0 << COMPUTE_PGM_RSRC2__EXCP_EN__SHIFT)
|
||||
& COMPUTE_PGM_RSRC2__EXCP_EN_MASK;
|
||||
pgmRsrc2 |= (1 << COMPUTE_PGM_RSRC2__EXCP_EN_MSB__SHIFT)
|
||||
& COMPUTE_PGM_RSRC2__EXCP_EN_MSB_MASK;
|
||||
|
||||
const unsigned int COMPUTE_PGM_RSRC[] = {
|
||||
pgmRsrc1,
|
||||
pgmRsrc2
|
||||
};
|
||||
|
||||
// Starts at COMPUTE_PGM_LO
|
||||
const unsigned int COMPUTE_PGM_VALUES_GFX8[] = {
|
||||
static_cast<uint32_t>(shiftedIsaAddr), // PGM_LO
|
||||
static_cast<uint32_t>(shiftedIsaAddr >> 32) // PGM_HI
|
||||
| (hsakmt_is_dgpu() ? 0 : (1<<8)) // including PGM_ATC=?
|
||||
};
|
||||
|
||||
// Starts at COMPUTE_PGM_LO
|
||||
const unsigned int COMPUTE_PGM_VALUES_GFX9[] = {
|
||||
static_cast<uint32_t>(shiftedIsaAddr), // PGM_LO
|
||||
static_cast<uint32_t>(shiftedIsaAddr >> 32) // PGM_HI
|
||||
| (hsakmt_is_dgpu() ? 0 : (1<<8)), // including PGM_ATC=?
|
||||
0,
|
||||
0,
|
||||
static_cast<uint32_t>(m_scratch_base >> 8), // compute_dispatch_scratch_base
|
||||
static_cast<uint32_t>(m_scratch_base >> 40)
|
||||
};
|
||||
|
||||
// Starts at COMPUTE_RESOURCE_LIMITS
|
||||
const unsigned int COMPUTE_RESOURCE_LIMITS[] = {
|
||||
0, // COMPUTE_RESOURCE_LIMITS
|
||||
};
|
||||
|
||||
// Starts at COMPUTE_TMPRING_SIZE
|
||||
const unsigned int COMPUTE_TMPRING_SIZE[] = {
|
||||
m_ComputeTmpringSize, // COMPUTE_TMPRING_SIZE
|
||||
};
|
||||
|
||||
// Starts at COMPUTE_RESTART_X
|
||||
const unsigned int COMPUTE_RESTART_VALUES[] = {
|
||||
0, // COMPUTE_RESTART_X
|
||||
0, // COMPUTE_RESTART_Y
|
||||
0, // COMPUTE_RESTART_Z
|
||||
0 // COMPUTE_THREAD_TRACE_ENABLE
|
||||
};
|
||||
|
||||
// Starts at COMPUTE_USER_DATA_0
|
||||
const unsigned int COMPUTE_USER_DATA_VALUES[] = {
|
||||
// Reg name - use in KFDtest - use in ABI
|
||||
arg0, // COMPUTE_USER_DATA_0 - arg0 - resource descriptor for the scratch buffer - 1st dword
|
||||
arg1, // COMPUTE_USER_DATA_1 - arg1 - resource descriptor for the scratch buffer - 2nd dword
|
||||
arg2, // COMPUTE_USER_DATA_2 - arg2 - resource descriptor for the scratch buffer - 3rd dword
|
||||
arg3, // COMPUTE_USER_DATA_3 - arg3 - resource descriptor for the scratch buffer - 4th dword
|
||||
static_cast<uint32_t>(m_scratch_base), // COMPUTE_USER_DATA_4 - flat_scratch_lo
|
||||
static_cast<uint32_t>(m_scratch_base >> 32), // COMPUTE_USER_DATA_4 - flat_scratch_hi
|
||||
0, // COMPUTE_USER_DATA_6 - - AQL queue address, low part
|
||||
0, // COMPUTE_USER_DATA_7 - - AQL queue address, high part
|
||||
0, // COMPUTE_USER_DATA_8 - - kernel arguments block, low part
|
||||
0, // COMPUTE_USER_DATA_9 - - kernel arguments block, high part
|
||||
0, // COMPUTE_USER_DATA_10 - - unused
|
||||
0, // COMPUTE_USER_DATA_11 - - unused
|
||||
0, // COMPUTE_USER_DATA_12 - - unused
|
||||
0, // COMPUTE_USER_DATA_13 - - unused
|
||||
0, // COMPUTE_USER_DATA_14 - - unused
|
||||
0, // COMPUTE_USER_DATA_15 - - unused
|
||||
};
|
||||
|
||||
const unsigned int DISPATCH_INIT_VALUE = 0x00000021 | (hsakmt_is_dgpu() ? 0 : 0x1000) |
|
||||
((m_FamilyId >= FAMILY_NV) ? 0x8000 : 0);
|
||||
// {COMPUTE_SHADER_EN=1, PARTIAL_TG_EN=0, FORCE_START_AT_000=0, ORDERED_APPEND_ENBL=0,
|
||||
// ORDERED_APPEND_MODE=0, USE_THREAD_DIMENSIONS=1, ORDER_MODE=0, DISPATCH_CACHE_CNTL=0,
|
||||
// SCALAR_L1_INV_VOL=0, VECTOR_L1_INV_VOL=0, DATA_ATC=?, RESTORE=0}
|
||||
// Set CS_W32_EN for wave32 workloads for gfx10 since all the shaders used in KFDTest is 32 bit .
|
||||
|
||||
m_IndirectBuf.AddPacket(PM4AcquireMemoryPacket(m_FamilyId));
|
||||
|
||||
m_IndirectBuf.AddPacket(PM4SetShaderRegPacket(mmCOMPUTE_START_X, COMPUTE_DISPATCH_DIMS_VALUES,
|
||||
ARRAY_SIZE(COMPUTE_DISPATCH_DIMS_VALUES)));
|
||||
|
||||
m_IndirectBuf.AddPacket(PM4SetShaderRegPacket(mmCOMPUTE_PGM_LO,
|
||||
(m_FamilyId >= FAMILY_AI) ? COMPUTE_PGM_VALUES_GFX9 : COMPUTE_PGM_VALUES_GFX8,
|
||||
(m_FamilyId >= FAMILY_AI) ? ARRAY_SIZE(COMPUTE_PGM_VALUES_GFX9) : ARRAY_SIZE(COMPUTE_PGM_VALUES_GFX8)));
|
||||
m_IndirectBuf.AddPacket(PM4SetShaderRegPacket(mmCOMPUTE_PGM_RSRC1, COMPUTE_PGM_RSRC,
|
||||
ARRAY_SIZE(COMPUTE_PGM_RSRC)));
|
||||
|
||||
if (m_FamilyId == FAMILY_AL || m_FamilyId == FAMILY_AV) {
|
||||
const unsigned int COMPUTE_PGM_RSRC3[] = {9};
|
||||
m_IndirectBuf.AddPacket(PM4SetShaderRegPacket(mmCOMPUTE_PGM_RSRC3, COMPUTE_PGM_RSRC3,
|
||||
ARRAY_SIZE(COMPUTE_PGM_RSRC3)));
|
||||
}
|
||||
|
||||
m_IndirectBuf.AddPacket(PM4SetShaderRegPacket(mmCOMPUTE_RESOURCE_LIMITS, COMPUTE_RESOURCE_LIMITS,
|
||||
ARRAY_SIZE(COMPUTE_RESOURCE_LIMITS)));
|
||||
m_IndirectBuf.AddPacket(PM4SetShaderRegPacket(mmCOMPUTE_TMPRING_SIZE, COMPUTE_TMPRING_SIZE,
|
||||
ARRAY_SIZE(COMPUTE_TMPRING_SIZE)));
|
||||
m_IndirectBuf.AddPacket(PM4SetShaderRegPacket(mmCOMPUTE_RESTART_X, COMPUTE_RESTART_VALUES,
|
||||
ARRAY_SIZE(COMPUTE_RESTART_VALUES)));
|
||||
|
||||
m_IndirectBuf.AddPacket(PM4SetShaderRegPacket(mmCOMPUTE_USER_DATA_0, COMPUTE_USER_DATA_VALUES,
|
||||
ARRAY_SIZE(COMPUTE_USER_DATA_VALUES)));
|
||||
|
||||
m_IndirectBuf.AddPacket(PM4DispatchDirectPacket(m_DimX, m_DimY, m_DimZ, DISPATCH_INIT_VALUE));
|
||||
|
||||
// EVENT_WRITE.partial_flush causes problems with preemptions in
|
||||
// GWS testing. Since this is specific to this PM4 command and
|
||||
// doesn't affect AQL, it's easier to fix KFDTest than the
|
||||
// firmware.
|
||||
//
|
||||
// Replace PartialFlush with an ReleaseMem (with no interrupt) + WaitRegMem
|
||||
//
|
||||
// Original: m_IndirectBuf.AddPacket(PM4PartialFlushPacket());
|
||||
uint32_t *nop = m_IndirectBuf.AddPacket(PM4NopPacket(2)); // NOP packet with one dword payload for the release-mem fence
|
||||
m_IndirectBuf.AddPacket(PM4ReleaseMemoryPacket(m_FamilyId, true, (uint64_t)&nop[1], 0xdeadbeef));
|
||||
m_IndirectBuf.AddPacket(PM4WaitRegMemPacket(true, (uint64_t)&nop[1], 0xdeadbeef, 4));
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __KFD_DISPATCH__H__
|
||||
#define __KFD_DISPATCH__H__
|
||||
#include "KFDTestUtil.hpp"
|
||||
#include "IndirectBuffer.hpp"
|
||||
#include "BaseQueue.hpp"
|
||||
|
||||
class Dispatch {
|
||||
public:
|
||||
Dispatch(const HsaMemoryBuffer& isaBuf, const bool eventAutoReset = false);
|
||||
~Dispatch();
|
||||
|
||||
void SetArgs(void* pArg1, void* pArg2);
|
||||
|
||||
void SetDim(unsigned int x, unsigned int y, unsigned int z);
|
||||
|
||||
void Submit(BaseQueue& queue);
|
||||
|
||||
void Sync(unsigned int timeout = HSA_EVENTTIMEOUT_INFINITE);
|
||||
|
||||
int SyncWithStatus(unsigned int timeout);
|
||||
|
||||
void SetScratch(int numWaves, int waveSize, HSAuint64 scratch_base);
|
||||
|
||||
void SetSpiPriority(unsigned int priority);
|
||||
|
||||
void SetPriv(bool priv);
|
||||
|
||||
HsaEvent *GetHsaEvent() { return m_pEop; }
|
||||
|
||||
private:
|
||||
void BuildIb();
|
||||
|
||||
private:
|
||||
const HsaMemoryBuffer& m_IsaBuf;
|
||||
|
||||
IndirectBuffer m_IndirectBuf;
|
||||
|
||||
unsigned int m_DimX;
|
||||
unsigned int m_DimY;
|
||||
unsigned int m_DimZ;
|
||||
|
||||
void* m_pArg1;
|
||||
void* m_pArg2;
|
||||
|
||||
HsaEvent* m_pEop;
|
||||
|
||||
bool m_ScratchEn;
|
||||
unsigned int m_ComputeTmpringSize;
|
||||
|
||||
HSAuint64 m_scratch_base;
|
||||
unsigned int m_SpiPriority;
|
||||
unsigned int m_FamilyId;
|
||||
bool m_NeedCwsrWA;
|
||||
};
|
||||
|
||||
#endif // __KFD_DISPATCH__H__
|
||||
@@ -0,0 +1,76 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#include "GoogleTestExtension.hpp"
|
||||
#include "OSWrapper.hpp"
|
||||
|
||||
bool Ok2Run(unsigned int testProfile) {
|
||||
bool testMatchProfile = true;
|
||||
if ((testProfile & g_TestRunProfile) == 0) {
|
||||
WARN() << "Test is skipped beacuse profile does not match current run mode" << std::endl;
|
||||
testMatchProfile = false;
|
||||
}
|
||||
|
||||
return testMatchProfile;
|
||||
}
|
||||
|
||||
// This predication is used when specific HW capabilities must exist for the test to succeed.
|
||||
bool TestReqEnvCaps(unsigned int envCaps) {
|
||||
bool testMatchEnv = true;
|
||||
if ((envCaps & g_TestENVCaps) != envCaps) {
|
||||
WARN() << "Test is skipped due to HW capability issues" << std::endl;
|
||||
testMatchEnv = false;
|
||||
}
|
||||
|
||||
return testMatchEnv;
|
||||
}
|
||||
|
||||
// This predication is used when specific HW capabilities must be absent for the test to succeed.
|
||||
// e.g Testing capabilities not supported by HW scheduling
|
||||
bool TestReqNoEnvCaps(unsigned int envCaps) {
|
||||
bool testMatchEnv = true;
|
||||
if ((envCaps & g_TestENVCaps) != 0) {
|
||||
WARN() << "Test is skipped due to HW capability issues" << std::endl;
|
||||
testMatchEnv = false;
|
||||
}
|
||||
|
||||
return testMatchEnv;
|
||||
}
|
||||
|
||||
std::ostream& operator<< (KFDLog log, LOGTYPE level) {
|
||||
const char *heading;
|
||||
|
||||
if (level == LOGTYPE_WARNING) {
|
||||
SetConsoleTextColor(TEXTCOLOR_YELLOW);
|
||||
heading = "[----------] ";
|
||||
} else {
|
||||
SetConsoleTextColor(TEXTCOLOR_GREEN);
|
||||
heading = "[ ] ";
|
||||
}
|
||||
|
||||
std::clog << heading;
|
||||
SetConsoleTextColor(TEXTCOLOR_WHITE);
|
||||
|
||||
return std::clog;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __GOOGLETEST_EXTENSION__H__
|
||||
#define __GOOGLETEST_EXTENSION__H__
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
#include "hsakmt/hsakmt.h"
|
||||
#include "KFDTestFlags.hpp"
|
||||
|
||||
enum LOGTYPE {
|
||||
LOGTYPE_INFO, // msg header in green
|
||||
LOGTYPE_WARNING // msg header in yellow
|
||||
};
|
||||
|
||||
class KFDLog{};
|
||||
std::ostream& operator << (KFDLog log, LOGTYPE level);
|
||||
|
||||
// @brief Log additional details, to be displayed in the same format as other google test outputs
|
||||
// Currently not supported by gtest
|
||||
// Should be used like cout: LOG() << "message" << value << std::endl;
|
||||
#define LOG() KFDLog() << LOGTYPE_INFO
|
||||
#define WARN() KFDLog() << LOGTYPE_WARNING
|
||||
|
||||
class KFDRecord: public testing::Test {
|
||||
public:
|
||||
KFDRecord(const char *val): m_val(val) {}
|
||||
KFDRecord(std::string &val): m_val(val) {}
|
||||
KFDRecord(HSAint64 val): m_val(std::to_string(val)) {}
|
||||
KFDRecord(HSAuint64 val): m_val(std::to_string(val)) {}
|
||||
KFDRecord(double val): m_val(std::to_string(val)) {}
|
||||
~KFDRecord() {
|
||||
RecordProperty(m_key.str().c_str(), m_val.c_str());
|
||||
}
|
||||
std::stringstream &get_key_stream() {
|
||||
return m_key;
|
||||
}
|
||||
virtual void TestBody() {};
|
||||
private:
|
||||
std::string m_val;
|
||||
std::stringstream m_key;
|
||||
};
|
||||
|
||||
#define RECORD(val) (KFDRecord(val).get_key_stream())
|
||||
|
||||
// All tests MUST be in a try catch since the gtest flag to throw an exception on any fatal failure is enabled
|
||||
#define TEST_START(testProfile) if (Ok2Run(testProfile)) try {
|
||||
#define TEST_END } catch (...) {}
|
||||
|
||||
// Used to wrap setup and teardown functions, anything that is built-in gtest and is not a test
|
||||
#define ROUTINE_START try {
|
||||
#define ROUTINE_END }catch(...) {}
|
||||
|
||||
#define TEST_REQUIRE_ENV_CAPABILITIES(envCaps) if (!TestReqEnvCaps(envCaps)) return;
|
||||
#define TEST_REQUIRE_NO_ENV_CAPABILITIES(envCaps) if (!TestReqNoEnvCaps(envCaps)) return;
|
||||
|
||||
#define ASSERT_SUCCESS(_val) ASSERT_EQ(HSAKMT_STATUS_SUCCESS, (_val))
|
||||
#define EXPECT_SUCCESS(_val) EXPECT_EQ(HSAKMT_STATUS_SUCCESS, (_val))
|
||||
|
||||
#define EXPECT_EQ_GPU(expected, actual , gpuNode) EXPECT_EQ((expected), (actual)) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
#define ASSERT_SUCCESS_GPU(_val, gpuNode) ASSERT_EQ(HSAKMT_STATUS_SUCCESS, (_val)) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
#define EXPECT_SUCCESS_GPU(_val, gpuNode) EXPECT_EQ(HSAKMT_STATUS_SUCCESS, (_val)) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
|
||||
#define ASSERT_NOTNULL(_val) ASSERT_NE((void *)NULL, _val)
|
||||
#define EXPECT_NOTNULL(_val) EXPECT_NE((void *)NULL, _val)
|
||||
|
||||
#define ASSERT_NOTNULL_GPU(_val, gpuNode) ASSERT_NE((void *)NULL, _val) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
#define EXPECT_NOTNULL_GPU(_val, gpuNode) EXPECT_NE((void *)NULL, _val) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
|
||||
#define EXPECT_NE_GPU(expected, actual, gpuNode) EXPECT_NE((expected), (actual)) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
#define EXPECT_GE_GPU(expected, actual, gpuNode) EXPECT_GE((expected), (actual)) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
|
||||
#define ASSERT_GE_GPU(val1, val2, gpuNode) ASSERT_GE((val1), (val2)) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
#define ASSERT_NE_GPU(val1, val2, gpuNode) ASSERT_NE((val1), (val2)) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
#define ASSERT_EQ_GPU(val1, val2, gpuNode) ASSERT_EQ((val1), (val2)) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
|
||||
#define EXPECT_TRUE_GPU(condition, gpuNode) EXPECT_TRUE(condition) << "gpuNodeID: " << std::to_string(gpuNode) << "\n"
|
||||
|
||||
// @brief Determines if it is ok to run a test given input flags
|
||||
bool Ok2Run(unsigned int testProfile);
|
||||
|
||||
// @brief Checks if all HW capabilities needed for a test to run exist
|
||||
bool TestReqEnvCaps(unsigned int hwCaps);
|
||||
|
||||
// @brief Checks if all HW capabilities that prevents a test from running are absent
|
||||
bool TestReqNoEnvCaps(unsigned int hwCaps);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,53 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#include "IndirectBuffer.hpp"
|
||||
#include "GoogleTestExtension.hpp"
|
||||
#include "pm4_pkt_struct_common.h"
|
||||
#include "PM4Packet.hpp"
|
||||
|
||||
|
||||
IndirectBuffer::IndirectBuffer(PACKETTYPE type, unsigned int sizeInDWords, unsigned int NodeId)
|
||||
:m_NumOfPackets(0), m_MaxSize(sizeInDWords), m_ActualSize(0), m_PacketTypeAllowed(type) {
|
||||
m_IndirectBuf = new HsaMemoryBuffer(sizeInDWords*sizeof(unsigned int), NodeId, true/*zero*/,
|
||||
false/*local*/, true/*exec*/, false/*isScratch*/,
|
||||
false/*isReadOnly*/, true/*isUncached*/);
|
||||
}
|
||||
|
||||
IndirectBuffer::~IndirectBuffer(void) {
|
||||
delete m_IndirectBuf;
|
||||
}
|
||||
|
||||
uint32_t *IndirectBuffer::AddPacket(const BasePacket &packet) {
|
||||
EXPECT_EQ(packet.PacketType(), m_PacketTypeAllowed) << "Cannot add a packet since packet type doesn't match queue";
|
||||
|
||||
unsigned int writePtr = m_ActualSize;
|
||||
|
||||
EXPECT_GE(m_MaxSize, packet.SizeInDWords() + writePtr) << "Cannot add a packet, not enough room";
|
||||
|
||||
memcpy(m_IndirectBuf->As<unsigned int*>() + writePtr , packet.GetPacket(), packet.SizeInBytes());
|
||||
m_ActualSize += packet.SizeInDWords();
|
||||
m_NumOfPackets++;
|
||||
|
||||
return m_IndirectBuf->As<HSAuint32 *>() + writePtr;
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
/*
|
||||
* Copyright (C) 2014-2018 Advanced Micro Devices, Inc. All Rights Reserved.
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a
|
||||
* copy of this software and associated documentation files (the "Software"),
|
||||
* to deal in the Software without restriction, including without limitation
|
||||
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
* and/or sell copies of the Software, and to permit persons to whom the
|
||||
* Software is furnished to do so, subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in
|
||||
* all copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
|
||||
* OTHER DEALINGS IN THE SOFTWARE.
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef __INDIRECT_BUFFER__H__
|
||||
#define __INDIRECT_BUFFER__H__
|
||||
|
||||
#include "BasePacket.hpp"
|
||||
#include "KFDTestUtil.hpp"
|
||||
|
||||
/** @class IndirectBuffer
|
||||
* When working with an indirect buffer, create IndirectBuffer, fill it with all the packets you want,
|
||||
* create an indirect packet to point to it, and submit the packet to queue
|
||||
*/
|
||||
class IndirectBuffer {
|
||||
public:
|
||||
// @param[size] Queue max size in DWords
|
||||
// @param[type] Packet type allowed in queue
|
||||
IndirectBuffer(PACKETTYPE type, unsigned int sizeInDWords, unsigned int NodeId);
|
||||
~IndirectBuffer(void);
|
||||
|
||||
// @brief Add packet to queue, all validations are done with gtest ASSERT and EXPECT
|
||||
uint32_t *AddPacket(const BasePacket &packet);
|
||||
// @returns Actual size of the indirect queue in DWords, equivalent to write pointer
|
||||
unsigned int SizeInDWord() { return m_ActualSize; }
|
||||
// @returns Indirect queue address
|
||||
unsigned int *Addr() { return m_IndirectBuf->As<unsigned int*>(); }
|
||||
|
||||
protected:
|
||||
// Number of packets in the queue
|
||||
unsigned int m_NumOfPackets;
|
||||
// Max size of queue in DWords
|
||||
unsigned int m_MaxSize;
|
||||
// Current size of queue in DWords
|
||||
unsigned int m_ActualSize;
|
||||
HsaMemoryBuffer *m_IndirectBuf;
|
||||
// What packets are supported in this queue
|
||||
PACKETTYPE m_PacketTypeAllowed;
|
||||
};
|
||||
|
||||
#endif // __INDIRECT_BUFFER__H__
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user