Merge commit '996e8bbfb71310d9a1b641bc954b69138cf85daa' into develop

This commit is contained in:
systems-assistant[bot]
2025-07-30 15:21:37 +00:00
کامیت c9993f5447
44فایلهای تغییر یافته به همراه5482 افزوده شده و 438 حذف شده
@@ -108,12 +108,20 @@ if (HSA_DEP_ROCPROFILER_REGISTER)
string(APPEND CPACK_RPM_BINARY_PACKAGE_REQUIRES " rocprofiler-register")
endif()
if (NOT DEFINED BUILD_THUNK_VIRTIO)
set(BUILD_THUNK_VIRTIO OFF)
endif()
add_rocm_subdir(libhsakmt "${THUNK_DEFINITIONS}")
set_target_properties(hsakmt PROPERTIES
ARCHIVE_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/archive"
LIBRARY_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/lib"
RUNTIME_OUTPUT_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/libhsakmt/runtime")
if (BUILD_THUNK_VIRTIO)
add_rocm_subdir(libhsakmt/src/virtio "${THUNK_VIRTIO_DEFINITIONS}")
endif()
if (BUILD_ROCR)
add_rocm_subdir(runtime/hsa-runtime "${ROCR_DEFINITIONS}")
set_target_properties(hsa-runtime64 PROPERTIES
@@ -123,6 +131,9 @@ if (BUILD_ROCR)
if (BUILD_SHARED_LIBS)
add_dependencies(hsa-runtime64 hsakmt)
if (BUILD_THUNK_VIRTIO)
add_dependencies(hsa-runtime64 hsakmt_virtio)
endif()
else()
add_dependencies(hsa-runtime64 hsakmt-staticdrm)
endif()
@@ -211,7 +211,7 @@ install ( TARGETS ${HSAKMT_TARGET}
# Install public headers
install ( DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/include/${HSAKMT_TARGET} DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}
COMPONENT dev PATTERN "linux" EXCLUDE )
COMPONENT dev PATTERN "linux" EXCLUDE PATTERN "*virtio*" EXCLUDE)
# Record our usage data for clients find_package calls.
install ( EXPORT ${HSAKMT_TARGET}Targets
@@ -0,0 +1,113 @@
/*
* Copyright © 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person
* obtaining a copy of this software and associated documentation
* files (the "Software"), to deal in the Software without
* restriction, including without limitation the rights to use, copy,
* modify, merge, publish, distribute, sublicense, and/or sell copies
* of the Software, and to permit persons to whom the Software is
* furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice (including
* the next paragraph) shall be included in all copies or substantial
* portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
* EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
* MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
* NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT
* HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY,
* WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
* DEALINGS IN THE SOFTWARE.
*/
#ifndef HSAKMT_VIRTIO_H
#define HSAKMT_VIRTIO_H
#include "hsakmt/linux/kfd_ioctl.h"
#include "hsakmt/hsakmt.h"
#include <libdrm/amdgpu.h>
#ifdef __cplusplus
extern "C" {
#endif
HSAKMT_STATUS HSAKMTAPI vhsaKmtOpenKFD(void);
HSAKMT_STATUS HSAKMTAPI vhsaKmtCloseKFD(void);
HSAKMT_STATUS HSAKMTAPI vhsaKmtAllocMemory(HSAuint32 PreferredNode, HSAuint64 SizeInBytes,
HsaMemFlags MemFlags, void** MemoryAddress);
HSAKMT_STATUS HSAKMTAPI vhsaKmtFreeMemory(void* MemoryAddress, HSAuint64 SizeInBytes);
HSAKMT_STATUS HSAKMTAPI vhsaKmtMapMemoryToGPUNodes(void* MemoryAddress, HSAuint64 MemorySizeInBytes,
HSAuint64* AlternateVAGPU,
HsaMemMapFlags MemMapFlags,
HSAuint64 NumberOfNodes, HSAuint32* NodeArray);
HSAKMT_STATUS HSAKMTAPI vhsaKmtUnmapMemoryToGPU(void* MemoryAddress);
HSAKMT_STATUS HSAKMTAPI vhsaKmtAvailableMemory(HSAuint32 Node, HSAuint64* AvailableBytes);
HSAKMT_STATUS HSAKMTAPI vhsaKmtMapMemoryToGPU(void* MemoryAddress, HSAuint64 MemorySizeInBytes,
HSAuint64* AlternateVAGPU);
HSAKMT_STATUS HSAKMTAPI vhsaKmtRegisterMemoryWithFlags(void* MemoryAddress,
HSAuint64 MemorySizeInBytes,
HsaMemFlags MemFlags);
HSAKMT_STATUS HSAKMTAPI vhsaKmtDeregisterMemory(void* MemoryAddress);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetVersion(HsaVersionInfo* v);
HSAKMT_STATUS HSAKMTAPI vhsaKmtAcquireSystemProperties(HsaSystemProperties* SystemProperties);
HSAKMT_STATUS HSAKMTAPI vhsaKmtReleaseSystemProperties(void);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetNodeProperties(HSAuint32 NodeId,
HsaNodeProperties* NodeProperties);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetXNACKMode(HSAint32* enable);
HSAKMT_STATUS HSAKMTAPI vhsaKmtRuntimeEnable(void* rDebug, bool setupTtmp);
HSAKMT_STATUS HSAKMTAPI vhsaKmtRuntimeDisable(void);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetNodeMemoryProperties(HSAuint32 NodeId, HSAuint32 NumBanks,
HsaMemoryProperties* MemoryProperties);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetNodeCacheProperties(HSAuint32 NodeId, HSAuint32 ProcessorId,
HSAuint32 NumCaches,
HsaCacheProperties* CacheProperties);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetNodeIoLinkProperties(HSAuint32 NodeId, HSAuint32 NumIoLinks,
HsaIoLinkProperties* IoLinkProperties);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetClockCounters(HSAuint32 NodeId, HsaClockCounters* Counters);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetAMDGPUDeviceHandle(HSAuint32 NodeId,
HsaAMDGPUDeviceHandle* DeviceHandle);
HSAKMT_STATUS HSAKMTAPI vhsaKmtQueryPointerInfo(const void* Pointer, HsaPointerInfo* PointerInfo);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetTileConfig(HSAuint32 NodeId, HsaGpuTileConfig* config);
HSAKMT_STATUS HSAKMTAPI vhsaKmtCreateEvent(HsaEventDescriptor* EventDesc, _Bool ManualReset,
_Bool IsSignaled, HsaEvent** Event);
HSAKMT_STATUS HSAKMTAPI vhsaKmtDestroyEvent(HsaEvent* Event);
HSAKMT_STATUS HSAKMTAPI vhsaKmtSetEvent(HsaEvent* Event);
HSAKMT_STATUS HSAKMTAPI vhsaKmtResetEvent(HsaEvent* Event);
HSAKMT_STATUS HSAKMTAPI vhsaKmtQueryEventState(HsaEvent* Event);
HSAKMT_STATUS HSAKMTAPI vhsaKmtWaitOnMultipleEvents(HsaEvent* Events[], HSAuint32 NumEvents,
bool WaitOnAll, HSAuint32 Milliseconds);
HSAKMT_STATUS HSAKMTAPI vhsaKmtWaitOnEvent(HsaEvent* Event, HSAuint32 Milliseconds);
HSAKMT_STATUS HSAKMTAPI vhsaKmtWaitOnEvent_Ext(HsaEvent* Event, HSAuint32 Milliseconds,
uint64_t* event_age);
HSAKMT_STATUS HSAKMTAPI vhsaKmtWaitOnMultipleEvents_Ext(HsaEvent* Events[], HSAuint32 NumEvents,
bool WaitOnAll, HSAuint32 Milliseconds,
uint64_t* event_age);
HSAKMT_STATUS HSAKMTAPI vhsaKmtSetTrapHandler(HSAuint32 NodeId, void* TrapHandlerBaseAddress,
HSAuint64 TrapHandlerSizeInBytes,
void* TrapBufferBaseAddress,
HSAuint64 TrapBufferSizeInBytes);
HSAKMT_STATUS HSAKMTAPI vhsaKmtCreateQueueExt(HSAuint32 NodeId, HSA_QUEUE_TYPE Type,
HSAuint32 QueuePercentage,
HSA_QUEUE_PRIORITY Priority, HSAuint32 SdmaEngineId,
void* QueueAddress, HSAuint64 QueueSizeInBytes,
HsaEvent* Event, HsaQueueResource* QueueResource);
HSAKMT_STATUS HSAKMTAPI vhsaKmtCreateQueue(HSAuint32 NodeId, HSA_QUEUE_TYPE Type,
HSAuint32 QueuePercentage, HSA_QUEUE_PRIORITY Priority,
void* QueueAddress, HSAuint64 QueueSizeInBytes,
HsaEvent* Event, HsaQueueResource* QueueResource);
HSAKMT_STATUS HSAKMTAPI vhsaKmtDestroyQueue(HSA_QUEUEID QueueId);
HSAKMT_STATUS HSAKMTAPI vhsaKmtRegisterGraphicsHandleToNodes(
HSAuint64 GraphicsResourceHandle, HsaGraphicsResourceInfo* GraphicsResourceInfo,
HSAuint64 NumberOfNodes, HSAuint32* NodeArray);
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetRuntimeCapabilities(HSAuint32* caps_mask);
int vamdgpu_query_gpu_info(amdgpu_device_handle dev, void* out);
#ifdef __cplusplus
}
#endif
#endif /* HSAKMT_VIRTIO_H */
@@ -0,0 +1,38 @@
/* GPL-2.0 WITH Linux-syscall-note */
/*
* This file was copied from inux-libc-dev package
* This header provides interface to linux kernel udmabuf drver
* Modifications may have been made.
*/
#ifndef _THUNK_UDMABUF_H
#define _THUNK_UDMABUF_H
#include <linux/types.h>
#include <linux/ioctl.h>
#define UDMABUF_FLAGS_CLOEXEC 0x01
struct udmabuf_create {
__u32 memfd;
__u32 flags;
__u64 offset;
__u64 size;
};
struct udmabuf_create_item {
__u32 memfd;
__u32 __pad;
__u64 offset;
__u64 size;
};
struct udmabuf_create_list {
__u32 flags;
__u32 count;
struct udmabuf_create_item list[];
};
#define UDMABUF_CREATE _IOW('u', 0x42, struct udmabuf_create)
#define UDMABUF_CREATE_LIST _IOW('u', 0x43, struct udmabuf_create_list)
#endif /* _THUNK_UDMABUF_H */
@@ -23,6 +23,7 @@
* DEALINGS IN THE SOFTWARE.
*/
#define _GNU_SOURCE
#include "libhsakmt.h"
#include "fmm.h"
#include "hsakmt/hsakmtmodel.h"
@@ -44,6 +45,11 @@
#include "rbtree.h"
#include <amdgpu.h>
#include <fcntl.h>
#include <sys/ioctl.h>
#include <sys/stat.h>
#include "hsakmt/linux/udmabuf.h"
#ifndef MPOL_F_STATIC_NODES
/* Bug in numaif.h, this should be defined in there. Definition copied
* from linux/mempolicy.h.
@@ -148,6 +154,10 @@ static void *reserved_aperture_allocate_aligned(manageable_aperture_t *aper,
uint64_t size, uint64_t align);
static void reserved_aperture_release(manageable_aperture_t *aper,
void *addr, uint64_t size);
static int bind_mem_to_numa(uint32_t node_id, void *mem,
uint64_t SizeInBytes, HsaMemFlags mflags);
static const manageable_aperture_ops_t reserved_aperture_ops = {
reserved_aperture_allocate_aligned,
reserved_aperture_release
@@ -758,7 +768,7 @@ static void *reserved_aperture_allocate_aligned(manageable_aperture_t *app,
}
void *hsakmt_mmap_allocate_aligned(int prot, int flags, uint64_t size, uint64_t align,
uint64_t guard_size, void *aper_base, void *aper_limit)
uint64_t guard_size, void *aper_base, void *aper_limit, int fd)
{
void *addr, *aligned_addr, *aligned_end, *mapping_end;
uint64_t aligned_padded_size;
@@ -766,7 +776,7 @@ void *hsakmt_mmap_allocate_aligned(int prot, int flags, uint64_t size, uint64_t
aligned_padded_size = size + guard_size * 2 + (align - PAGE_SIZE);
/* Map memory PROT_NONE to alloc address space only */
addr = mmap(0, aligned_padded_size, PROT_NONE, flags, -1, 0);
addr = mmap(0, aligned_padded_size, PROT_NONE, flags | MAP_ANONYMOUS, -1, 0);
if (addr == MAP_FAILED) {
pr_err("mmap failed: %s\n", strerror(errno));
return NULL;
@@ -795,7 +805,7 @@ void *hsakmt_mmap_allocate_aligned(int prot, int flags, uint64_t size, uint64_t
return aligned_addr;
/* MAP_FIXED to the aligned address with required prot */
addr = mmap(aligned_addr, size, prot, flags | MAP_FIXED, -1, 0);
addr = mmap(aligned_addr, size, prot, flags | MAP_FIXED, fd, 0);
if (addr == MAP_FAILED) {
pr_err("mmap failed: %s\n", strerror(errno));
return NULL;
@@ -859,7 +869,7 @@ static void *mmap_aperture_allocate_aligned(manageable_aperture_t *aper,
guard_size = (uint64_t)aper->guard_pages * PAGE_SIZE;
return hsakmt_mmap_allocate_aligned(PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE,
size, align, guard_size, aper->base, aper->limit);
size, align, guard_size, aper->base, aper->limit, -1);
}
static void mmap_aperture_release(manageable_aperture_t *aper,
@@ -1494,7 +1504,7 @@ void *hsakmt_fmm_allocate_scratch(uint32_t gpu_id, void *address, uint64_t Memor
mem = hsakmt_mmap_allocate_aligned(PROT_READ | PROT_WRITE,
MAP_PRIVATE | MAP_ANONYMOUS,
aligned_size, SCRATCH_ALIGN, 0,
0, (void *)LONG_MAX);
0, (void *)LONG_MAX, -1);
}
/* Remember scratch backing aperture for later */
@@ -1603,6 +1613,126 @@ static void *fmm_allocate_va(uint32_t gpu_id, void *address, uint64_t size,
return mem;
}
/* use udmabuf driver to allocate buf */
static void* udmabuf_allocation(uint32_t gpu_id, uint32_t node_id, uint64_t size,
manageable_aperture_t *aperture, uint64_t alignment,
HsaMemFlags mflags, vm_object_t** vm_obj)
{
struct kfd_ioctl_import_dmabuf_args importArgs = {0};
int memfd, dmabuf_fd;
long long node_size, free_size;
struct udmabuf_create create;
uint64_t alignment_size;
uint32_t numa_node_id;
uint64_t guard_size;
void *mem;
int ret;
dmabuf_fd = -1;
memfd = -1;
*vm_obj = NULL;
memfd = memfd_create("thunk_memfd", MFD_ALLOW_SEALING);
if (memfd == -1) {
pr_debug("running kernel does not support memfd\n");
return NULL;
}
if (ftruncate(memfd, size) == -1) {
pr_debug("ftruncate fail\n");
goto error_release_memfd;
}
pr_debug("PID: %jd; fd: %d; /proc/%jd/fd/%d\n",
(intmax_t) getpid(), memfd, (intmax_t) getpid(), memfd);
if (fcntl(memfd, F_ADD_SEALS, F_SEAL_SHRINK | F_SEAL_GROW) < 0) {
pr_debug("fcntl fail %s\n", strerror(errno));
goto error_release_memfd;
}
alignment_size = PAGE_SIZE << svm.alignment_order;
alignment = alignment ? alignment : aperture->align;
while (alignment < alignment_size && size >= (alignment << 1))
alignment <<= 1;
guard_size = (uint64_t)aperture->guard_pages * PAGE_SIZE;
mem = hsakmt_mmap_allocate_aligned(PROT_WRITE | PROT_READ, MAP_NORESERVE | MAP_SHARED,
size, alignment, guard_size, aperture->base, aperture->limit, memfd);
if (!mem)
goto error_release_memfd;
/* set madvise flags to HUGEPAGE if allocate more than 2MB */
if (size >= (2 * 1024 * 1024))
madvise(mem, size, MADV_HUGEPAGE);
/* always bind to numa node */
mflags.ui32.NoSubstitute = 1;
/* Bind to NUMA node */
/* node_id is gpu id, get closed numa id */
numa_node_id = hsakmt_get_direct_link_cpu(node_id);
if (bind_mem_to_numa(numa_node_id, mem, size, mflags))
goto error_release_aperture;
node_size = numa_node_size(numa_node_id, &free_size);
pr_debug("udmabuf_allocation: numa_node_id %d, node_size %lld, free_size %lld\n",
numa_node_id, node_size, free_size);
/* compare free size at numa_node_id with size */
if ((uint64_t)free_size < size) {
pr_debug("udmabuf_allocation: has no enough ram on numa_node_id %d, node_size %lld, free_size %lld\n",
numa_node_id, node_size, free_size);
goto error_release_aperture;
}
create.memfd = memfd;
create.flags = UDMABUF_FLAGS_CLOEXEC;
create.offset = 0;
create.size = size;
dmabuf_fd = ioctl(hsakmt_udmabuf_dev_fd, UDMABUF_CREATE, &create);
if (dmabuf_fd < 0) {
pr_debug("ioctl UDMABUF_CREATE failed\n");
goto error_release_aperture;
}
importArgs.va_addr = (uint64_t)mem;
importArgs.gpu_id = gpu_id;
importArgs.dmabuf_fd = dmabuf_fd;
ret = hsakmt_ioctl(hsakmt_kfd_fd, AMDKFD_IOC_IMPORT_DMABUF, (void *)&importArgs);
if (ret) {
pr_debug("ioctl AMDKFD_IOC_IMPORT_DMABUF failed\n, ret 0x%x", ret);
goto error_release_dmabuf;
}
/* Allocate object */
pthread_mutex_lock(&aperture->fmm_mutex);
*vm_obj = aperture_allocate_object(aperture, mem, importArgs.handle,
size, mflags);
pthread_mutex_unlock(&aperture->fmm_mutex);
if (*vm_obj == NULL)
goto error_release_dmabuf;
/* after import udmabuf into kfd driver close dmabuf_fd
* as kfd driver holds the dmabuf
*/
close(dmabuf_fd);
close(memfd);
return mem;
error_release_dmabuf:
close(dmabuf_fd);
error_release_aperture:
aperture_release_area(aperture, mem, size);
error_release_memfd:
close(memfd);
return NULL;
}
void *hsakmt_fmm_allocate_device(uint32_t gpu_id, uint32_t node_id, void *address,
uint64_t MemorySizeInBytes, uint64_t alignment, HsaMemFlags mflags)
{
@@ -1653,35 +1783,51 @@ void *hsakmt_fmm_allocate_device(uint32_t gpu_id, uint32_t node_id, void *addres
if (mflags.ui32.Contiguous)
ioc_flags |= KFD_IOC_ALLOC_MEM_FLAGS_CONTIGUOUS_BEST_EFFORT;
mem = __fmm_allocate_device(gpu_id, address, size, aperture, &mmap_offset,
ioc_flags, alignment, &vm_obj);
if (mem && vm_obj) {
pthread_mutex_lock(&aperture->fmm_mutex);
/* Store memory allocation flags, not ioc flags */
vm_obj->mflags = mflags;
hsakmt_gpuid_to_nodeid(gpu_id, &vm_obj->node_id);
pthread_mutex_unlock(&aperture->fmm_mutex);
mem = NULL;
if (hsakmt_udmabuf_dev_fd > 0 && aperture == svm.dgpu_aperture && !hsakmt_is_dgpu
&& aperture->ops == &mmap_aperture_ops) {
mem = udmabuf_allocation(gpu_id, node_id, size, aperture, alignment,
mflags, &vm_obj);
pr_debug("udmabuf_allocation mem %p\n", mem);
if (!mem)
pr_debug("udmabuf_allocation allocation fail\n");
}
/* if alloc vram-only not mmap to cpu vm since no va */
if (mem && !mflags.ui32.NoAddress) {
void *ret = fmm_map_to_cpu(mem, MemorySizeInBytes,
/* env HSA_USE_UDMABUF not set, or not apu, or cannot use udmabuf,
* fall back to use device driver to allocate memory
*/
if (!mem) {
mem = __fmm_allocate_device(gpu_id, address, size, aperture, &mmap_offset,
ioc_flags, alignment, &vm_obj);
/* if alloc vram-only not mmap to cpu vm since no va */
if (mem && !mflags.ui32.NoAddress) {
void *ret = fmm_map_to_cpu(mem, MemorySizeInBytes,
mflags.ui32.HostAccess,
gpu_mem[gpu_mem_id].drm_render_fd,
mmap_offset);
if (ret == MAP_FAILED) {
__fmm_release(vm_obj, aperture);
return NULL;
}
if (ret == MAP_FAILED) {
__fmm_release(vm_obj, aperture);
return NULL;
}
#ifdef SANITIZER_AMDGPU
if (vm_obj) {
vm_obj->mmap_flags = mflags.ui32.HostAccess ? PROT_READ | PROT_WRITE : PROT_NONE;
vm_obj->mmap_fd = gpu_mem[gpu_mem_id].drm_render_fd;
vm_obj->mmap_offset = mmap_offset;
}
if (vm_obj) {
vm_obj->mmap_flags = mflags.ui32.HostAccess ? PROT_READ | PROT_WRITE : PROT_NONE;
vm_obj->mmap_fd = gpu_mem[gpu_mem_id].drm_render_fd;
vm_obj->mmap_offset = mmap_offset;
}
#endif
}
}
if (mem && vm_obj) {
pthread_mutex_lock(&aperture->fmm_mutex);
/* Store memory allocation flags, not ioc flags */
vm_obj->mflags = mflags;
hsakmt_gpuid_to_nodeid(gpu_id, &vm_obj->node_id);
pthread_mutex_unlock(&aperture->fmm_mutex);
}
return mem;
@@ -1773,7 +1919,7 @@ static void *fmm_allocate_host_cpu(void *address, uint64_t MemorySizeInBytes,
return mem;
}
static int bind_mem_to_numa(uint32_t node_id, void *mem,
static int bind_mem_to_numa(uint32_t numa_node_id, void *mem,
uint64_t SizeInBytes, HsaMemFlags mflags)
{
int mode = MPOL_F_STATIC_NODES;
@@ -1782,34 +1928,37 @@ static int bind_mem_to_numa(uint32_t node_id, void *mem,
long r;
pr_debug("%s mem %p flags 0x%x size 0x%lx node_id %d\n", __func__,
mem, mflags.Value, SizeInBytes, node_id);
mem, mflags.Value, SizeInBytes, numa_node_id);
if (mflags.ui32.NoNUMABind)
return 0;
if (numa_available() == -1)
return 0;
if (mflags.ui32.NoNUMABind || numa_available() == -1) {
/* but need bind to a numa node */
if (mflags.ui32.NoSubstitute)
return -EFAULT;
else
return 0;
}
num_node = numa_max_node() + 1;
/* Ignore binding requests to invalid nodes IDs */
if (node_id >= (unsigned)num_node) {
pr_warn("node_id %d >= num_node %d\n", node_id, num_node);
return 0;
if (numa_node_id >= (unsigned)num_node || numa_node_id == INVALID_NODEID || num_node <= 1) {
pr_warn("numa_node_id is out range: numa_node_id %d, num_node %d\n", numa_node_id, num_node);
if (mflags.ui32.NoSubstitute)
return -EFAULT;
else
return 0;
}
if (num_node <= 1)
return 0;
node_mask = numa_bitmask_alloc(num_node);
if (!node_mask)
return -ENOMEM;
#ifdef __PPC64__
numa_bitmask_setbit(node_mask, node_id * 8);
numa_bitmask_setbit(node_mask, numa_node_id * 8);
#else
numa_bitmask_setbit(node_mask, node_id);
numa_bitmask_setbit(node_mask, numa_node_id);
#endif
mode |= mflags.ui32.NoSubstitute ? MPOL_BIND : MPOL_PREFERRED;
r = mbind(mem, SizeInBytes, mode, node_mask->maskp, num_node + 1, 0);
numa_bitmask_free(node_mask);
@@ -100,7 +100,7 @@ HSAKMT_STATUS hsakmt_fmm_map_to_gpu_nodes(void *address, uint64_t size,
int hsakmt_open_drm_render_device(int minor);
void *hsakmt_mmap_allocate_aligned(int prot, int flags, uint64_t size, uint64_t align,
uint64_t guard_size, void *aper_base, void *aper_limit);
uint64_t guard_size, void *aper_base, void *aper_limit, int fd);
extern int (*hsakmt_fn_amdgpu_device_get_fd)(HsaAMDGPUDeviceHandle device_handle);
#endif /* FMM_H_ */
@@ -28,6 +28,7 @@
// HSAKMT global data
int hsakmt_kfd_fd = -1;
int hsakmt_udmabuf_dev_fd = -1;
unsigned long hsakmt_kfd_open_count;
unsigned long hsakmt_system_properties_count;
pthread_mutex_t hsakmt_mutex = PTHREAD_MUTEX_INITIALIZER;
@@ -33,6 +33,7 @@
#include <limits.h>
extern int hsakmt_kfd_fd;
extern int hsakmt_udmabuf_dev_fd;
extern unsigned long hsakmt_kfd_open_count;
extern bool hsakmt_forked;
extern pthread_mutex_t hsakmt_mutex;
@@ -46,6 +46,7 @@
int (*hsakmt_fn_amdgpu_device_get_fd)(HsaAMDGPUDeviceHandle device_handle);
static const char kfd_device_name[] = "/dev/kfd";
static const char kfd_udmabuf_device_name[] = "/dev/udmabuf";
static pid_t parent_pid = -1;
int hsakmt_debug_level;
bool hsakmt_forked;
@@ -108,6 +109,10 @@ static void clear_after_fork(void)
close(hsakmt_kfd_fd);
hsakmt_kfd_fd = -1;
}
if (hsakmt_udmabuf_dev_fd > 0) {
close(hsakmt_udmabuf_dev_fd);
hsakmt_udmabuf_dev_fd = -1;
}
hsakmt_kfd_open_count = 0;
parent_pid = -1;
hsakmt_forked = false;
@@ -152,6 +157,7 @@ HSAKMT_STATUS HSAKMTAPI hsaKmtOpenKFD(void)
HsaSystemProperties sys_props;
char *error;
char *useSvmStr;
char *useUdmaBuf;
pthread_mutex_lock(&hsakmt_mutex);
@@ -195,6 +201,18 @@ HSAKMT_STATUS HSAKMTAPI hsaKmtOpenKFD(void)
if (result != HSAKMT_STATUS_SUCCESS)
goto kfd_version_failed;
/* check if udmabuf is enabled by env HSA_USE_UDMABUF */
useUdmaBuf = getenv("HSA_USE_UDMABUF");
if (useUdmaBuf && atoi(useUdmaBuf)) {
/* open udmabuf device */
hsakmt_udmabuf_dev_fd = open(kfd_udmabuf_device_name, 0);
if (hsakmt_udmabuf_dev_fd < 0)
pr_debug("running kernel does not support udmabuf\n");
else
pr_debug("udmabuf is enabled\n");
} else
pr_debug("udmabuf is not enabled\n");
useSvmStr = getenv("HSA_USE_SVM");
hsakmt_is_svm_api_supported = !(useSvmStr && !strcmp(useSvmStr, "0"));
if(!hsakmt_use_model)
@@ -541,7 +541,7 @@ static int handle_concrete_asic(struct queue *q,
void *addr = hsakmt_mmap_allocate_aligned(PROT_READ | PROT_WRITE,
MAP_ANONYMOUS | MAP_PRIVATE,
size, GPU_HUGE_PAGE_SIZE, 0,
0, (void *)LONG_MAX);
0, (void *)LONG_MAX, -1);
if (!addr) {
pr_err("mmap failed to alloc ctx area size 0x%x: %s\n",
size, strerror(errno));
@@ -1794,9 +1794,11 @@ static int32_t gpu_get_direct_link_cpu(uint32_t gpu_node, node_props_t *node_pro
return -1;
for (i = 0; i < node_props[gpu_node].node.NumIOLinks; i++)
if (props[i].IoLinkType == HSA_IOLINKTYPE_PCIEXPRESS &&
props[i].Weight <= 20) /* >20 is GPU->CPU->GPU */
return props[i].NodeTo;
if ((props[i].IoLinkType == HSA_IOLINKTYPE_PCIEXPRESS || props[i].IoLinkType == HSA_IOLINK_TYPE_XGMI) &&
props[i].Weight <= 20) /* >20 is GPU->CPU->GPU */{
if (!node_props[props[i].NodeTo].node.KFDGpuID)
return props[i].NodeTo;
}
return -1;
}
@@ -0,0 +1,88 @@
# Copyright 2025 Advanced Micro Devices, Inc.
# Permission is hereby granted, free of charge, to any person obtaining a
# copy of this software and associated documentation files (the "Software"),
# to deal in the Software without restriction, including without limitation
# the rights to use, copy, modify, merge, publish, distribute, sublicense,
# and/or sell copies of the Software, and to permit persons to whom the
# Software is furnished to do so, subject to the following conditions:
# The above copyright notice and this permission notice shall be included in
# all copies or substantial portions of the Software.
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
# THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
# OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
# ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
# OTHER DEALINGS IN THE SOFTWARE.
cmake_minimum_required ( VERSION 3.7 )
set (CMAKE_VERBOSE_MAKEFILE ON)
set ( HSAKMT_VIRTIO "hsakmt_virtio" )
set ( HSAKMT_VIRTIO_TARGET "${HSAKMT_VIRTIO}" )
project ( ${HSAKMT_VIRTIO_TARGET} VERSION 1.0)
## Compiler flags
set ( HSAKMT_VIRTIO_C_FLAGS -fPIC -W -Wall -Wextra -Wno-unused-parameter -Wformat-security -Wswitch-default -Wundef -Wshadow -Wpointer-arith -Wbad-function-cast -Wcast-qual -Wstrict-prototypes -Wmissing-prototypes -Wmissing-declarations -Wredundant-decls -Wunreachable-code -std=gnu99 -fvisibility=hidden )
if ( CMAKE_COMPILER_IS_GNUCC )
set ( HSAKMT_VIRTIO_C_FLAGS "${HSAKMT_VIRTIO_C_FLAGS}" -Wlogical-op )
endif ()
set ( HSAKMT_VIRTIO_LINKER_SCRIPT "${CMAKE_CURRENT_SOURCE_DIR}/libhsakmt_virtio.ver" )
set ( HSAKMT_VIRTIO_LINK_FLAGS "-Wl,--enable-new-dtags -Wl,--version-script=${HSAKMT_VIRTIO_LINKER_SCRIPT} -Wl,-z,nodelete")
if ( "${CMAKE_BUILD_TYPE}" STREQUAL Release )
set ( HSAKMT_VIRTIO_C_FLAGS "${HSAKMT_VIRTIO_C_FLAGS}" -O2 )
else ()
set ( HSAKMT_VIRTIO_C_FLAGS "${HSAKMT_VIRTIO_C_FLAGS}" -g )
endif ()
set ( HSAKMT_VIRTIO_SRC "virtio_gpu.c"
"hsakmt_virtio_vm.c"
"hsakmt_virtio_device.c"
"hsakmt_virtio_memory.c"
"hsakmt_virtio_amdgpu.c"
"hsakmt_virtio_events.c"
"hsakmt_virtio_queues.c"
"hsakmt_virtio_topology.c"
"hsakmt_virtio_openclose.c"
"../rbtree.c" )
add_library ( ${HSAKMT_VIRTIO_TARGET} STATIC ${HSAKMT_VIRTIO_SRC} )
target_sources ( ${HSAKMT_VIRTIO_TARGET} PRIVATE ${HSAKMT_VIRTIO_SRC} )
target_compile_options ( ${HSAKMT_VIRTIO_TARGET} PRIVATE ${HSAKMT_VIRTIO_C_FLAGS} )
target_include_directories ( ${HSAKMT_VIRTIO_TARGET}
PUBLIC
$<BUILD_INTERFACE:${CMAKE_CURRENT_SOURCE_DIR}/include>
$<INSTALL_INTERFACE:${CMAKE_INSTALL_INCLUDEDIR}>
PRIVATE
${CMAKE_CURRENT_SOURCE_DIR}/virtio
${CMAKE_CURRENT_SOURCE_DIR}/../
${CMAKE_CURRENT_SOURCE_DIR}/../../include
${CMAKE_CURRENT_SOURCE_DIR}/include/linux )
set_property(TARGET ${HSAKMT_VIRTIO_TARGET} PROPERTY LINK_FLAGS ${HSAKMT_VIRTIO_LINK_FLAGS})
find_package ( PkgConfig )
## If environment variable DRM_DIR is set, the script
## will pick up the corresponding libraries from that path.
list ( PREPEND CMAKE_PREFIX_PATH "${DRM_DIR}" )
pkg_check_modules ( DRM REQUIRED IMPORTED_TARGET libdrm )
pkg_check_modules ( DRM_AMDGPU REQUIRED IMPORTED_TARGET libdrm_amdgpu )
target_include_directories ( ${HSAKMT_VIRTIO_TARGET} PRIVATE ${DRM_AMDGPU_INCLUDE_DIRS} )
target_include_directories ( ${HSAKMT_VIRTIO_TARGET} PRIVATE ${DRM_INCLUDE_DIRS} )
target_link_libraries ( ${HSAKMT_VIRTIO_TARGET}
PRIVATE ${DRM_LDFLAGS} ${DRM_AMDGPU_LDFLAGS} pthread rt c ${CMAKE_DL_LIBS} )
@@ -0,0 +1,50 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#include "hsakmt/hsakmt_virtio.h"
#include "hsakmt_virtio_device.h"
int vamdgpu_query_gpu_info(amdgpu_device_handle handle, void* out) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GPU_INFO,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
int ret = vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (!ret) memcpy(out, &rsp->gpu_info, sizeof(struct amdgpu_gpu_info));
return ret;
}
HSAKMT_STATUS vhsaKmtGetAMDGPUDeviceHandle(HSAuint32 NodeId, HsaAMDGPUDeviceHandle* DeviceHandle) {
CHECK_VIRTIO_KFD_OPEN();
return HSAKMT_STATUS_SUCCESS;
}
@@ -0,0 +1,33 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
* All Rights Reserved.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* on the rights to use, copy, modify, merge, publish, distribute, sub
* license, and/or sell copies of the Software, and to permit persons to whom
* the Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice (including the next
* paragraph) shall be included in all copies or substantial portions of the
* Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT. IN NO EVENT SHALL
* THE AUTHOR(S) AND/OR THEIR SUPPLIERS BE LIABLE FOR ANY CLAIM,
* DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
* OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE
* USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
#include "hsakmt_virtio_device.h"
int vhsakmt_execbuf_cpu(vhsakmt_device_handle dev, struct vhsakmt_ccmd_req* req, const char* from) {
return virtio_gpu_exec_cmd(dev->vgdev, req, true);
}
void* vhsakmt_alloc_rsp(vhsakmt_device_handle dev, struct vhsakmt_ccmd_req* req, uint32_t sz) {
return virtio_gpu_alloc_rsp(dev->vgdev, req, sz);
}
@@ -0,0 +1,189 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
* All Rights Reserved.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* on the rights to use, copy, modify, merge, publish, distribute, sub
* license, and/or sell copies of the Software, and to permit persons to whom
* the Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice (including the next
* paragraph) shall be included in all copies or substantial portions of the
* Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT. IN NO EVENT SHALL
* THE AUTHOR(S) AND/OR THEIR SUPPLIERS BE LIABLE FOR ANY CLAIM,
* DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
* OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE
* USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
#ifndef HSAKMT_VIRTIO_DEVICE_H
#define HSAKMT_VIRTIO_DEVICE_H
#include "hsakmt_virtio_proto.h"
#include "rbtree.h"
#include "virtio_gpu.h"
#include <stdatomic.h>
#ifdef __cplusplus
extern "C" {
#endif
#define vhsakmt_atomic_inc_return(ptr) (atomic_fetch_add((ptr), 1) + 1)
#define vhsakmt_atomic_dec_return(ptr) (atomic_fetch_sub((ptr), 1) - 1)
#define VHSA_VPTR_TO_UINT64(vptr) ((uint64_t)(unsigned long)(vptr))
#define VHSA_UINT64_TO_VPTR(v) ((void*)(unsigned long)(v))
extern int vhsakmt_debug_level;
#define vhsakmt_print(level, fmt, ...) \
do { \
if (level <= vhsakmt_debug_level) fprintf(stderr, fmt, ##__VA_ARGS__); \
} while (0)
#define VHSAKMT_DEBUG_LEVEL_DEFAULT -1
#define VHSAKMT_DEBUG_LEVEL_ERR 3
#define VHSAKMT_DEBUG_LEVEL_WARNING 4
#define VHSAKMT_DEBUG_LEVEL_INFO 6
#define VHSAKMT_DEBUG_LEVEL_DEBUG 7
#define vhsa_err(fmt, ...) vhsakmt_print(VHSAKMT_DEBUG_LEVEL_ERR, fmt, ##__VA_ARGS__)
#define vhsa_warn(fmt, ...) vhsakmt_print(VHSAKMT_DEBUG_LEVEL_WARNING, fmt, ##__VA_ARGS__)
#define vhsa_info(fmt, ...) vhsakmt_print(VHSAKMT_DEBUG_LEVEL_INFO, fmt, ##__VA_ARGS__)
#define vhsa_debug(fmt, ...) vhsakmt_print(VHSAKMT_DEBUG_LEVEL_DEBUG, fmt, ##__VA_ARGS__)
struct vhsakmt_device;
struct vhsakmt_bo;
typedef struct vhsakmt_device* vhsakmt_device_handle;
typedef struct vhsakmt_bo* vhsakmt_bo_handle;
typedef rbtree_node_t* bo_entry;
extern pthread_mutex_t dev_mutex;
extern vhsakmt_device_handle dev_list;
#define VHSA_BO_KFD_MEM 1 << 0 /* allocated from KFD (hsaKmtAllocMemory) */
#define VHSA_BO_USERPTR 1 << 1
#define VHSA_BO_QUEUE_BUFFER 1 << 2 /* allocated from KFD, but used for queue CMD submit */
#define VHSA_BO_QUEUE_DOORBELL 1 << 3 /* doorbell memory */
#define VHSA_BO_QUEUE_RW_PTR 1 << 4 /* queue read write ptr, from host map to guest*/
/* allocated from KFD, but used for AQL queue read write ptr */
#define VHSA_BO_QUEUE_AQL_RW_PTR 1 << 5
#define VHSA_BO_CLGL 1 << 6 /* CLGL memory, imported from mesa GL */
/* allocated from KFD, but is scratch memory, do not need map and unmap in ioctrl */
#define VHSA_BO_SCRATCH 1 << 7
#define VHSA_BO_QUEUE 1 << 8
#define VHSA_BO_EVENT 1 << 9
#define VHSA_BO_SCRATCH_MAP 1 << 10
#define VHSA_SDMA_NONE UINT32_MAX
#define CHECK_VIRTIO_KFD_OPEN() \
do { \
if (dev_list == NULL) return HSAKMT_STATUS_KERNEL_IO_CHANNEL_NOT_OPENED; \
} while (0)
struct vhsakmt_node {
HsaNodeProperties node_props;
void* doorbell_base;
uint64_t scratch_start;
uint64_t scratch_size;
};
struct vhsakmt_device {
struct virtio_gpu_device* vgdev;
int refcount;
pthread_mutex_t bo_handles_mutex;
rbtree_t bo_rbt;
struct vhsakmt_bo* shmem_bo;
uint32_t reqbuf_max;
uint32_t next_blob_id;
uint64_t vm_start;
uint64_t vm_size;
pthread_mutex_t vhsakmt_mutex;
struct vhsakmt_node* vhsakmt_nodes;
HsaSystemProperties* sys_props;
};
struct vhsakmt_bo {
rbtree_node_t rbtn;
struct vhsakmt_device* dev;
int refcount;
unsigned size;
void* cpu_addr;
void* host_addr;
HsaMemFlags flags;
uint32_t bo_type;
uint32_t blob_id;
pthread_mutex_t map_mutex;
union {
struct {
uint32_t handle;
uint32_t res_id;
uint64_t offset;
uint64_t alloc_size;
int map_count;
} real;
};
vHsaEvent* event;
uint64_t queue_id;
vhsakmt_bo_handle rw_bo;
void* gl_meta_data;
};
/*hsakmt_virtio_memory.c*/
vhsakmt_bo_handle vhsakmt_entry_to_bo_handle(bo_entry e);
bo_entry vhsakmt_bo_handle_to_entry(vhsakmt_bo_handle bo);
void vhsakmt_insert_bo(vhsakmt_device_handle dev, vhsakmt_bo_handle bo, void* addr, uint64_t size);
void vhsakmt_remove_bo(vhsakmt_device_handle dev, vhsakmt_bo_handle bo);
vhsakmt_bo_handle vhsakmt_find_bo_by_addr(vhsakmt_device_handle dev, void* addr);
void* vhsakmt_gpu_va(vhsakmt_device_handle dev, void* va);
int vhsakmt_bo_cpu_unmap(vhsakmt_bo_handle bo);
int vhsakmt_bo_cpu_map(vhsakmt_bo_handle bo_handle, void** cpu, void* fixed_cpu);
int vhsakmt_create_mappable_blob_bo(vhsakmt_device_handle dev, size_t size, uint32_t blob_id,
uint32_t bo_type, void* va_handle,
vhsakmt_bo_handle* bo_handle);
int vhsakmt_bo_free(vhsakmt_device_handle dev, vhsakmt_bo_handle bo);
int vhsakmt_init_host_blob(vhsakmt_device_handle dev, size_t size, uint32_t blob_type,
uint32_t blob_flag, uint32_t blob_id, uint32_t bo_type, void* va_handle,
vhsakmt_bo_handle* bo_handle);
/*hsakmt_virtio_openclose.c*/
vhsakmt_device_handle vhsakmt_dev(void);
/*hsakmt_virtio_vm.c*/
void* vhsakmt_vm_start(void);
int vhsakmt_reserve_va(uint64_t start, uint64_t size);
void vhsakmt_dereserve_va(uint64_t start, uint64_t size);
void vhsakmt_set_scratch_area(vhsakmt_device_handle dev, uint32_t node, uint64_t start,
uint64_t size);
void vhsakmt_set_vm_area(vhsakmt_device_handle dev, uint64_t start, uint64_t size);
int vhsakmt_set_node_doorbell(vhsakmt_device_handle dev, uint32_t node, void* doorbell);
void* vhsakmt_node_doorbell(vhsakmt_device_handle dev, uint32_t node);
bool vhsakmt_is_scratch_mem(vhsakmt_device_handle dev, void* addr);
bool vhsakmt_is_userptr(vhsakmt_device_handle dev, void* addr);
/*hsakmt_virtio_device.c*/
int vhsakmt_execbuf_cpu(vhsakmt_device_handle dev, struct vhsakmt_ccmd_req* req, const char* from);
void* vhsakmt_alloc_rsp(vhsakmt_device_handle dev, struct vhsakmt_ccmd_req* req, uint32_t sz);
/*hsakmt_virtio_event.c*/
void* vhsakmt_event_host_handle(HsaEvent* h);
#ifdef __cplusplus
}
#endif
#endif
@@ -0,0 +1,189 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#include "hsakmt/hsakmt_virtio.h"
#include "hsakmt_virtio_device.h"
int vhsakmt_debug_level;
void* vhsakmt_event_host_handle(HsaEvent* h) { return (void*)((vHsaEvent*)h)->event_handle; }
static inline int32_t vhsakmt_event_res_id(HsaEvent* h) { return ((vHsaEvent*)h)->res_id; }
static inline vhsakmt_bo_handle vhsakmt_event_bo_handle(HsaEvent* h) {
return (vhsakmt_bo_handle)((vHsaEvent*)h)->bo_handle;
}
static int vhsakmt_create_event_blob_bo(vhsakmt_device_handle dev, size_t size, uint32_t blob_id,
vHsaEvent* vevent_handle, vhsakmt_bo_handle* bo_handle) {
int r;
r = vhsakmt_init_host_blob(dev, size, VIRTGPU_BLOB_MEM_HOST3D, 0, blob_id, VHSA_BO_EVENT,
(void*)vevent_handle->event_handle, bo_handle);
if (r) return r;
(*bo_handle)->event = vevent_handle;
vevent_handle->bo_handle = (uint64_t)(*bo_handle);
vevent_handle->res_id = (*bo_handle)->real.res_id;
vhsakmt_insert_bo(dev, *bo_handle, vevent_handle, size);
return r;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtCreateEvent(HsaEventDescriptor* EventDesc, _Bool ManualReset,
_Bool IsSignaled, HsaEvent** Event) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_event_rsp* rsp;
vhsakmt_bo_handle event_bo;
vHsaEvent* e;
int r;
struct vhsakmt_ccmd_event_req req = {
.hdr = VHSAKMT_CCMD(EVENT, sizeof(struct vhsakmt_ccmd_event_req)),
.type = VHSAKMT_CCMD_EVENT_CREATE,
.create_args.EventDesc = *EventDesc,
.create_args.ManualReset = ManualReset,
.create_args.IsSignaled = IsSignaled,
.blob_id = vhsakmt_atomic_inc_return(&dev->next_blob_id),
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_event_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (rsp->ret) return rsp->ret;
e = calloc(1, sizeof(vHsaEvent));
if (!e) return -ENOMEM;
memcpy(e, &rsp->vevent, sizeof(vHsaEvent));
r = vhsakmt_create_event_blob_bo(dev, sizeof(vHsaEvent), req.blob_id, e, &event_bo);
if (r) {
free(e);
return -ENOMEM;
}
*Event = (HsaEvent*)e;
vhsa_debug(
"%s: event addr: %p, hw123: %lx, %lx, %x, type: %d, id: %x, host handle: 0x%lx, res id: %d\n",
__FUNCTION__, e, e->event.EventData.HWData1, e->event.EventData.HWData2,
e->event.EventData.HWData3, e->event.EventData.EventType, e->event.EventId, e->event_handle,
event_bo->real.res_id);
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtDestroyEvent(HsaEvent* Event) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_bo* bo;
if (Event == NULL) return HSAKMT_STATUS_SUCCESS;
bo = vhsakmt_event_bo_handle(Event);
if (!bo) return HSAKMT_STATUS_SUCCESS;
return vhsakmt_bo_free(dev, bo);
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtSetEvent(HsaEvent* Event) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_event_rsp* rsp;
struct vhsakmt_ccmd_event_req req = {
.hdr = VHSAKMT_CCMD(EVENT, sizeof(struct vhsakmt_ccmd_event_req)),
.type = VHSAKMT_CCMD_EVENT_SET,
.event_hanele = vhsakmt_event_host_handle(Event),
.res_id = vhsakmt_event_res_id(Event),
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_event_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtResetEvent(HsaEvent* Event) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_event_rsp* rsp;
struct vhsakmt_ccmd_event_req req = {
.hdr = VHSAKMT_CCMD(EVENT, sizeof(struct vhsakmt_ccmd_event_req)),
.type = VHSAKMT_CCMD_EVENT_RESET,
.event_hanele = vhsakmt_event_host_handle(Event),
.res_id = vhsakmt_event_res_id(Event),
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_event_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtQueryEventState(HsaEvent* Event) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_event_rsp* rsp;
struct vhsakmt_ccmd_event_req req = {
.hdr = VHSAKMT_CCMD(EVENT, sizeof(struct vhsakmt_ccmd_event_req)),
.type = VHSAKMT_CCMD_EVENT_QUERY_STATE,
.event_hanele = vhsakmt_event_host_handle(Event),
.res_id = vhsakmt_event_res_id(Event),
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_event_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtWaitOnMultipleEvents(HsaEvent* Events[], HSAuint32 NumEvents,
bool WaitOnAll, HSAuint32 Milliseconds) {
return HSAKMT_STATUS_ERROR;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtWaitOnEvent(HsaEvent* Event, HSAuint32 Milliseconds) {
return vhsaKmtWaitOnMultipleEvents(&Event, 1, true, Milliseconds);
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtWaitOnEvent_Ext(HsaEvent* Event, HSAuint32 Milliseconds,
uint64_t* event_age) {
return vhsaKmtWaitOnMultipleEvents(&Event, 1, true, Milliseconds);
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtWaitOnMultipleEvents_Ext(HsaEvent* Events[], HSAuint32 NumEvents,
bool WaitOnAll, HSAuint32 Milliseconds,
uint64_t* event_age) {
return vhsaKmtWaitOnMultipleEvents(Events, NumEvents, WaitOnAll, Milliseconds);
}
@@ -0,0 +1,852 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#include "hsakmt/hsakmt_virtio.h"
#include "hsakmt_virtio_device.h"
#define VHSA_GL_METADATA_MAX_SIZE (0x50)
vhsakmt_bo_handle vhsakmt_entry_to_bo_handle(bo_entry e) { return (vhsakmt_bo_handle)e; }
bo_entry vhsakmt_bo_handle_to_entry(vhsakmt_bo_handle bo) { return &bo->rbtn; }
static inline bool vhsakmt_is_mem_bo(vhsakmt_bo_handle bo) { return (!bo->queue_id && !bo->event); }
static bool vhsakmt_mappable(HsaMemFlags flags) { return (!flags.ui32.Scratch); }
static bool vhsakmt_bo_mappable(vhsakmt_bo_handle bo) { return vhsakmt_mappable(bo->flags); }
void vhsakmt_insert_bo(vhsakmt_device_handle dev, vhsakmt_bo_handle bo, void* addr, uint64_t size) {
bo->rbtn.key.addr = (unsigned long)addr;
bo->rbtn.key.size = (unsigned long)size;
pthread_mutex_lock(&dev->bo_handles_mutex);
hsakmt_rbtree_insert(&dev->bo_rbt, &bo->rbtn);
pthread_mutex_unlock(&dev->bo_handles_mutex);
}
static void vhsakmt_remove_entry(vhsakmt_device_handle dev, bo_entry entry) {
if (!entry) return;
pthread_mutex_lock(&dev->bo_handles_mutex);
hsakmt_rbtree_delete(&dev->bo_rbt, entry);
pthread_mutex_unlock(&dev->bo_handles_mutex);
}
void vhsakmt_remove_bo(vhsakmt_device_handle dev, vhsakmt_bo_handle bo) {
bo_entry entry = vhsakmt_bo_handle_to_entry(bo);
if (entry->key.addr == 0 && entry->key.size == 0) return;
vhsakmt_remove_entry(dev, entry);
}
static bo_entry vhsakmt_rbt_search(vhsakmt_device_handle dev, void* addr) {
vhsakmt_bo_handle bo;
rbtree_key_t key = rbtree_key((uint64_t)addr, 0);
pthread_mutex_lock(&dev->bo_handles_mutex);
bo_entry n = rbtree_lookup_nearest(&dev->bo_rbt, &key, LKP_ADDR, RIGHT);
pthread_mutex_unlock(&dev->bo_handles_mutex);
if (n) {
bo = vhsakmt_entry_to_bo_handle(n);
if (bo->cpu_addr != addr) return NULL;
return n;
}
return NULL;
}
static bo_entry vhsakmt_find_entry_by_addr(vhsakmt_device_handle dev, void* addr) {
return vhsakmt_rbt_search(dev, addr);
}
vhsakmt_bo_handle vhsakmt_find_bo_by_addr(vhsakmt_device_handle dev, void* addr) {
bo_entry entry = vhsakmt_find_entry_by_addr(dev, addr);
if (entry) {
vhsakmt_bo_handle bo = vhsakmt_entry_to_bo_handle(entry);
if (!vhsakmt_is_mem_bo(bo)) return NULL;
return bo;
}
return NULL;
}
void* vhsakmt_gpu_va(vhsakmt_device_handle dev, void* va) {
if (!vhsakmt_is_userptr(dev, va)) return va;
bo_entry entry = vhsakmt_find_entry_by_addr(dev, va);
if (!entry) return NULL;
return vhsakmt_entry_to_bo_handle(entry)->host_addr;
}
int vhsakmt_bo_cpu_map(vhsakmt_bo_handle bo, void** cpu, void* fixed_cpu) {
int r;
if (!vhsakmt_bo_mappable(bo)) return 0;
pthread_mutex_lock(&bo->map_mutex);
if (!bo->cpu_addr) {
r = virtio_gpu_map_handle(bo->dev->vgdev, bo->real.handle, bo->size, cpu, fixed_cpu);
if (r) {
pthread_mutex_unlock(&bo->map_mutex);
return r;
}
bo->cpu_addr = *cpu;
atomic_fetch_add(&bo->real.map_count, 1);
}
pthread_mutex_unlock(&bo->map_mutex);
return *cpu == MAP_FAILED;
}
int vhsakmt_bo_cpu_unmap(vhsakmt_bo_handle bo) {
int r = 0;
if (!vhsakmt_bo_mappable(bo)) return 0;
pthread_mutex_lock(&bo->map_mutex);
if (!bo->cpu_addr || bo->real.map_count == 0) {
pthread_mutex_unlock(&bo->map_mutex);
return 0;
}
if (vhsakmt_atomic_dec_return(&bo->real.map_count) <= 0) {
if (bo->bo_type & VHSA_BO_KFD_MEM) {
virtio_gpu_unmap(bo->cpu_addr, bo->size);
vhsakmt_reserve_va(VHSA_VPTR_TO_UINT64(bo->cpu_addr), bo->size);
bo->cpu_addr = NULL;
}
}
pthread_mutex_unlock(&bo->map_mutex);
return r;
}
static int vhsakmt_destroy_handle(vhsakmt_device_handle dev, vhsakmt_bo_handle bo) {
int r = virtio_gpu_destroy_handle(dev->vgdev, bo->real.handle);
free(bo);
return r;
}
int vhsakmt_init_host_blob(vhsakmt_device_handle dev, size_t size, uint32_t blob_type,
uint32_t blob_flag, uint32_t blob_id, uint32_t bo_type, void* va_handle,
vhsakmt_bo_handle* bo_handle) {
int r;
vhsakmt_bo_handle bo;
struct drm_virtgpu_resource_create_blob args = {
.blob_mem = blob_type,
.size = size,
.blob_id = blob_id,
.blob_flags = blob_flag,
};
r = virtio_gpu_create_blob(dev->vgdev, &args);
if (r) return -EINVAL;
bo = calloc(1, sizeof(struct vhsakmt_bo));
if (!bo) {
virtio_gpu_destroy_handle(dev->vgdev, args.bo_handle);
return -ENOMEM;
}
bo->dev = dev;
bo->size = size;
bo->real.alloc_size = size;
bo->bo_type = bo_type;
bo->host_addr = va_handle;
pthread_mutex_init(&bo->map_mutex, NULL);
atomic_store(&bo->real.map_count, 0);
atomic_store(&bo->refcount, 1);
bo->real.handle = args.bo_handle;
virtio_gpu_res_id(dev->vgdev, bo->real.handle, &bo->real.res_id);
*bo_handle = bo;
return 0;
}
static int vhsakmt_init_userptr_blob(vhsakmt_device_handle dev, void* addr, size_t size,
vhsakmt_bo_handle* bo_handle, uint64_t* offset) {
int r;
struct drm_virtgpu_resource_create_blob args = {
.blob_mem = VIRTGPU_BLOB_MEM_HOST3D_GUEST,
.blob_flags = VIRTGPU_BLOB_FLAG_USE_USERPTR,
.size = size,
.blob_id = vhsakmt_atomic_inc_return(&dev->next_blob_id),
.blob_userptr = (uint64_t)addr,
};
r = virtio_gpu_create_blob(dev->vgdev, &args);
if (r < 0) return r;
vhsakmt_bo_handle userptr = calloc(1, sizeof(struct vhsakmt_bo));
if (!userptr) {
virtio_gpu_destroy_handle(dev->vgdev, args.bo_handle);
return -ENOMEM;
}
userptr->dev = dev;
userptr->size = size;
userptr->real.alloc_size = size;
userptr->bo_type = VHSA_BO_USERPTR;
userptr->cpu_addr = addr;
pthread_mutex_init(&userptr->map_mutex, NULL);
atomic_store(&userptr->real.map_count, 0);
atomic_store(&userptr->refcount, 1);
userptr->real.handle = args.bo_handle;
virtio_gpu_res_id(dev->vgdev, userptr->real.handle, &userptr->real.res_id);
*bo_handle = userptr;
*offset = args.offset;
return r;
}
int vhsakmt_create_mappable_blob_bo(vhsakmt_device_handle dev, size_t size, uint32_t blob_id,
uint32_t bo_type, void* va_handle,
vhsakmt_bo_handle* bo_handle) {
int r;
r = vhsakmt_init_host_blob(dev, size, VIRTGPU_BLOB_MEM_HOST3D, VIRTGPU_BLOB_FLAG_USE_MAPPABLE,
blob_id, bo_type, va_handle, bo_handle);
if (r) return r;
r = vhsakmt_bo_cpu_map(*bo_handle, &((*bo_handle)->cpu_addr), va_handle);
if (r) {
free(*bo_handle);
*bo_handle = NULL;
return -EINVAL;
}
if (va_handle && (va_handle != (*bo_handle)->cpu_addr))
vhsa_warn("%s: target map: %p != real map: %p\n", __FUNCTION__, va_handle,
(*bo_handle)->cpu_addr);
vhsakmt_insert_bo(dev, *bo_handle, (*bo_handle)->cpu_addr, (*bo_handle)->size);
return r;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtAllocMemory(HSAuint32 PreferredNode, HSAuint64 SizeInBytes,
HsaMemFlags MemFlags, void** MemoryAddress) {
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_memory_rsp* rsp;
vhsakmt_bo_handle bo;
int r;
struct vhsakmt_ccmd_memory_req req = {
.hdr = VHSAKMT_CCMD(MEMORY, sizeof(struct vhsakmt_ccmd_memory_req)),
.type = VHSAKMT_CCMD_MEMORY_ALLOC,
.blob_id = vhsakmt_atomic_inc_return(&dev->next_blob_id),
.alloc_args =
{
.PreferredNode = PreferredNode,
.SizeInBytes = SizeInBytes,
.MemFlags = MemFlags,
},
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (rsp->ret) return rsp->ret;
if (!rsp->memory_handle) return -ENOMEM;
r = vhsakmt_init_host_blob(dev, SizeInBytes, VIRTGPU_BLOB_MEM_HOST3D,
vhsakmt_mappable(MemFlags) ? VIRTGPU_BLOB_FLAG_USE_MAPPABLE : 0,
req.blob_id, VHSA_BO_KFD_MEM, (void*)rsp->memory_handle, &bo);
if (r) return r;
if (!vhsakmt_mappable(MemFlags)) {
bo->cpu_addr = bo->host_addr;
if (MemFlags.ui32.Scratch) {
vhsakmt_set_scratch_area(dev, PreferredNode, (uint64_t)bo->cpu_addr, SizeInBytes);
bo->bo_type |= VHSA_BO_SCRATCH;
}
} else {
r = vhsakmt_bo_cpu_map(bo, &bo->cpu_addr, bo->host_addr);
if (r) {
free(bo);
return -ENOMEM;
}
}
if (!MemFlags.ui32.Scratch) vhsakmt_insert_bo(dev, bo, bo->cpu_addr, bo->size);
*MemoryAddress = bo->cpu_addr;
vhsa_debug("alloc mem addr: %p, host addr: %p, size: %lx, res-id: %d, handble: %d\n",
*MemoryAddress, bo->host_addr, SizeInBytes, bo->real.res_id, bo->real.handle);
return rsp->ret;
}
int vhsakmt_bo_free(vhsakmt_device_handle dev, vhsakmt_bo_handle bo) {
bo_entry entry;
int r;
if (vhsakmt_atomic_dec_return(&bo->refcount) > 0) return 0;
entry = vhsakmt_bo_handle_to_entry(bo);
if (entry->key.addr == 0 && entry->key.size == 0) return -EINVAL;
/* do not free BOs of queue, let them be freed with queue */
if (bo->bo_type & VHSA_BO_QUEUE_DOORBELL) {
vhsa_err("%s: Try to free VHSA_BO_QUEUE_DOORBELL memory: %p\n", __FUNCTION__, bo->cpu_addr);
return 0;
}
vhsakmt_remove_bo(dev, bo);
if (bo->cpu_addr) vhsakmt_bo_cpu_unmap(bo);
if (bo->event) free(bo->event);
if (bo->gl_meta_data) free(bo->gl_meta_data);
pthread_mutex_destroy(&bo->map_mutex);
r = vhsakmt_destroy_handle(dev, bo);
return r;
}
/* Only remove bo in rbtree */
static void vhsakmt_remove_userptr_bo(vhsakmt_device_handle dev, vhsakmt_bo_handle bo) {
vhsakmt_remove_bo(dev, bo);
free(bo);
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtFreeMemory(void* MemoryAddress, HSAuint64 SizeInBytes) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
vhsakmt_bo_handle bo = vhsakmt_find_bo_by_addr(dev, MemoryAddress);
if (!bo) return HSAKMT_STATUS_SUCCESS;
vhsa_debug("%s: addr: %p, size: %lx, res_id: %d\n", __FUNCTION__, MemoryAddress, SizeInBytes,
bo->real.res_id);
return vhsakmt_bo_free(dev, bo);
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtMapMemoryToGPUNodes(void* MemoryAddress, HSAuint64 MemorySizeInBytes,
HSAuint64* AlternateVAGPU,
HsaMemMapFlags MemMapFlags,
HSAuint64 NumberOfNodes, HSAuint32* NodeArray) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
size_t req_len =
VHSA_ALIGN_UP(sizeof(struct vhsakmt_ccmd_memory_req) + NumberOfNodes * sizeof(*NodeArray), 8);
struct vhsakmt_ccmd_memory_req* req;
struct vhsakmt_ccmd_memory_rsp* rsp;
vhsakmt_bo_handle bo;
req = (void*)calloc(1, req_len);
if (!req) return -ENOMEM;
req->hdr = VHSAKMT_CCMD(MEMORY, req_len);
req->type = VHSAKMT_CCMD_MEMORY_MAP_TO_GPU_NODES;
req->map_to_GPU_nodes_args.MemorySizeInBytes = MemorySizeInBytes;
req->map_to_GPU_nodes_args.MemMapFlags = MemMapFlags;
req->map_to_GPU_nodes_args.NumberOfNodes = NumberOfNodes;
bo = vhsakmt_find_bo_by_addr(dev, MemoryAddress);
if (bo) {
req->map_to_GPU_nodes_args.MemoryAddress = (uint64_t)bo->host_addr;
if (bo->bo_type & VHSA_BO_USERPTR) vhsakmt_remove_userptr_bo(dev, bo);
} else
req->map_to_GPU_nodes_args.MemoryAddress = (uint64_t)MemoryAddress;
memcpy(req->payload, NodeArray, NumberOfNodes * sizeof(*NodeArray));
rsp = vhsakmt_alloc_rsp(dev, &req->hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) {
free(req);
return -ENOMEM;
}
vhsakmt_execbuf_cpu(dev, &req->hdr, __FUNCTION__);
*AlternateVAGPU = rsp->alternate_vagpu;
vhsa_debug("%s: gva: %p, hva: 0x%lx, size: %lx, AlternateVAGPU: %lx, ret: %d\n", __FUNCTION__,
MemoryAddress, req->map_to_GPU_nodes_args.MemoryAddress, MemorySizeInBytes,
*AlternateVAGPU, rsp->ret);
free(req);
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtUnmapMemoryToGPU(void* MemoryAddress) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
vhsakmt_bo_handle bo = vhsakmt_find_bo_by_addr(dev, MemoryAddress);
if (!bo) return HSAKMT_STATUS_SUCCESS;
struct vhsakmt_ccmd_memory_rsp* rsp;
struct vhsakmt_ccmd_memory_req req = {
.hdr = VHSAKMT_CCMD(MEMORY, sizeof(struct vhsakmt_ccmd_memory_req)),
.type = VHSAKMT_CCMD_MEMORY_UNMAP_TO_GPU,
.MemoryAddress = (uint64_t)bo->host_addr,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
vhsa_debug("%s: gva: %p, hva: 0x%lx\n", __FUNCTION__, MemoryAddress, req.MemoryAddress);
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtAvailableMemory(HSAuint32 Node, HSAuint64* AvailableBytes) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_memory_rsp* rsp;
struct vhsakmt_ccmd_memory_req req = {
.hdr = VHSAKMT_CCMD(MEMORY, sizeof(struct vhsakmt_ccmd_memory_req)),
.type = VHSAKMT_CCMD_MEMORY_AVAIL_MEM,
.Node = Node,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
*AvailableBytes = rsp->available_bytes;
return rsp->ret;
}
static int vhsakmt_create_scratch_map_memory(vhsakmt_device_handle dev, void* MemoryAddress,
HSAuint64 MemorySizeInBytes,
HSAuint64* AlternateVAGPU) {
vhsakmt_bo_handle out;
int r;
struct vhsakmt_ccmd_memory_req req = {
.hdr = VHSAKMT_CCMD(MEMORY, sizeof(struct vhsakmt_ccmd_memory_req)),
.type = VHSAKMT_CCMD_MEMORY_MAP_MEM_TO_GPU,
.blob_id = vhsakmt_atomic_inc_return(&dev->next_blob_id),
.map_to_GPU_args =
{
.MemoryAddress = (uint64_t)MemoryAddress,
.MemorySizeInBytes = MemorySizeInBytes,
.need_create_bo = true,
},
};
struct vhsakmt_ccmd_memory_rsp* rsp =
vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (rsp->ret) return rsp->ret;
r = vhsakmt_init_host_blob(dev, MemorySizeInBytes, VIRTGPU_BLOB_MEM_HOST3D, 0, req.blob_id,
VHSA_BO_SCRATCH_MAP, NULL, &out);
if (r) return r;
// TODO: insert scratch bo into rbtree, or insert it in dev nodes.
out->cpu_addr = MemoryAddress;
out->host_addr = (void*)rsp->memory_handle;
*AlternateVAGPU = rsp->alternate_vagpu;
vhsa_debug(
"%s: create scratch memory, gva: %p, memory_handle: 0x%p, alternate_vagpu: %p, size: %lx\n",
__FUNCTION__, MemoryAddress, (void*)rsp->memory_handle, (void*)rsp->alternate_vagpu,
MemorySizeInBytes);
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtMapMemoryToGPU(void* MemoryAddress, HSAuint64 MemorySizeInBytes,
HSAuint64* AlternateVAGPU) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_memory_rsp* rsp;
vhsakmt_bo_handle bo = vhsakmt_find_bo_by_addr(dev, MemoryAddress);
if (!bo && vhsakmt_is_scratch_mem(dev, MemoryAddress))
return vhsakmt_create_scratch_map_memory(dev, MemoryAddress, MemorySizeInBytes, AlternateVAGPU);
struct vhsakmt_ccmd_memory_req req = {
.hdr = VHSAKMT_CCMD(MEMORY, sizeof(struct vhsakmt_ccmd_memory_req)),
.type = VHSAKMT_CCMD_MEMORY_MAP_MEM_TO_GPU,
.map_to_GPU_args =
{
.MemoryAddress = bo ? (uint64_t)bo->host_addr : (uint64_t)MemoryAddress,
.MemorySizeInBytes = MemorySizeInBytes,
},
};
if (bo && (bo->bo_type & VHSA_BO_USERPTR)) vhsakmt_remove_userptr_bo(dev, bo);
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
vhsa_debug("%s: gva: %p, hva: 0x%lx, size: %lx\n", __FUNCTION__, MemoryAddress, req.MemoryAddress,
MemorySizeInBytes);
*AlternateVAGPU = rsp->alternate_vagpu;
return rsp->ret;
}
static int vhsakmt_map_userptr(vhsakmt_device_handle dev, void* addr, size_t size, uint32_t res_id,
uint64_t* userptr_handle) {
struct vhsakmt_ccmd_memory_req req = {
.hdr = VHSAKMT_CCMD(MEMORY, sizeof(struct vhsakmt_ccmd_memory_req)),
.type = VHSAKMT_CCMD_MEMORY_MAP_USERPTR,
.res_id = res_id,
};
struct vhsakmt_ccmd_memory_rsp* rsp =
vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) return -ENOMEM;
rsp->map_userptr_rsp.userptr_handle = 0;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
*userptr_handle = rsp->map_userptr_rsp.userptr_handle;
return rsp->ret;
}
static void* vhsakmt_map_to_gpu(void* addr, size_t size) {
vhsakmt_device_handle dev = vhsakmt_dev();
size_t offset = (uint64_t)addr % getpagesize();
size_t map_size = (VHSA_ALIGN_UP(size + offset, getpagesize()) / getpagesize()) * getpagesize();
uint64_t userptr_offset, userptr_handle = 0;
vhsakmt_bo_handle userptr;
int r;
vhsa_debug("%s: addr: %p, size: 0x%lx, size + offset: 0x%lx, map_size: 0x%lx\n", __FUNCTION__,
addr, size, size + offset, map_size);
r = vhsakmt_init_userptr_blob(dev, addr, size, &userptr, &userptr_offset);
if (r < 0) {
vhsa_debug("%s: userptr create failed at address: %p, ret = %d\n", __FUNCTION__, addr, r);
return NULL;
}
vhsakmt_map_userptr(dev, addr, size, userptr->real.res_id, &userptr_handle);
if (!userptr_handle) {
vhsa_debug("%s: map userptr failed at address: %p, ret = %d\n", __FUNCTION__, addr, r);
vhsakmt_destroy_handle(dev, userptr);
vhsakmt_remove_userptr_bo(dev, userptr);
return NULL;
}
userptr->host_addr = VHSA_UINT64_TO_VPTR(VHSA_VPTR_TO_UINT64(userptr_handle) + offset);
if (r > 0) {
vhsa_debug("%s: userptr: %p already registered, offset: %lx\n", __FUNCTION__, addr,
userptr_offset);
userptr->host_addr =
VHSA_UINT64_TO_VPTR(VHSA_VPTR_TO_UINT64(userptr->host_addr) + userptr_offset);
}
vhsakmt_insert_bo(dev, userptr, userptr->cpu_addr, userptr->size);
vhsa_debug("%s: real gva: %p, gva: %p, hva: %p, size: %lx, offset: %" PRIu64
", map_size: 0x%lx\n",
__FUNCTION__, addr, userptr->cpu_addr, userptr->host_addr, size, offset, map_size);
return userptr->host_addr;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtRegisterMemoryWithFlags(void* MemoryAddress,
HSAuint64 MemorySizeInBytes,
HsaMemFlags MemFlags) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_memory_rsp* rsp;
void* addr;
struct vhsakmt_ccmd_memory_req req = {
.hdr = VHSAKMT_CCMD(MEMORY, sizeof(struct vhsakmt_ccmd_memory_req)),
.type = VHSAKMT_CCMD_MEMORY_REG_MEM_WITH_FLAG,
.reg_mem_with_flag =
{
.MemorySizeInBytes = MemorySizeInBytes,
.MemFlags = MemFlags,
},
};
/* no need to register memory from lihsakmt / not a userptr */
if (!vhsakmt_is_userptr(dev, MemoryAddress)) return HSAKMT_STATUS_SUCCESS;
addr = vhsakmt_map_to_gpu(MemoryAddress, MemorySizeInBytes);
if (!addr) {
vhsa_debug("%s: register memory failed, gva: %p, size: %lx\n", __FUNCTION__, MemoryAddress,
MemorySizeInBytes);
return HSAKMT_STATUS_ERROR;
}
req.reg_mem_with_flag.MemoryAddress = (uint64_t)addr;
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
return rsp->ret;
}
static int vhsakmt_remove_clgl_bo(vhsakmt_device_handle dev, vhsakmt_bo_handle bo) {
struct vhsakmt_ccmd_memory_rsp* rsp;
struct vhsakmt_ccmd_memory_req req = {
.hdr = VHSAKMT_CCMD(MEMORY, sizeof(struct vhsakmt_ccmd_memory_req)),
.type = VHSAKMT_CCMD_MEMORY_DEREG_MEM,
.res_id = bo->real.res_id,
.MemoryAddress = (uint64_t)bo->cpu_addr,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_memory_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (rsp->ret) vhsa_err("%s: deregister failed clgl memory gva: %p\n", __FUNCTION__, bo->cpu_addr);
vhsakmt_bo_free(dev, bo);
vhsa_debug("%s: deregister clgl memory gva: %p, ret: %d\n", __FUNCTION__, bo->cpu_addr, rsp->ret);
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtDeregisterMemory(void* MemoryAddress) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
vhsakmt_bo_handle bo = vhsakmt_find_bo_by_addr(dev, MemoryAddress);
if (!bo) return HSAKMT_STATUS_SUCCESS;
vhsa_debug("%s: remove userptr %p size: 0x%lx, res id: %d\n", __FUNCTION__, MemoryAddress,
(size_t)bo->size, bo->real.res_id);
if (bo->bo_type & VHSA_BO_CLGL)
return vhsakmt_remove_clgl_bo(dev, bo);
else {
vhsakmt_remove_bo(dev, bo);
free(bo);
}
return 0;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtQueryPointerInfo(const void* Pointer, HsaPointerInfo* PointerInfo) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
void* gpu_va = vhsakmt_gpu_va(dev, VHSA_UINT64_TO_VPTR(Pointer));
if (!gpu_va) return -HSAKMT_STATUS_ERROR;
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_POINTER_INFO,
.pointer = VHSA_VPTR_TO_UINT64(gpu_va),
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr,
sizeof(struct vhsakmt_ccmd_query_info_rsp) +
QUERY_PTR_INFO_MAX_MAPPED_NODES * sizeof(uint32_t));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
memcpy(PointerInfo, &rsp->ptr_info, sizeof(HsaPointerInfo));
if (PointerInfo->NMappedNodes && PointerInfo->MappedNodes) {
if (PointerInfo->NMappedNodes > QUERY_PTR_INFO_MAX_MAPPED_NODES) {
PointerInfo->NMappedNodes = QUERY_PTR_INFO_MAX_MAPPED_NODES;
vhsa_debug(
"%s: query pointer: %p info mapped nodes greater than QUERY_PTR_INFO_MAX_MAPPED_NODES\n",
__FUNCTION__, Pointer);
}
PointerInfo->MappedNodes = calloc(PointerInfo->NMappedNodes, sizeof(uint32_t));
if (!PointerInfo->MappedNodes) {
PointerInfo->NMappedNodes = 0;
return -HSAKMT_STATUS_NO_MEMORY;
}
memcpy(VHSA_UINT64_TO_VPTR(PointerInfo->MappedNodes), rsp->payload,
PointerInfo->NMappedNodes * sizeof(uint32_t));
}
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetTileConfig(HSAuint32 NodeId, HsaGpuTileConfig* config) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
uint8_t* config_cpy_addr = NULL;
struct vhsakmt_ccmd_query_info_rsp* rsp;
unsigned req_len = sizeof(struct vhsakmt_ccmd_query_info_req);
unsigned rsp_len = sizeof(struct vhsakmt_ccmd_query_info_rsp) +
config->NumTileConfigs * sizeof(HSAuint32) + config->NumMacroTileConfigs * sizeof(HSAuint32);
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, req_len),
.type = VHSAKMT_CCMD_QUERY_TILE_CONFIG,
.tile_config_args.NodeId = NodeId,
.tile_config_args.config = *config,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, rsp_len);
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
memcpy(config, &rsp->tile_config_rsp, sizeof(HsaGpuTileConfig));
config_cpy_addr = ((uint8_t*)rsp->payload);
memcpy(config->TileConfig, config_cpy_addr, config->NumTileConfigs * sizeof(HSAuint32));
config_cpy_addr += config->NumTileConfigs * sizeof(HSAuint32);
memcpy(config->MacroTileConfig, config_cpy_addr, config->NumMacroTileConfigs * sizeof(HSAuint32));
return rsp->ret;
}
static int vhsakmt_create_clgl_bo(vhsakmt_device_handle dev, void* addr, size_t size,
uint32_t res_id, uint32_t bo_handle, void* meta_data) {
vhsakmt_bo_handle out = calloc(1, sizeof(struct vhsakmt_bo));
if (!out) return -ENOMEM;
out->dev = dev;
out->size = size;
atomic_store(&out->real.map_count, 0);
atomic_store(&out->refcount, 1);
#ifdef CLGL_EXPORT_RESID
out->real.res_id = GraphicsResourceHandle;
#else
out->real.res_id = res_id;
#endif
/* GL bo handle from GL context*/
out->real.handle = bo_handle;
out->bo_type |= VHSA_BO_CLGL;
if (meta_data) out->gl_meta_data = meta_data;
out->host_addr = addr;
vhsakmt_insert_bo(dev, out, addr, out->size);
return 0;
}
static int vhsakmt_gfxhandle_to_resid(vhsakmt_device_handle dev, uint32_t gfx_handle,
uint32_t* res_id, uint32_t* bo_handle) {
int r = drmPrimeFDToHandle(dev->vgdev->fd, gfx_handle, bo_handle);
if (r) {
vhsa_err("%s: drmPrimeFDToHandle failed for handle: %u\n", __FUNCTION__, gfx_handle);
return r;
}
virtio_gpu_res_id(dev->vgdev, *bo_handle, res_id);
vhsa_debug("%s: register praphics handle: handle: %d, bo_handle: %d, res_id: %d\n", __FUNCTION__,
gfx_handle, *bo_handle, *res_id);
return 0;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtRegisterGraphicsHandleToNodes(
HSAuint64 GraphicsResourceHandle, HsaGraphicsResourceInfo* GraphicsResourceInfo,
HSAuint64 NumberOfNodes, HSAuint32* NodeArray) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
uint32_t bo_handle, res_id;
uint64_t meta_data_size = VHSA_GL_METADATA_MAX_SIZE;
unsigned req_len = sizeof(struct vhsakmt_ccmd_gl_inter_req) + NumberOfNodes * sizeof(NodeArray);
struct vhsakmt_ccmd_gl_inter_req* req;
struct vhsakmt_ccmd_gl_inter_rsp* rsp;
int r;
req = calloc(1, req_len);
if (!req) return -ENOMEM;
req->hdr = VHSAKMT_CCMD(GL_INTER, req_len);
req->type = VHSAKMT_CCMD_GL_REG_GHD_TO_NODES;
req->reg_ghd_to_nodes.NumberOfNodes = NumberOfNodes;
req->reg_ghd_to_nodes.res_handle = GraphicsResourceHandle;
#ifdef CLGL_EXPORT_RESID
req->reg_ghd_to_nodes.GraphicsResourceHandle = GraphicsResourceHandle;
#else
r = vhsakmt_gfxhandle_to_resid(dev, GraphicsResourceHandle, &res_id, &bo_handle);
if (r) return r;
req->reg_ghd_to_nodes.GraphicsResourceHandle = bo_handle;
req->reg_ghd_to_nodes.res_handle = res_id;
#endif
memcpy(req->payload, NodeArray, NumberOfNodes * sizeof(NodeArray));
rsp =
vhsakmt_alloc_rsp(dev, &req->hdr, sizeof(struct vhsakmt_ccmd_gl_inter_rsp) + meta_data_size);
if (!rsp) {
r = -ENOMEM;
goto free_out;
}
vhsakmt_execbuf_cpu(dev, &req->hdr, __FUNCTION__);
if (rsp->ret) return rsp->ret;
memcpy(GraphicsResourceInfo, &rsp->info, sizeof(HsaGraphicsResourceInfo));
if (rsp->info.MetadataSizeInBytes) {
GraphicsResourceInfo->Metadata = calloc(1, GraphicsResourceInfo->MetadataSizeInBytes);
if (!GraphicsResourceInfo->Metadata) {
r = -ENOMEM;
goto free_out;
}
memcpy(VHSA_UINT64_TO_VPTR(GraphicsResourceInfo->Metadata), rsp->payload,
GraphicsResourceInfo->MetadataSizeInBytes);
} else
GraphicsResourceInfo->Metadata = NULL;
vhsa_debug("%s: register graphics handle: handle: %ld hva: %p, size: %lx\n", __FUNCTION__,
GraphicsResourceHandle, GraphicsResourceInfo->MemoryAddress,
GraphicsResourceInfo->SizeInBytes);
r = vhsakmt_create_clgl_bo(dev, GraphicsResourceInfo->MemoryAddress,
GraphicsResourceInfo->SizeInBytes, res_id, bo_handle,
VHSA_UINT64_TO_VPTR(GraphicsResourceInfo->Metadata));
if (r) goto free_out;
r = rsp->ret;
free_out:
/* close exported FD after register or close it when deregistre. Close after register here. */
close(GraphicsResourceHandle);
free(req);
return r;
}
@@ -0,0 +1,132 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#include "hsakmt/hsakmt_virtio.h"
#include "hsakmt_virtio_device.h"
pthread_mutex_t dev_mutex = PTHREAD_MUTEX_INITIALIZER;
vhsakmt_device_handle dev_list = NULL;
vhsakmt_device_handle vhsakmt_dev(void) { return dev_list; }
static HSAKMT_STATUS vhsakmt_openKFD_cmd(vhsakmt_device_handle dev) {
void* vm_start = vhsakmt_vm_start();
if (!vm_start) return -HSAKMT_STATUS_NO_MEMORY;
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_OPEN_KFD,
.open_kfd_args =
{
.cur_vm_start = VHSA_VPTR_TO_UINT64(vm_start),
},
};
if (!req.open_kfd_args.cur_vm_start) {
vhsa_err("%s: failed to get current heap start address\n", __FUNCTION__);
return -HSAKMT_STATUS_ERROR;
}
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -HSAKMT_STATUS_NO_MEMORY;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (!rsp->open_kfd_rsp.vm_start || !rsp->open_kfd_rsp.vm_size) {
vhsa_err("%s: failed to get KFD VM area\n", __FUNCTION__);
return -HSAKMT_STATUS_ERROR;
}
vhsakmt_set_vm_area(dev, rsp->open_kfd_rsp.vm_start, rsp->open_kfd_rsp.vm_size);
if (vhsakmt_reserve_va(dev->vm_start, dev->vm_size)) {
vhsa_err("%s: failed to reserve VM area: [%lx-%lx]-0x%lx\n", __FUNCTION__, dev->vm_start,
dev->vm_start + dev->vm_size, dev->vm_size);
return -HSAKMT_STATUS_NO_MEMORY;
}
vhsa_debug("%s: kfd vm range: [%lx-%lx]-0x%lx\n", __FUNCTION__, dev->vm_start,
dev->vm_start + dev->vm_size, dev->vm_size);
return rsp->ret;
}
static vhsakmt_device_handle vhsakmt_device_init(void) {
int fd;
vhsakmt_device_handle dev = NULL;
if (vhsakmt_dev()) return vhsakmt_dev();
pthread_mutex_lock(&dev_mutex);
fd = virtio_gpu_kfd_open();
if (fd < 0) goto open_failed;
dev = calloc(1, sizeof(struct vhsakmt_device));
if (!dev) goto open_failed;
dev->vgdev = virtio_gpu_init(fd, 0);
if (!dev->vgdev) goto malloc_failed;
rbtree_init(&dev->bo_rbt);
atomic_store(&dev->next_blob_id, 1);
atomic_store(&dev->refcount, 1);
pthread_mutex_init(&dev->bo_handles_mutex, NULL);
pthread_mutex_init(&dev->vhsakmt_mutex, NULL);
dev_list = dev;
pthread_mutex_unlock(&dev_mutex);
return dev;
malloc_failed:
free(dev);
dev = NULL;
open_failed:
pthread_mutex_unlock(&dev_mutex);
return dev;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtOpenKFD(void) {
vhsakmt_device_handle dev;
char* d = getenv("VHSAKMT_DEBUG_LEVEL");
if (d) vhsakmt_debug_level = atoi(d);
dev = vhsakmt_device_init();
if (!dev) return HSAKMT_STATUS_ERROR;
return vhsakmt_openKFD_cmd(vhsakmt_dev());
}
static void vhsakmt_device_destroy(struct vhsakmt_device* dev) {
pthread_mutex_destroy(&dev->bo_handles_mutex);
vhsakmt_dereserve_va(dev->vm_start, dev->vm_size);
if (dev->sys_props) free(dev->sys_props);
if (dev->vhsakmt_nodes) free(dev->vhsakmt_nodes);
virtio_gpu_close(dev->vgdev);
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtCloseKFD(void) {
vhsakmt_device_handle dev = vhsakmt_dev();
pthread_mutex_lock(&dev_mutex);
if (vhsakmt_atomic_dec_return(&dev->refcount) <= 0) vhsakmt_device_destroy(dev);
pthread_mutex_unlock(&dev_mutex);
return 0;
}
@@ -0,0 +1,503 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#ifndef VHSAKMT_VIRTIO_PROTO_H
#define VHSAKMT_VIRTIO_PROTO_H
#include "hsakmt/linux/kfd_ioctl.h"
#include "hsakmt/hsakmt.h"
#include <drm/amdgpu_drm.h>
#include <libdrm/amdgpu.h>
#include <stdint.h>
#include "virtio_gpu.h"
#ifdef __GNUC__
#pragma GCC diagnostic push
#pragma GCC diagnostic error "-Wpadded"
#endif
/* defined in other header file in virglrenderer */
#define VHSAKMT_DEFINE_CAST(parent, child) \
static inline struct child* to_##child(struct parent* x) { return (struct child*)x; }
#if defined(__STDC_VERSION__) && __STDC_VERSION__ >= 201112L
#define VHSAKMT_STATIC_ASSERT_SIZE(t) \
static_assert(sizeof(struct t) % 8 == 0, "sizeof(struct " #t ") not multiple of 8"); \
static_assert(_Alignof(struct t) <= 8, "alignof(struct " #t ") too large");
#else
#define VHSAKMT_STATIC_ASSERT_SIZE(t)
#endif
enum vhsakmt_ccmd {
VHSAKMT_CCMD_NOP = 1, /* No payload, can be used to sync with host */
VHSAKMT_CCMD_QUERY_INFO,
VHSAKMT_CCMD_EVENT,
VHSAKMT_CCMD_MEMORY,
VHSAKMT_CCMD_QUEUE,
VHSAKMT_CCMD_GL_INTER,
};
typedef struct _vHsaEvent {
HsaEvent event;
uint64_t event_handle;
uint64_t bo_handle;
uint32_t res_id;
uint32_t pad;
} vHsaEvent;
VHSAKMT_STATIC_ASSERT_SIZE(_vHsaEvent)
struct vhsakmt_event_shmem {
uint32_t trigered_events_num;
uint32_t pad;
HsaEvent trigered_events[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_event_shmem)
#define VHSAKMT_CCMD(_cmd, _len) \
((struct vhsakmt_ccmd_req){ \
.cmd = VHSAKMT_CCMD_##_cmd, \
.len = (_len), \
})
struct vhsakmt_ccmd_nop_req {
struct vhsakmt_ccmd_req hdr;
};
/*
* VHSAKMT_CCMD_QUERY
*/
enum vhsakmt_ccmd_query_type {
VHSAKMT_CCMD_QUERY_GPU_INFO = 0,
VHSAKMT_CCMD_QUERY_OPEN_KFD,
VHSAKMT_CCMD_QUERY_GET_VER,
VHSAKMT_CCMD_QUERY_REL_SYS_PROP,
VHSAKMT_CCMD_QUERY_GET_SYS_PROP,
VHSAKMT_CCMD_QUERY_GET_NODE_PROP,
VHSAKMT_CCMD_QUERY_GET_XNACK_MODE,
VHSAKMT_CCMD_QUERY_RUN_TIME_ENABLE,
VHSAKMT_CCMD_QUERY_RUN_TIME_DISABLE,
VHSAKMT_CCMD_QUERY_GET_NOD_MEM_PROP,
VHSAKMT_CCMD_QUERY_GET_NOD_CACHE_PROP,
VHSAKMT_CCMD_QUERY_GET_NOD_IO_LINK_PROP,
VHSAKMT_CCMD_QUERY_GET_CLOCK_COUNTERS,
VHSAKMT_CCMD_QUERY_POINTER_INFO,
VHSAKMT_CCMD_QUERY_TILE_CONFIG,
VHSAKMT_CCMD_QUERY_NANO_TIME,
VHSAKMT_CCMD_QUERY_GET_RUNTIME_CAPS,
};
#define QUERY_PTR_INFO_MAX_MAPPED_NODES 3
typedef struct _query_req_run_time_enable_args {
/* void* rDebug, bypassed by payload */
uint8_t pad[3];
uint8_t setupTtmp;
uint32_t __pad;
} query_req_run_time_enable_args;
VHSAKMT_STATIC_ASSERT_SIZE(_query_req_run_time_enable_args)
typedef struct _query_req_node_mem_prop_args {
uint32_t NodeId;
uint32_t NumBanks;
} query_req_node_mem_prop_args;
VHSAKMT_STATIC_ASSERT_SIZE(_query_req_node_mem_prop_args)
typedef struct _query_req_node_cache_prop_args {
uint32_t NodeId;
uint32_t ProcessorId;
uint32_t NumCaches;
uint32_t pad;
} query_req_node_cache_prop_args;
VHSAKMT_STATIC_ASSERT_SIZE(_query_req_node_cache_prop_args)
typedef struct _query_req_node_io_link_args {
uint32_t NodeId;
uint32_t NumIoLinks;
} query_req_node_io_link_args;
VHSAKMT_STATIC_ASSERT_SIZE(_query_req_node_io_link_args)
typedef struct _query_tile_config {
HsaGpuTileConfig config;
uint32_t NodeId;
uint32_t pad;
} query_tile_config;
VHSAKMT_STATIC_ASSERT_SIZE(_query_tile_config)
typedef struct _query_open_kfd_args {
uint64_t cur_vm_start;
} query_open_kfd_args;
VHSAKMT_STATIC_ASSERT_SIZE(_query_open_kfd_args)
typedef struct _query_open_kfd_rsp {
uint64_t vm_start;
uint64_t vm_size;
} query_open_kfd_rsp;
VHSAKMT_STATIC_ASSERT_SIZE(_query_open_kfd_rsp)
typedef struct _query_nano_time_rsp {
uint64_t nano_time;
} query_nano_time_rsp;
VHSAKMT_STATIC_ASSERT_SIZE(_query_nano_time_rsp)
struct vhsakmt_ccmd_query_info_req {
struct vhsakmt_ccmd_req hdr;
struct drm_amdgpu_info info;
uint32_t type;
uint32_t pad;
union {
uint64_t pointer;
uint32_t NodeID; /* some query API just need node ID */
query_req_run_time_enable_args run_time_enable_args;
query_req_node_mem_prop_args node_mem_prop_args;
query_req_node_cache_prop_args node_cache_prop_args;
query_req_node_io_link_args node_io_link_args;
query_tile_config tile_config_args;
query_open_kfd_args open_kfd_args;
};
uint8_t payload[];
};
VHSAKMT_DEFINE_CAST(vhsakmt_ccmd_req, vhsakmt_ccmd_query_info_req)
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_query_info_req)
#define VHSAKMT_CCMD_QUERY_MAX_TILE_CONFIG 128
#define VHSAKMT_CCMD_QUERY_MAX_GET_NOD_MEM_PROP 128
#define VHSAKMT_CCMD_QUERY_MAX_GET_NOD_CACHE_PROP 128
#define VHSAKMT_CCMD_QUERY_MAX_GET_NOD_IO_LINK_PROP 128
struct vhsakmt_ccmd_query_info_rsp {
struct vhsakmt_ccmd_rsp hdr;
int32_t ret;
union {
query_open_kfd_rsp open_kfd_rsp;
query_nano_time_rsp nano_time_rsp;
HsaGpuTileConfig tile_config_rsp;
HsaPointerInfo ptr_info;
struct amdgpu_gpu_info gpu_info;
HsaVersionInfo kfd_version;
HsaSystemProperties sys_props;
HsaNodeProperties node_props;
int32_t xnack_mode;
HsaClockCounters clock_counters;
uint32_t caps;
uint64_t pad[9];
};
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_query_info_rsp)
/*
* VHSAKMT_CCMD_EVENT
*/
enum vhsakmt_ccmd_event_type {
VHSAKMT_CCMD_EVENT_CREATE,
VHSAKMT_CCMD_EVENT_DESTROY,
VHSAKMT_CCMD_EVENT_SET,
VHSAKMT_CCMD_EVENT_RESET,
VHSAKMT_CCMD_EVENT_QUERY_STATE,
VHSAKMT_CCMD_EVENT_WAIT_ON_MULTI_EVENTS,
VHSAKMT_CCMD_EVENT_SET_TRAP,
};
typedef struct _event_req_create_args {
HsaEventDescriptor EventDesc;
uint8_t ManualReset;
uint8_t IsSignaled;
uint8_t pad[6];
} event_req_create_args;
VHSAKMT_STATIC_ASSERT_SIZE(_event_req_create_args)
typedef struct _event_req_wait_args {
HsaEvent Event;
uint32_t Milliseconds;
uint32_t pad;
} event_req_wait_args;
VHSAKMT_STATIC_ASSERT_SIZE(_event_req_wait_args)
typedef struct _event_req_wait_ext_args {
HsaEvent Event;
uint64_t event_age;
uint32_t Milliseconds;
uint32_t pad;
} event_req_wait_ext_args;
VHSAKMT_STATIC_ASSERT_SIZE(_event_req_wait_ext_args)
typedef struct _event_req_wait_on_multi_args {
/*HsaEvent* Events[], in playloud*/
uint32_t NumEvents;
uint32_t Milliseconds;
uint8_t WaitOnAll;
uint8_t pad[7];
} event_req_wait_on_multi_args;
VHSAKMT_STATIC_ASSERT_SIZE(_event_req_wait_on_multi_args)
typedef struct _event_req_wait_on_multi_ext_args {
/*HsaEvent* Events[], in playloud*/
uint32_t NumEvents;
uint32_t Milliseconds;
uint64_t event_age;
uint8_t WaitOnAll;
uint8_t pad[7];
} event_req_wait_on_multi_ext_args;
VHSAKMT_STATIC_ASSERT_SIZE(_event_req_wait_on_multi_ext_args)
typedef struct _event_set_trap_handler_args {
uint64_t TrapHandlerBaseAddress;
uint64_t TrapHandlerSizeInBytes;
uint64_t TrapBufferBaseAddress;
uint64_t TrapBufferSizeInBytes;
uint32_t NodeId;
uint32_t pad;
} event_set_trap_handler_args;
VHSAKMT_STATIC_ASSERT_SIZE(_event_set_trap_handler_args)
struct vhsakmt_ccmd_event_req {
struct vhsakmt_ccmd_req hdr;
union {
HsaEvent Event; /* For set, reset, query. */
HsaEvent* event_hanele;
event_req_wait_args wait_args;
event_req_create_args create_args;
event_req_wait_ext_args wait_ext_args;
event_req_wait_on_multi_args wait_on_multi_args;
event_req_wait_on_multi_ext_args wait_on_multi_ext_args;
event_set_trap_handler_args set_trap_handler_args;
};
uint32_t type;
uint32_t sync_shmem_res_id;
uint64_t blob_id;
uint32_t res_id;
uint32_t pad;
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_event_req)
VHSAKMT_DEFINE_CAST(vhsakmt_ccmd_req, vhsakmt_ccmd_event_req)
struct vhsakmt_ccmd_event_rsp {
struct vhsakmt_ccmd_rsp hdr;
int32_t ret;
vHsaEvent vevent;
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_event_rsp)
/*
* VHSAKMT_CCMD_MEMORY
*/
enum vhsakmt_ccmd_memory_type {
VHSAKMT_CCMD_MEMORY_ALLOC,
VHSAKMT_CCMD_MEMORY_MAP_TO_GPU_NODES,
VHSAKMT_CCMD_MEMORY_FREE,
VHSAKMT_CCMD_MEMORY_UNMAP_TO_GPU,
VHSAKMT_CCMD_MEMORY_AVAIL_MEM,
VHSAKMT_CCMD_MEMORY_MAP_MEM_TO_GPU,
VHSAKMT_CCMD_MEMORY_REG_MEM_WITH_FLAG,
VHSAKMT_CCMD_MEMORY_DEREG_MEM,
VHSAKMT_CCMD_MEMORY_MAP_USERPTR,
};
typedef struct _memory_req_alloc_args {
uint32_t PreferredNode;
HsaMemFlags MemFlags;
uint64_t SizeInBytes;
uint64_t MemoryAddress;
} memory_req_alloc_args;
VHSAKMT_STATIC_ASSERT_SIZE(_memory_req_alloc_args)
typedef struct _memory_req_free_args {
uint64_t MemoryAddress;
uint64_t SizeInBytes;
} memory_req_free_args;
VHSAKMT_STATIC_ASSERT_SIZE(_memory_req_free_args)
typedef struct _memory_req_map_to_GPU_nodes_args {
uint64_t MemoryAddress;
uint64_t MemorySizeInBytes;
uint64_t AlternateVAGPU;
HsaMemMapFlags MemMapFlags;
uint32_t pad;
uint64_t NumberOfNodes;
uint32_t* NodeArray;
} memory_req_map_to_GPU_nodes_args;
VHSAKMT_STATIC_ASSERT_SIZE(_memory_req_map_to_GPU_nodes_args)
typedef struct _memory_map_mem_to_gpu_args {
uint64_t MemoryAddress;
uint64_t MemorySizeInBytes;
uint8_t need_create_bo;
uint8_t pad[7];
} memory_map_mem_to_gpu_args;
VHSAKMT_STATIC_ASSERT_SIZE(_memory_map_mem_to_gpu_args)
typedef struct _memory_reg_mem_with_flag {
uint64_t MemoryAddress;
uint64_t MemorySizeInBytes;
HsaMemFlags MemFlags;
uint32_t pad;
} memory_reg_mem_with_flag;
VHSAKMT_STATIC_ASSERT_SIZE(_memory_reg_mem_with_flag)
struct vhsakmt_ccmd_memory_req {
struct vhsakmt_ccmd_req hdr;
union {
uint64_t MemoryAddress;
uint32_t Node;
memory_req_alloc_args alloc_args;
memory_req_map_to_GPU_nodes_args map_to_GPU_nodes_args;
memory_req_free_args free_args;
memory_map_mem_to_gpu_args map_to_GPU_args;
memory_reg_mem_with_flag reg_mem_with_flag;
};
uint64_t blob_id;
uint32_t type;
uint32_t res_id;
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_memory_req)
VHSAKMT_DEFINE_CAST(vhsakmt_ccmd_req, vhsakmt_ccmd_memory_req)
typedef struct _vhsakmt_ccmd_memory_map_userptr_rsp {
uint64_t userptr_handle;
uint32_t npfns;
uint32_t pad;
} vhsakmt_ccmd_memory_map_userptr_rsp;
VHSAKMT_STATIC_ASSERT_SIZE(_vhsakmt_ccmd_memory_map_userptr_rsp)
struct vhsakmt_ccmd_memory_rsp {
struct vhsakmt_ccmd_rsp hdr;
int32_t ret;
union {
vhsakmt_ccmd_memory_map_userptr_rsp map_userptr_rsp;
uint64_t memory_handle;
uint64_t alternate_vagpu;
uint64_t available_bytes;
};
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_memory_rsp)
/*
* VHSAKMT_CCMD_QUEUE
*/
enum vhsakmt_ccmd_queue_type {
VHSAKMT_CCMD_QUEUE_CREATE,
VHSAKMT_CCMD_QUEUE_DESTROY,
};
typedef struct _vHsaQueueResource {
HsaQueueResource r;
uint64_t host_doorbell;
uint64_t host_doorbell_offset;
uint64_t host_write_offset;
uint64_t host_read_offset;
uint64_t host_rw_handle;
uint64_t queue_handle;
} vHsaQueueResource;
VHSAKMT_STATIC_ASSERT_SIZE(_vHsaQueueResource)
typedef struct _queue_req_create {
uint32_t NodeId;
HSA_QUEUE_TYPE Type;
uint32_t QueuePercentage;
uint32_t pad;
HSA_QUEUE_PRIORITY Priority;
uint32_t pad1;
uint32_t SdmaEngineId;
uint64_t QueueAddress;
uint64_t QueueSizeInBytes;
HsaEvent* Event;
HsaQueueResource* QueueResource;
uint64_t* Queue_write_ptr_aql;
uint64_t* Queue_read_ptr_aql;
} queue_req_create;
VHSAKMT_STATIC_ASSERT_SIZE(_queue_req_create)
struct vhsakmt_ccmd_queue_req {
struct vhsakmt_ccmd_req hdr;
union {
HSA_QUEUEID QueueId;
queue_req_create create_queue_args;
};
uint64_t blob_id; /* For queue create, queue resource */
uint64_t rw_ptr_blob_id; /* For queue create, r/w ptr memory mapping */
uint64_t doorbell_blob_id; /* For queue create, doorbell ptr memory mapping */
uint32_t res_id;
uint32_t type;
uint32_t queue_mem_res_id;
uint32_t pad;
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_queue_req)
VHSAKMT_DEFINE_CAST(vhsakmt_ccmd_req, vhsakmt_ccmd_queue_req)
struct vhsakmt_ccmd_queue_rsp {
struct vhsakmt_ccmd_rsp hdr;
int32_t ret;
vHsaQueueResource vqueue_res;
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_queue_rsp)
/*
* VHSAKMT_CCMD_GL_INTER
*/
enum vhsakmt_ccmd_gl_inter_type {
VHSAKMT_CCMD_GL_REG_GHD_TO_NODES,
};
typedef struct _gl_inter_req_reg_ghd_to_nodes {
uint64_t GraphicsResourceHandle;
uint64_t NumberOfNodes; // NodeArray in payload
uint32_t res_handle;
uint32_t pad;
} gl_inter_req_reg_ghd_to_nodes;
VHSAKMT_STATIC_ASSERT_SIZE(_gl_inter_req_reg_ghd_to_nodes)
struct vhsakmt_ccmd_gl_inter_req {
struct vhsakmt_ccmd_req hdr;
union {
gl_inter_req_reg_ghd_to_nodes reg_ghd_to_nodes;
};
uint32_t type;
uint32_t pad;
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_gl_inter_req)
VHSAKMT_DEFINE_CAST(vhsakmt_ccmd_req, vhsakmt_ccmd_gl_inter_req)
struct vhsakmt_ccmd_gl_inter_rsp {
struct vhsakmt_ccmd_rsp hdr;
int32_t ret;
union {
HsaGraphicsResourceInfo info;
};
uint8_t payload[];
};
VHSAKMT_STATIC_ASSERT_SIZE(vhsakmt_ccmd_gl_inter_rsp)
#ifdef __GNUC__
#pragma GCC diagnostic pop
#endif
#endif
@@ -0,0 +1,253 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#include "hsakmt/hsakmt_virtio.h"
#include "hsakmt_virtio_device.h"
static inline uint64_t vhsakmt_doorbell_page_size(void) { return 0x2000; }
static inline uint64_t vhsakmt_queue_page_size(void) { return getpagesize(); }
HSAKMT_STATUS HSAKMTAPI vhsaKmtSetTrapHandler(HSAuint32 NodeId, void* TrapHandlerBaseAddress,
HSAuint64 TrapHandlerSizeInBytes,
void* TrapBufferBaseAddress,
HSAuint64 TrapBufferSizeInBytes) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_event_rsp* rsp;
struct vhsakmt_ccmd_event_req req = {
.hdr = VHSAKMT_CCMD(EVENT, sizeof(struct vhsakmt_ccmd_event_req)),
.type = VHSAKMT_CCMD_EVENT_SET_TRAP,
.set_trap_handler_args =
{
.NodeId = NodeId,
.TrapHandlerBaseAddress = (uint64_t)TrapHandlerBaseAddress,
.TrapHandlerSizeInBytes = TrapHandlerSizeInBytes,
.TrapBufferBaseAddress = (uint64_t)TrapBufferBaseAddress,
.TrapBufferSizeInBytes = TrapBufferSizeInBytes,
},
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_event_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
return rsp->ret;
}
static int vhsakmt_find_aql_rw_bo(vhsakmt_device_handle dev, uint64_t aql_ptr,
uint32_t* aql_bo_res_id) {
uint64_t aql_base_ptr = VHSA_ALIGN_DOWN(aql_ptr, getpagesize());
vhsakmt_bo_handle bo = vhsakmt_find_bo_by_addr(dev, (void*)aql_base_ptr);
if (!bo) return -EINVAL;
bo->bo_type |= VHSA_BO_QUEUE_AQL_RW_PTR;
*aql_bo_res_id = bo->real.res_id;
return 0;
}
static int vhsakmt_create_doorbell_blob_bo(vhsakmt_device_handle dev, uint32_t node, size_t size,
uint32_t blob_id, uint64_t host_handle,
vhsakmt_bo_handle* bo_handle) {
int r;
r = vhsakmt_create_mappable_blob_bo(dev, size, blob_id, VHSA_BO_QUEUE_DOORBELL,
(void*)host_handle, bo_handle);
if (r) return r;
r = vhsakmt_set_node_doorbell(dev, node, (*bo_handle)->cpu_addr);
return r;
}
static int vhsakmt_create_queue_rw_blob_bo(vhsakmt_device_handle dev, size_t size, uint32_t blob_id,
uint64_t host_handle, vhsakmt_bo_handle* bo_handle) {
int r;
r = vhsakmt_create_mappable_blob_bo(dev, size, blob_id, VHSA_BO_QUEUE_RW_PTR, NULL, bo_handle);
if (r) return r;
(*bo_handle)->host_addr = (void*)host_handle;
return r;
}
static int vhsakmt_create_queue_blob_bo(vhsakmt_device_handle dev, size_t size, uint32_t blob_id,
uint64_t queue_id, vhsakmt_bo_handle rw_bo_handle,
vhsakmt_bo_handle* bo_handle) {
int r;
r = vhsakmt_init_host_blob(dev, size, VIRTGPU_BLOB_MEM_HOST3D, 0, blob_id, VHSA_BO_QUEUE, NULL,
bo_handle);
if (r) return r;
vhsakmt_insert_bo(dev, *bo_handle, *bo_handle, (*bo_handle)->size);
(*bo_handle)->queue_id = queue_id;
(*bo_handle)->rw_bo = rw_bo_handle;
return r;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtCreateQueueExt(HSAuint32 NodeId, HSA_QUEUE_TYPE Type,
HSAuint32 QueuePercentage,
HSA_QUEUE_PRIORITY Priority, HSAuint32 SdmaEngineId,
void* QueueAddress, HSAuint64 QueueSizeInBytes,
HsaEvent* Event, HsaQueueResource* QueueResource) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
vhsakmt_bo_handle rw_bo_handle = NULL, doorbell_bo, queue_bo, queue_mem_bo;
struct vhsakmt_ccmd_queue_rsp* rsp;
struct vhsakmt_ccmd_queue_req req = {
.hdr = VHSAKMT_CCMD(QUEUE, sizeof(struct vhsakmt_ccmd_queue_req)),
.type = VHSAKMT_CCMD_QUEUE_CREATE,
.create_queue_args =
{
.NodeId = NodeId,
.Type = Type,
.QueuePercentage = QueuePercentage,
.Priority = Priority,
.SdmaEngineId = SdmaEngineId,
.QueueAddress = (uint64_t)QueueAddress,
.QueueSizeInBytes = QueueSizeInBytes,
.Event = Event ? vhsakmt_event_host_handle(Event) : 0,
.Queue_write_ptr_aql = QueueResource->Queue_write_ptr_aql,
.Queue_read_ptr_aql = QueueResource->Queue_read_ptr_aql,
},
.blob_id = vhsakmt_atomic_inc_return(&dev->next_blob_id), /* For queue resource */
.doorbell_blob_id = vhsakmt_node_doorbell(dev, NodeId)
? 0
: vhsakmt_atomic_inc_return(&dev->next_blob_id), /* For queue doorbell memory map */
};
int r;
/* Queue ptr memory is allocated by hsakmtallocmemory in host then mapped into guest, but their
* address are not aligned. */
if (Type == HSA_QUEUE_COMPUTE_AQL) {
r = vhsakmt_find_aql_rw_bo(dev, QueueResource->QueueWptrValue, &req.res_id);
if (r) {
vhsa_debug("%s: can not find the AQL queue R/W BO: %p\n", __FUNCTION__,
QueueResource->Queue_write_ptr_aql);
return HSAKMT_STATUS_NO_MEMORY;
}
vhsa_debug("%s: create AQL queue, read ptr: %p, write ptr: %p, res id: %d\n", __FUNCTION__,
QueueResource->Queue_read_ptr_aql, QueueResource->Queue_write_ptr_aql, req.res_id);
} else
/* For queue not CP AQL, it use r/w ptr by itself. */
req.rw_ptr_blob_id = vhsakmt_atomic_inc_return(&dev->next_blob_id);
queue_mem_bo = vhsakmt_find_bo_by_addr(dev, QueueAddress);
if (!queue_mem_bo) {
vhsa_err("%s: can not find the queue memory BO: %p\n", __FUNCTION__, QueueAddress);
return HSAKMT_STATUS_NO_MEMORY;
}
queue_mem_bo->bo_type |= VHSA_BO_QUEUE_AQL_RW_PTR;
req.queue_mem_res_id = queue_mem_bo->real.res_id;
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_queue_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (rsp->ret) {
vhsa_err("%s: queue create failed, ret: %d", __FUNCTION__, rsp->ret);
return rsp->ret;
}
/* Map doorbell */
if (req.doorbell_blob_id) {
r = vhsakmt_create_doorbell_blob_bo(
dev, NodeId, vhsakmt_doorbell_page_size(), req.doorbell_blob_id,
rsp->vqueue_res.host_doorbell - rsp->vqueue_res.host_doorbell_offset, &doorbell_bo);
if (r) {
vhsa_err("%s: doorbell create failed, doorbell: %lx\n", __FUNCTION__,
rsp->vqueue_res.host_doorbell);
return r;
}
vhsa_debug("%s: create doorbell: %p, size: 0x%x\n", __FUNCTION__, doorbell_bo->cpu_addr,
doorbell_bo->size);
}
QueueResource->Queue_DoorBell_aql = (void*)rsp->vqueue_res.host_doorbell;
vhsa_debug("%s: queue create, Doorbell: %p\n", __FUNCTION__, QueueResource->Queue_DoorBell_aql);
/* Map R/W pointer.
* For a queue is not a COMPUTE AQL, the R/W PTR not using the input address,
* uses the queue memory allocated by hsakmtallocmemory, a page align address.
*/
if (Type != HSA_QUEUE_COMPUTE_AQL) {
r = vhsakmt_create_queue_rw_blob_bo(dev, vhsakmt_queue_page_size(), req.rw_ptr_blob_id,
rsp->vqueue_res.host_rw_handle, &rw_bo_handle);
if (r) {
vhsa_debug("%s: queue rw ptr create failed, host addr: %p\n", __FUNCTION__,
(void*)rsp->vqueue_res.host_rw_handle);
return r;
}
QueueResource->Queue_write_ptr_aql = VHSA_UINT64_TO_VPTR(
VHSA_VPTR_TO_UINT64(rw_bo_handle->cpu_addr) + rsp->vqueue_res.host_write_offset);
QueueResource->Queue_read_ptr_aql = VHSA_UINT64_TO_VPTR(
VHSA_VPTR_TO_UINT64(rw_bo_handle->cpu_addr) + rsp->vqueue_res.host_read_offset);
vhsa_debug("%s: queue create: write ptr gva: %p, read ptr gva: %p, base hva: %lx\n",
__FUNCTION__, QueueResource->Queue_write_ptr_aql, QueueResource->Queue_read_ptr_aql,
rsp->vqueue_res.host_rw_handle);
}
r = vhsakmt_create_queue_blob_bo(dev, QueueSizeInBytes, req.blob_id, rsp->vqueue_res.r.QueueId,
rw_bo_handle, &queue_bo);
if (r) {
vhsa_err("%s: queue create failed, queue ID: 0x%lx\n", __FUNCTION__, rsp->vqueue_res.r.QueueId);
return r;
}
QueueResource->QueueId = (uint64_t)queue_bo;
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtCreateQueue(HSAuint32 NodeId, HSA_QUEUE_TYPE Type,
HSAuint32 QueuePercentage, HSA_QUEUE_PRIORITY Priority,
void* QueueAddress, HSAuint64 QueueSizeInBytes,
HsaEvent* Event, HsaQueueResource* QueueResource) {
return vhsaKmtCreateQueueExt(NodeId, Type, QueuePercentage, Priority, VHSA_SDMA_NONE,
QueueAddress, QueueSizeInBytes, Event, QueueResource);
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtDestroyQueue(HSA_QUEUEID QueueId) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
int r;
/* queue ID: vhsakmt_bo_handle -> real queue ID*/
vhsakmt_bo_handle bo = (vhsakmt_bo_handle)QueueId;
vhsakmt_bo_handle rw_bo = bo->rw_bo;
r = vhsakmt_bo_free(dev, bo);
if (rw_bo) vhsakmt_bo_free(dev, rw_bo);
vhsa_debug("%s: queue res id: %d, queue ID: %" PRIu64 ", ret = %d\n", __FUNCTION__,
bo->real.res_id, bo->queue_id, r);
return r;
}
@@ -0,0 +1,342 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#include "hsakmt/hsakmt_virtio.h"
#include "hsakmt_virtio_device.h"
static int vhsakmt_set_sys_props(vhsakmt_device_handle dev, HsaSystemProperties* sys_props) {
int r = 0;
pthread_mutex_lock(&dev->vhsakmt_mutex);
if (dev->sys_props) {
r = 0;
goto out;
}
dev->sys_props = calloc(1, sizeof(HsaSystemProperties));
if (!dev->sys_props) {
r = -ENOMEM;
goto out;
}
memcpy(dev->sys_props, sys_props, sizeof(HsaSystemProperties));
out:
pthread_mutex_unlock(&dev->vhsakmt_mutex);
return r;
}
static int vhsakmt_set_node_props(vhsakmt_device_handle dev, uint32_t node,
HsaNodeProperties* node_props) {
int r = 0;
if (!dev->sys_props) return -EINVAL;
if (node >= dev->sys_props->NumNodes) return -EINVAL;
pthread_mutex_lock(&dev->vhsakmt_mutex);
if (!dev->vhsakmt_nodes) {
dev->vhsakmt_nodes = calloc(dev->sys_props->NumNodes, sizeof(struct vhsakmt_node));
if (!dev->vhsakmt_nodes) {
r = -ENOMEM;
goto out;
}
}
memcpy(&dev->vhsakmt_nodes[node].node_props, node_props, sizeof(HsaNodeProperties));
out:
pthread_mutex_unlock(&dev->vhsakmt_mutex);
return r;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetVersion(HsaVersionInfo* v) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GET_VER,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
memcpy(v, &rsp->kfd_version, sizeof(HsaVersionInfo));
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtAcquireSystemProperties(HsaSystemProperties* SystemProperties) {
CHECK_VIRTIO_KFD_OPEN();
int r;
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GET_SYS_PROP,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (!rsp) return -ENOMEM;
memcpy(SystemProperties, &rsp->sys_props, sizeof(HsaSystemProperties));
r = vhsakmt_set_sys_props(dev, SystemProperties);
if (r) return r;
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtReleaseSystemProperties(void) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_REL_SYS_PROP,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (!rsp) return -ENOMEM;
if (dev->sys_props) {
free(dev->sys_props);
dev->sys_props = NULL;
}
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetNodeProperties(HSAuint32 NodeId,
HsaNodeProperties* NodeProperties) {
CHECK_VIRTIO_KFD_OPEN();
int r;
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.NodeID = NodeId,
.type = VHSAKMT_CCMD_QUERY_GET_NODE_PROP,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (!rsp) return -ENOMEM;
memcpy(NodeProperties, &rsp->node_props, sizeof(HsaNodeProperties));
r = vhsakmt_set_node_props(dev, NodeId, NodeProperties);
if (r) return r;
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetXNACKMode(HSAint32* enable) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GET_XNACK_MODE,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (!rsp) return -ENOMEM;
memcpy(enable, &rsp->xnack_mode, sizeof(HSAint32));
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtRuntimeEnable(void* rDebug, bool setupTtmp) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.run_time_enable_args.setupTtmp = setupTtmp,
.type = VHSAKMT_CCMD_QUERY_RUN_TIME_ENABLE,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (!rsp) return -ENOMEM;
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtRuntimeDisable(void) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_RUN_TIME_DISABLE,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
if (!rsp) return -ENOMEM;
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetNodeMemoryProperties(HSAuint32 NodeId, HSAuint32 NumBanks,
HsaMemoryProperties* MemoryProperties) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GET_NOD_MEM_PROP,
.node_mem_prop_args.NodeId = NodeId,
.node_mem_prop_args.NumBanks = NumBanks,
};
rsp = vhsakmt_alloc_rsp(
dev, &req.hdr,
sizeof(struct vhsakmt_ccmd_query_info_rsp) + NumBanks * sizeof(HsaMemoryProperties));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
memcpy(MemoryProperties, rsp->payload, NumBanks * sizeof(HsaMemoryProperties));
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetNodeCacheProperties(HSAuint32 NodeId, HSAuint32 ProcessorId,
HSAuint32 NumCaches,
HsaCacheProperties* CacheProperties) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GET_NOD_CACHE_PROP,
.node_cache_prop_args.NodeId = NodeId,
.node_cache_prop_args.ProcessorId = ProcessorId,
.node_cache_prop_args.NumCaches = NumCaches,
};
rsp = vhsakmt_alloc_rsp(
dev, &req.hdr,
sizeof(struct vhsakmt_ccmd_query_info_rsp) + NumCaches * sizeof(HsaCacheProperties));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
memcpy(CacheProperties, rsp->payload, NumCaches * sizeof(HsaCacheProperties));
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetNodeIoLinkProperties(HSAuint32 NodeId, HSAuint32 NumIoLinks,
HsaIoLinkProperties* IoLinkProperties) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GET_NOD_IO_LINK_PROP,
.node_io_link_args.NodeId = NodeId,
.node_io_link_args.NumIoLinks = NumIoLinks,
};
rsp = vhsakmt_alloc_rsp(
dev, &req.hdr,
sizeof(struct vhsakmt_ccmd_query_info_rsp) + NumIoLinks * sizeof(HsaIoLinkProperties));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
memcpy(IoLinkProperties, rsp->payload, NumIoLinks * sizeof(HsaIoLinkProperties));
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetClockCounters(HSAuint32 NodeId, HsaClockCounters* Counters) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GET_CLOCK_COUNTERS,
.NodeID = NodeId,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
memcpy(Counters, &rsp->clock_counters, sizeof(HsaClockCounters));
return rsp->ret;
}
HSAKMT_STATUS HSAKMTAPI vhsaKmtGetRuntimeCapabilities(HSAuint32* caps_mask) {
CHECK_VIRTIO_KFD_OPEN();
vhsakmt_device_handle dev = vhsakmt_dev();
struct vhsakmt_ccmd_query_info_rsp* rsp;
struct vhsakmt_ccmd_query_info_req req = {
.hdr = VHSAKMT_CCMD(QUERY_INFO, sizeof(struct vhsakmt_ccmd_query_info_req)),
.type = VHSAKMT_CCMD_QUERY_GET_RUNTIME_CAPS,
};
rsp = vhsakmt_alloc_rsp(dev, &req.hdr, sizeof(struct vhsakmt_ccmd_query_info_rsp));
if (!rsp) return -ENOMEM;
vhsakmt_execbuf_cpu(dev, &req.hdr, __FUNCTION__);
*caps_mask = rsp->caps;
return rsp->ret;
}
@@ -0,0 +1,113 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#include <sys/mman.h>
#include <unistd.h>
#include "hsakmt_virtio_device.h"
void* vhsakmt_vm_start(void) {
void* vm_start = malloc(getpagesize());
if (!vm_start) return NULL;
free(vm_start);
return vm_start;
}
int vhsakmt_reserve_va(uint64_t start, uint64_t size) {
int32_t protFlags = PROT_NONE;
int32_t mapFlags = MAP_PRIVATE | MAP_ANONYMOUS | MAP_FIXED;
void* va = mmap((void*)start, size, protFlags, mapFlags, -1, 0);
if (va == MAP_FAILED) return -ENOMEM;
if (va != (void*)start) return -ENOMEM;
madvise(va, size, MADV_DONTFORK);
return 0;
}
void vhsakmt_dereserve_va(uint64_t start, uint64_t size) { munmap((void*)start, size); }
void vhsakmt_set_scratch_area(vhsakmt_device_handle dev, uint32_t node, uint64_t start,
uint64_t size) {
if (!dev->vhsakmt_nodes || !dev->sys_props) return;
if (node >= dev->sys_props->NumNodes) return;
pthread_mutex_lock(&dev->vhsakmt_mutex);
if (dev->vhsakmt_nodes[node].scratch_start && dev->vhsakmt_nodes[node].scratch_size) goto out;
dev->vhsakmt_nodes[node].scratch_start = start;
dev->vhsakmt_nodes[node].scratch_size = size;
out:
pthread_mutex_unlock(&dev->vhsakmt_mutex);
}
bool vhsakmt_is_scratch_mem(vhsakmt_device_handle dev, void* addr) {
uint32_t i;
if (!dev->vhsakmt_nodes || !dev->sys_props) return false;
for (i = 0; i < dev->sys_props->NumNodes; i++) {
if ((uint64_t)addr >= dev->vhsakmt_nodes[i].scratch_start &&
(uint64_t)addr <= dev->vhsakmt_nodes[i].scratch_start + dev->vhsakmt_nodes[i].scratch_size)
return true;
}
return false;
}
void vhsakmt_set_vm_area(vhsakmt_device_handle dev, uint64_t start, uint64_t size) {
pthread_mutex_lock(&dev->vhsakmt_mutex);
if (dev->vm_start && dev->vm_size) goto out;
dev->vm_start = start;
dev->vm_size = size;
out:
pthread_mutex_unlock(&dev->vhsakmt_mutex);
}
bool vhsakmt_is_userptr(vhsakmt_device_handle dev, void* addr) {
return !((uint64_t)addr >= dev->vm_start && (uint64_t)addr <= dev->vm_start + dev->vm_size);
}
int vhsakmt_set_node_doorbell(vhsakmt_device_handle dev, uint32_t node, void* doorbell) {
if (!dev->vhsakmt_nodes || !dev->sys_props) return -EINVAL;
if (node >= dev->sys_props->NumNodes) return -EINVAL;
pthread_mutex_lock(&dev->vhsakmt_mutex);
dev->vhsakmt_nodes[node].doorbell_base = doorbell;
pthread_mutex_unlock(&dev->vhsakmt_mutex);
return 0;
}
void* vhsakmt_node_doorbell(vhsakmt_device_handle dev, uint32_t node) {
if (!dev->vhsakmt_nodes || !dev->sys_props) return NULL;
if (node >= dev->sys_props->NumNodes) return NULL;
return dev->vhsakmt_nodes[node].doorbell_base;
}
@@ -0,0 +1,279 @@
/*
* Copyright 2013 Red Hat
* All Rights Reserved.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice (including the next
* paragraph) shall be included in all copies or substantial portions of the
* Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE AUTHORS AND/OR ITS SUPPLIERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#ifndef VIRTGPU_DRM_H
#define VIRTGPU_DRM_H
#include "drm.h"
#if defined(__cplusplus)
extern "C" {
#endif
/* Please note that modifications to all structs defined here are
* subject to backwards-compatibility constraints.
*
* Do not use pointers, use __u64 instead for 32 bit / 64 bit user/kernel
* compatibility Keep fields aligned to their size
*/
#define DRM_VIRTGPU_MAP 0x01
#define DRM_VIRTGPU_EXECBUFFER 0x02
#define DRM_VIRTGPU_GETPARAM 0x03
#define DRM_VIRTGPU_RESOURCE_CREATE 0x04
#define DRM_VIRTGPU_RESOURCE_INFO 0x05
#define DRM_VIRTGPU_TRANSFER_FROM_HOST 0x06
#define DRM_VIRTGPU_TRANSFER_TO_HOST 0x07
#define DRM_VIRTGPU_WAIT 0x08
#define DRM_VIRTGPU_GET_CAPS 0x09
#define DRM_VIRTGPU_RESOURCE_CREATE_BLOB 0x0a
#define DRM_VIRTGPU_CONTEXT_INIT 0x0b
#define VIRTGPU_EXECBUF_FENCE_FD_IN 0x01
#define VIRTGPU_EXECBUF_FENCE_FD_OUT 0x02
#define VIRTGPU_EXECBUF_RING_IDX 0x04
#define VIRTGPU_EXECBUF_FLAGS (\
VIRTGPU_EXECBUF_FENCE_FD_IN |\
VIRTGPU_EXECBUF_FENCE_FD_OUT |\
VIRTGPU_EXECBUF_RING_IDX |\
0)
struct drm_virtgpu_map {
__u64 offset; /* use for mmap system call */
__u32 handle;
__u32 pad;
};
#define VIRTGPU_EXECBUF_SYNCOBJ_RESET 0x01
#define VIRTGPU_EXECBUF_SYNCOBJ_FLAGS ( \
VIRTGPU_EXECBUF_SYNCOBJ_RESET | \
0)
struct drm_virtgpu_execbuffer_syncobj {
__u32 handle;
__u32 flags;
__u64 point;
};
/* fence_fd is modified on success if VIRTGPU_EXECBUF_FENCE_FD_OUT flag is set. */
struct drm_virtgpu_execbuffer {
__u32 flags;
__u32 size;
__u64 command; /* void* */
__u64 bo_handles;
__u32 num_bo_handles;
__s32 fence_fd; /* in/out fence fd (see VIRTGPU_EXECBUF_FENCE_FD_IN/OUT) */
__u32 ring_idx; /* command ring index (see VIRTGPU_EXECBUF_RING_IDX) */
__u32 syncobj_stride; /* size of @drm_virtgpu_execbuffer_syncobj */
__u32 num_in_syncobjs;
__u32 num_out_syncobjs;
__u64 in_syncobjs;
__u64 out_syncobjs;
};
#define VIRTGPU_PARAM_3D_FEATURES 1 /* do we have 3D features in the hw */
#define VIRTGPU_PARAM_CAPSET_QUERY_FIX 2 /* do we have the capset fix */
#define VIRTGPU_PARAM_RESOURCE_BLOB 3 /* DRM_VIRTGPU_RESOURCE_CREATE_BLOB */
#define VIRTGPU_PARAM_HOST_VISIBLE 4 /* Host blob resources are mappable */
#define VIRTGPU_PARAM_CROSS_DEVICE 5 /* Cross virtio-device resource sharing */
#define VIRTGPU_PARAM_CONTEXT_INIT 6 /* DRM_VIRTGPU_CONTEXT_INIT */
#define VIRTGPU_PARAM_SUPPORTED_CAPSET_IDs 7 /* Bitmask of supported capability set ids */
#define VIRTGPU_PARAM_EXPLICIT_DEBUG_NAME 8 /* Ability to set debug name from userspace */
struct drm_virtgpu_getparam {
__u64 param;
__u64 value;
};
/* NO_BO flags? NO resource flag? */
/* resource flag for y_0_top */
struct drm_virtgpu_resource_create {
__u32 target;
__u32 format;
__u32 bind;
__u32 width;
__u32 height;
__u32 depth;
__u32 array_size;
__u32 last_level;
__u32 nr_samples;
__u32 flags;
__u32 bo_handle; /* if this is set - recreate a new resource attached to this bo ? */
__u32 res_handle; /* returned by kernel */
__u32 size; /* validate transfer in the host */
__u32 stride; /* validate transfer in the host */
};
struct drm_virtgpu_resource_info {
__u32 bo_handle;
__u32 res_handle;
__u32 size;
__u32 blob_mem;
};
struct drm_virtgpu_3d_box {
__u32 x;
__u32 y;
__u32 z;
__u32 w;
__u32 h;
__u32 d;
};
struct drm_virtgpu_3d_transfer_to_host {
__u32 bo_handle;
struct drm_virtgpu_3d_box box;
__u32 level;
__u32 offset;
__u32 stride;
__u32 layer_stride;
};
struct drm_virtgpu_3d_transfer_from_host {
__u32 bo_handle;
struct drm_virtgpu_3d_box box;
__u32 level;
__u32 offset;
__u32 stride;
__u32 layer_stride;
};
#define VIRTGPU_WAIT_NOWAIT 1 /* like it */
struct drm_virtgpu_3d_wait {
__u32 handle; /* 0 is an invalid handle */
__u32 flags;
};
#define VIRTGPU_DRM_CAPSET_VIRGL 1
#define VIRTGPU_DRM_CAPSET_VIRGL2 2
#define VIRTGPU_DRM_CAPSET_GFXSTREAM_VULKAN 3
#define VIRTGPU_DRM_CAPSET_VENUS 4
#define VIRTGPU_DRM_CAPSET_CROSS_DOMAIN 5
#define VIRTGPU_DRM_CAPSET_DRM 6
struct drm_virtgpu_get_caps {
__u32 cap_set_id;
__u32 cap_set_ver;
__u64 addr;
__u32 size;
__u32 pad;
};
struct drm_virtgpu_resource_create_blob {
#define VIRTGPU_BLOB_MEM_GUEST 0x0001
#define VIRTGPU_BLOB_MEM_HOST3D 0x0002
#define VIRTGPU_BLOB_MEM_HOST3D_GUEST 0x0003
#define VIRTGPU_BLOB_FLAG_USE_MAPPABLE 0x0001
#define VIRTGPU_BLOB_FLAG_USE_SHAREABLE 0x0002
#define VIRTGPU_BLOB_FLAG_USE_CROSS_DEVICE 0x0004
#define VIRTGPU_BLOB_FLAG_USE_USERPTR 0x0008
/* zero is invalid blob_mem */
__u32 blob_mem;
__u32 blob_flags;
__u32 bo_handle;
__u32 res_handle;
__u64 size;
/*
* for 3D contexts with VIRTGPU_BLOB_MEM_HOST3D_GUEST and
* VIRTGPU_BLOB_MEM_HOST3D otherwise, must be zero.
*/
__u32 pad;
__u32 cmd_size;
__u64 cmd;
__u64 blob_id;
__u64 blob_userptr;
__s64 offset;
};
#define VIRTGPU_CONTEXT_PARAM_CAPSET_ID 0x0001
#define VIRTGPU_CONTEXT_PARAM_NUM_RINGS 0x0002
#define VIRTGPU_CONTEXT_PARAM_POLL_RINGS_MASK 0x0003
#define VIRTGPU_CONTEXT_PARAM_DEBUG_NAME 0x0004
struct drm_virtgpu_context_set_param {
__u64 param;
__u64 value;
};
struct drm_virtgpu_context_init {
__u32 num_params;
__u32 pad;
/* pointer to drm_virtgpu_context_set_param array */
__u64 ctx_set_params;
};
/*
* Event code that's given when VIRTGPU_CONTEXT_PARAM_POLL_RINGS_MASK is in
* effect. The event size is sizeof(drm_event), since there is no additional
* payload.
*/
#define VIRTGPU_EVENT_FENCE_SIGNALED 0x90000000
#define DRM_IOCTL_VIRTGPU_MAP \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_MAP, struct drm_virtgpu_map)
#define DRM_IOCTL_VIRTGPU_EXECBUFFER \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_EXECBUFFER,\
struct drm_virtgpu_execbuffer)
#define DRM_IOCTL_VIRTGPU_GETPARAM \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_GETPARAM,\
struct drm_virtgpu_getparam)
#define DRM_IOCTL_VIRTGPU_RESOURCE_CREATE \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_RESOURCE_CREATE, \
struct drm_virtgpu_resource_create)
#define DRM_IOCTL_VIRTGPU_RESOURCE_INFO \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_RESOURCE_INFO, \
struct drm_virtgpu_resource_info)
#define DRM_IOCTL_VIRTGPU_TRANSFER_FROM_HOST \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_TRANSFER_FROM_HOST, \
struct drm_virtgpu_3d_transfer_from_host)
#define DRM_IOCTL_VIRTGPU_TRANSFER_TO_HOST \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_TRANSFER_TO_HOST, \
struct drm_virtgpu_3d_transfer_to_host)
#define DRM_IOCTL_VIRTGPU_WAIT \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_WAIT, \
struct drm_virtgpu_3d_wait)
#define DRM_IOCTL_VIRTGPU_GET_CAPS \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_GET_CAPS, \
struct drm_virtgpu_get_caps)
#define DRM_IOCTL_VIRTGPU_RESOURCE_CREATE_BLOB \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_RESOURCE_CREATE_BLOB, \
struct drm_virtgpu_resource_create_blob)
#define DRM_IOCTL_VIRTGPU_CONTEXT_INIT \
DRM_IOWR(DRM_COMMAND_BASE + DRM_VIRTGPU_CONTEXT_INIT, \
struct drm_virtgpu_context_init)
#if defined(__cplusplus)
}
#endif
#endif
@@ -0,0 +1,45 @@
{
global:
vhsaKmtOpenKFD;
vhsaKmtCloseKFD;
vhsaKmtAllocMemory;
vhsaKmtFreeMemory;
vhsaKmtMapMemoryToGPUNodes;
vhsaKmtUnmapMemoryToGPU;
vhsaKmtAvailableMemory;
vhsaKmtMapMemoryToGPU;
vhsaKmtRegisterMemoryWithFlags;
vhsaKmtDeregisterMemory;
vhsaKmtGetVersion;
vhsaKmtAcquireSystemProperties;
vhsaKmtReleaseSystemProperties;
vhsaKmtGetNodeProperties;
vhsaKmtGetXNACKMode;
vhsaKmtRuntimeEnable;
vhsaKmtRuntimeDisable;
vhsaKmtGetNodeMemoryProperties;
vhsaKmtGetNodeCacheProperties;
vhsaKmtGetNodeIoLinkProperties;
vhsaKmtGetClockCounters;
vhsaKmtGetAMDGPUDeviceHandle;
vhsaKmtQueryPointerInfo;
vhsaKmtGetTileConfig;
vhsaKmtCreateEvent;
vhsaKmtDestroyEvent;
vhsaKmtSetEvent;
vhsaKmtResetEvent;
vhsaKmtQueryEventState;
vhsaKmtWaitOnMultipleEvents;
vhsaKmtWaitOnEvent;
vhsaKmtWaitOnEvent_Ext;
vhsaKmtWaitOnMultipleEvents_Ext;
vhsaKmtSetTrapHandler;
vhsaKmtCreateQueueExt;
vhsaKmtCreateQueue;
vhsaKmtDestroyQueue;
vhsaKmtRegisterGraphicsHandleToNodes;
vhsaKmtGetRuntimeCapabilities;
vamdgpu_query_gpu_info;
local: *;
};
@@ -0,0 +1,324 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#include <errno.h>
#include <libsync.h>
#include <stdbool.h>
#include <stddef.h>
#include <stdlib.h>
#include <sys/mman.h>
#include <fcntl.h>
#include "virtio_gpu.h"
#define SHMEM_SZ (25 * 0x1000)
static int set_context(int fd) {
struct drm_virtgpu_context_set_param params[] = {
{VIRTGPU_CONTEXT_PARAM_CAPSET_ID, VIRGL_RENDERER_CAPSET_HSAKMT},
{VIRTGPU_CONTEXT_PARAM_NUM_RINGS, 64},
};
struct drm_virtgpu_context_init args = {
.num_params = ARRAY_SIZE(params),
.ctx_set_params = (uintptr_t)(params),
};
return virtio_gpu_ioctl(fd, VIRTGPU_CONTEXT_INIT, &args);
}
int virtio_gpu_map_handle(struct virtio_gpu_device* vgdev, uint32_t handle, uint64_t size,
void** addr, void* fixed_map) {
struct drm_virtgpu_map args = {
.handle = handle,
};
int r;
r = virtio_gpu_ioctl(vgdev->fd, VIRTGPU_MAP, &args);
if (r) return r;
*addr = mmap(fixed_map, size, PROT_READ | PROT_WRITE, MAP_SHARED | (fixed_map ? MAP_FIXED : 0),
vgdev->fd, args.offset);
if (*addr == MAP_FAILED) return -EINVAL;
return 0;
}
void virtio_gpu_unmap(void* addr, uint64_t size) { munmap(addr, size); }
static void virtio_gpu_bo_close(struct virtio_gpu_device* vgdev, uint32_t handle) {
struct drm_gem_close args = {
.handle = handle,
};
virtio_gpu_ioctl(vgdev->fd, GEM_CLOSE, &args);
}
static int virtio_gpu_shmem_init(struct virtio_gpu_device* vgdev, size_t size) {
struct drm_virtgpu_resource_create_blob args = {
.blob_mem = VIRTGPU_BLOB_MEM_HOST3D,
.blob_flags = VIRTGPU_BLOB_FLAG_USE_MAPPABLE,
.size = size,
.blob_id = 0,
};
int r = virtio_gpu_ioctl(vgdev->fd, VIRTGPU_RESOURCE_CREATE_BLOB, &args);
if (r) return r;
r = virtio_gpu_map_handle(vgdev, args.bo_handle, size, (void**)&vgdev->shmem, NULL);
if (r) {
virtio_gpu_bo_close(vgdev, args.bo_handle);
return r;
}
vgdev->shmem_handle = args.bo_handle;
uint32_t offset = vgdev->shmem->base.rsp_mem_offset;
vgdev->rsp_mem_len = size - offset;
vgdev->rsp_mem = &((uint8_t*)vgdev->shmem)[offset];
return 0;
}
struct virtio_gpu_device* virtio_gpu_init(int fd, uint32_t context_id) {
struct virtio_gpu_device* vgdev;
int r;
r = set_context(fd);
if (r) return NULL;
vgdev = calloc(1, sizeof(*vgdev));
if (!vgdev) return NULL;
vgdev->fd = fd;
vgdev->reqbuf = calloc(1, SHMEM_SZ);
if (!vgdev->reqbuf) {
free(vgdev);
return NULL;
}
r = virtio_gpu_shmem_init(vgdev, SHMEM_SZ);
if (r) {
free(vgdev);
return NULL;
}
pthread_mutex_init(&vgdev->rsp_lock, NULL);
pthread_mutex_init(&vgdev->eb_lock, NULL);
return vgdev;
}
void virtio_gpu_close(struct virtio_gpu_device* vgdev) {
virtio_gpu_unmap(vgdev->shmem, SHMEM_SZ);
virtio_gpu_bo_close(vgdev, vgdev->shmem_handle);
pthread_mutex_destroy(&vgdev->rsp_lock);
pthread_mutex_destroy(&vgdev->eb_lock);
close(vgdev->fd);
free(vgdev->reqbuf);
free(vgdev);
}
void* virtio_gpu_alloc_rsp(struct virtio_gpu_device* vgdev, struct virtio_gpu_ccmd_req* req,
uint32_t size) {
uint32_t off;
pthread_mutex_lock(&vgdev->rsp_lock);
size = VHSA_ALIGN_UP(size, 8);
if ((vgdev->next_rsp_off + size) >= vgdev->rsp_mem_len) vgdev->next_rsp_off = 0;
off = vgdev->next_rsp_off;
vgdev->next_rsp_off += size;
pthread_mutex_unlock(&vgdev->rsp_lock);
req->rsp_off = off;
struct virtio_gpu_ccmd_rsp* rsp = (void*)&vgdev->rsp_mem[off];
rsp->len = size;
return rsp;
}
static int virtio_gpu_execbuffer_locked(struct virtio_gpu_device* vgdev, void* cmd,
uint32_t cmd_size, uint32_t* handles, uint32_t num_handles,
int* fence_fd, int ring_idx, uint32_t num_in_syncobjs,
uint32_t num_out_syncobjs,
struct drm_virtgpu_execbuffer_syncobj* in_syncobjs,
struct drm_virtgpu_execbuffer_syncobj* out_syncobjs,
bool in_fence, bool out_fence) {
struct drm_virtgpu_execbuffer eb = {
.flags = (out_fence ? VIRTGPU_EXECBUF_FENCE_FD_OUT : 0) |
(in_fence ? VIRTGPU_EXECBUF_FENCE_FD_IN : 0) | VIRTGPU_EXECBUF_RING_IDX,
.size = cmd_size,
.command = (uintptr_t)cmd,
.bo_handles = (uintptr_t)handles,
.num_bo_handles = num_handles,
.fence_fd = *fence_fd,
.ring_idx = ring_idx,
.syncobj_stride = sizeof(struct drm_virtgpu_execbuffer_syncobj),
.num_in_syncobjs = num_in_syncobjs,
.num_out_syncobjs = num_out_syncobjs,
.in_syncobjs = (uintptr_t)in_syncobjs,
.out_syncobjs = (uintptr_t)out_syncobjs,
};
int r = virtio_gpu_ioctl(vgdev->fd, VIRTGPU_EXECBUFFER, &eb);
if (r) return r;
if (out_fence) *fence_fd = eb.fence_fd;
return 0;
}
static int virtio_gpu_flush_locked(struct virtio_gpu_device* vgdev, int* fence) {
int r;
if (!vgdev->reqbuf_len) return 0;
r = virtio_gpu_execbuffer_locked(vgdev, vgdev->reqbuf, vgdev->reqbuf_len, NULL, 0, fence, 0, 0, 0,
NULL, NULL, false, !!fence);
if (r) return r;
vgdev->reqbuf_len = 0;
vgdev->reqbuf_cnt = 0;
return 0;
}
static int virtio_gpu_add_cmd(struct virtio_gpu_device* vgdev, struct virtio_gpu_ccmd_req* req) {
req->seqno = ++vgdev->next_seqno;
int r;
if (vgdev->reqbuf_len + req->len > sizeof(vgdev->reqbuf)) {
r = virtio_gpu_flush_locked(vgdev, NULL);
if (r) return r;
}
memcpy(&vgdev->reqbuf[vgdev->reqbuf_len], req, req->len);
vgdev->reqbuf_len += req->len;
vgdev->reqbuf_cnt++;
return 0;
}
static inline bool fence_before(uint32_t a, uint32_t b) { return (int32_t)(a - b) < 0; }
static void virtio_gpu_seqno_sync(struct virtio_gpu_device* vgdev,
struct virtio_gpu_ccmd_req* req) {
while (fence_before(vgdev->shmem->base.seqno, req->seqno)) sched_yield();
}
int virtio_gpu_exec_cmd(struct virtio_gpu_device* vgdev, struct virtio_gpu_ccmd_req* req,
bool sync) {
int r = 0;
int fence;
pthread_mutex_lock(&vgdev->eb_lock);
r = virtio_gpu_add_cmd(vgdev, req);
if (r || !sync) goto out;
r = virtio_gpu_flush_locked(vgdev, &fence);
out:
pthread_mutex_unlock(&vgdev->eb_lock);
if (r) return r;
if (sync) {
sync_wait(fence, -1);
close(fence);
virtio_gpu_seqno_sync(vgdev, req);
}
return r;
}
int virtio_gpu_create_blob(struct virtio_gpu_device* vgdev,
struct drm_virtgpu_resource_create_blob* args) {
return virtio_gpu_ioctl(vgdev->fd, VIRTGPU_RESOURCE_CREATE_BLOB, args);
}
int virtio_gpu_destroy_handle(struct virtio_gpu_device* vgdev, uint32_t bo_handle) {
struct drm_gem_close args = {
.handle = bo_handle,
};
return virtio_gpu_ioctl(vgdev->fd, GEM_CLOSE, &args);
}
int virtio_gpu_res_id(struct virtio_gpu_device* vgdev, uint32_t handle, uint32_t* res_id) {
struct drm_virtgpu_resource_info args = {
.bo_handle = handle,
};
int r = virtio_gpu_ioctl(vgdev->fd, VIRTGPU_RESOURCE_INFO, &args);
if (r) return r;
*res_id = args.res_handle;
return 0;
}
static int virtio_gpu_get_capset(int fd, struct virgl_renderer_capset_hsakmt* caps) {
struct drm_virtgpu_get_caps args = {
.cap_set_id = VIRGL_RENDERER_CAPSET_HSAKMT,
.cap_set_ver = 0,
.addr = (uintptr_t)caps,
.size = sizeof(*caps),
};
memset(caps, 0, sizeof(*caps));
return virtio_gpu_ioctl(fd, VIRTGPU_GET_CAPS, &args);
}
int virtio_gpu_kfd_open(void) {
drmDevicePtr devices[VHSA_MAX_DEVICES];
int num_devices = 0;
int i, fd, ret;
num_devices = drmGetDevices2(0, devices, ARRAY_SIZE(devices));
if (num_devices <= 0) return -1;
for (i = 0; i < num_devices; i++) {
fd = open(devices[i]->nodes[DRM_NODE_RENDER], O_RDWR | O_CLOEXEC);
if (fd < 0) continue;
struct virgl_renderer_capset_hsakmt caps;
ret = virtio_gpu_get_capset(fd, &caps);
if (ret || caps.context_type != VIRTGPU_DRM_CONTEXT_AMDGPU) {
close(fd);
fd = -1;
continue;
}
goto out;
}
out:
drmFreeDevices(devices, num_devices);
return fd;
}
@@ -0,0 +1,118 @@
/*
* Copyright 2025 Advanced Micro Devices, Inc.
*
* Permission is hereby granted, free of charge, to any person obtaining a
* copy of this software and associated documentation files (the "Software"),
* to deal in the Software without restriction, including without limitation
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
* and/or sell copies of the Software, and to permit persons to whom the
* Software is furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in
* all copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
* THE COPYRIGHT HOLDER(S) OR AUTHOR(S) BE LIABLE FOR ANY CLAIM, DAMAGES OR
* OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
* ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR
* OTHER DEALINGS IN THE SOFTWARE.
*/
#ifndef VIRTIO_GPU_H
#define VIRTIO_GPU_H
#include <pthread.h>
#include <stdint.h>
#include <xf86drm.h>
#include "virtgpu_drm.h"
#define VIRGL_RENDERER_CAPSET_HSAKMT 8
#define VIRTGPU_DRM_CONTEXT_AMDGPU 1
#define VHSA_MAX_DEVICES 10
#ifndef ARRAY_SIZE
#define ARRAY_SIZE(x) (sizeof(x) / sizeof((x)[0]))
#endif
#define VHSA_ALIGN_UP(x, align) (((uint64_t)(x) + (align)-1) & ~(uint64_t)((align)-1))
#define VHSA_ALIGN_DOWN(x, align) ((uint64_t)(x) & ~(uint64_t)((align)-1))
#define virtio_gpu_ioctl(fd, name, args) \
({ \
int ret = drmIoctl((fd), DRM_IOCTL_##name, (args)); \
ret; \
})
struct virgl_renderer_capset_hsakmt {
uint32_t wire_format_version;
/* Underlying drm device version: */
uint32_t version_major;
uint32_t version_minor;
uint32_t version_patchlevel;
uint32_t context_type;
uint32_t pad;
};
struct virtio_gpu_shmem_base {
uint32_t seqno;
uint32_t rsp_mem_offset;
};
struct virtio_gpu_ccmd_req {
uint32_t cmd;
uint32_t len;
uint32_t seqno;
uint32_t rsp_off;
};
struct virtio_gpu_ccmd_rsp {
uint32_t len;
};
struct virtio_gpu_shmem {
struct virtio_gpu_shmem_base base;
uint32_t async_error;
uint32_t global_faults;
};
#define vhsakmt_shmem virtio_gpu_shmem
#define vhsakmt_ccmd_req virtio_gpu_ccmd_req
#define vhsakmt_ccmd_rsp virtio_gpu_ccmd_rsp
struct virtio_gpu_device {
int fd;
struct virtio_gpu_shmem* shmem;
uint32_t shmem_handle;
uint8_t* rsp_mem;
uint32_t rsp_mem_len;
uint32_t next_rsp_off;
pthread_mutex_t rsp_lock;
pthread_mutex_t eb_lock;
uint32_t next_seqno;
uint32_t reqbuf_len;
uint32_t reqbuf_cnt;
uint8_t* reqbuf;
};
struct virtio_gpu_device* virtio_gpu_init(int fd, uint32_t context_id);
void virtio_gpu_close(struct virtio_gpu_device* vgdev);
int virtio_gpu_exec_cmd(struct virtio_gpu_device* vgdev, struct virtio_gpu_ccmd_req* req,
bool sync);
void* virtio_gpu_alloc_rsp(struct virtio_gpu_device* vgdev, struct virtio_gpu_ccmd_req* req,
uint32_t size);
int virtio_gpu_map_handle(struct virtio_gpu_device* vgdev, uint32_t handle, uint64_t size,
void** addr, void* fixed_map);
void virtio_gpu_unmap(void* addr, uint64_t size);
int virtio_gpu_create_blob(struct virtio_gpu_device* vgdev,
struct drm_virtgpu_resource_create_blob* args);
int virtio_gpu_destroy_handle(struct virtio_gpu_device* vgdev, uint32_t bo_handle);
int virtio_gpu_res_id(struct virtio_gpu_device* vgdev, uint32_t handle, uint32_t* res_id);
int virtio_gpu_kfd_open(void);
#endif /* VIRTIO_GPU_H */
@@ -325,22 +325,6 @@ FILTER[gfx1036]=\
"$BLACKLIST_ALL_ASICS:"\
"$BLACKLIST_GFX10_NV2X"
FILTER[gfx940]=\
"$BLACKLIST_ALL_ASICS:"\
"KFDMemoryTest.LargestSysBufferTest:"\
"KFDMemoryTest.BigSysBufferStressTest:"\
"KFDMemoryTest.FlatScratchAccess:"\
"KFDIPCTest.BasicTest:"\
"KFDQMTest.QueueLatency"
FILTER[gfx941]=\
"$BLACKLIST_ALL_ASICS:"\
"KFDMemoryTest.LargestSysBufferTest:"\
"KFDMemoryTest.BigSysBufferStressTest:"\
"KFDMemoryTest.FlatScratchAccess:"\
"KFDIPCTest.BasicTest:"\
"KFDQMTest.QueueLatency"
FILTER[gfx942]=\
"$BLACKLIST_ALL_ASICS:"\
"KFDMemoryTest.LargestSysBufferTest:"\
@@ -43,7 +43,7 @@ static const std::vector<uint32_t> TargetList = {
0x090009,
0x09000a,
0x09000c,
0x090400,
0x090402,
0x0a0100,
0x0a0101,
0x0a0102,
@@ -204,6 +204,11 @@ set ( SRCS core/driver/driver.cpp
libamdhsacode/amd_hsa_code.cpp
libamdhsacode/amd_core_dump.cpp )
if ( BUILD_THUNK_VIRTIO )
list(APPEND SRCS core/driver/virtio/amd_kfd_virtio_driver.cpp)
target_compile_definitions(hsa-runtime64 PRIVATE HSAKMT_VIRTIO_ENABLED=1)
endif()
target_sources( ${CORE_RUNTIME_TARGET} PRIVATE ${SRCS} )
## Depend on trap handler target.
@@ -302,6 +307,10 @@ target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE elf::elf dl pthread rt )
# Link to hsakmt-staticdrm target for static library builds
if( BUILD_SHARED_LIBS )
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt::hsakmt PkgConfig::drm)
if( BUILD_THUNK_VIRTIO )
message(STATUS "Building with virtio support")
target_link_libraries ( ${CORE_RUNTIME_TARGET} PRIVATE hsakmt_virtio)
endif()
find_package(rocprofiler-register)
if(rocprofiler-register_FOUND)
target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE HSA_ROCPROFILER_REGISTER=1
@@ -0,0 +1,514 @@
////////////////////////////////////////////////////////////////////////////////
//
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2024-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
// AMD Research and AMD HSA Software Development
//
// Advanced Micro Devices, Inc.
//
// www.amd.com
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to
// deal with the Software without restriction, including without limitation
// the rights to use, copy, modify, merge, publish, distribute, sublicense,
// and/or sell copies of the Software, and to permit persons to whom the
// Software is furnished to do so, subject to the following conditions:
//
// - Redistributions of source code must retain the above copyright notice,
// this list of conditions and the following disclaimers.
// - Redistributions in binary form must reproduce the above copyright
// notice, this list of conditions and the following disclaimers in
// the documentation and/or other materials provided with the distribution.
// - Neither the names of Advanced Micro Devices, Inc,
// nor the names of its contributors may be used to endorse or promote
// products derived from this Software without specific prior written
// permission.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
// THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
// OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
// ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
// DEALINGS WITH THE SOFTWARE.
//
////////////////////////////////////////////////////////////////////////////////
#include "core/inc/amd_virtio_driver.h"
#include "hsakmt/hsakmt_virtio.h"
#include <link.h>
#include <vector>
#include "core/inc/amd_gpu_agent.h"
#include "core/inc/amd_memory_region.h"
#include "core/inc/runtime.h"
extern r_debug _amdgpu_r_debug;
namespace rocr {
namespace AMD {
KfdVirtioDriver::KfdVirtioDriver(std::string devnode_name)
: core::Driver(core::DriverType::KFD_VIRTIO, std::move(devnode_name)) {}
hsa_status_t KfdVirtioDriver::DiscoverDriver(std::unique_ptr<core::Driver>& driver) {
auto tmp_driver = std::unique_ptr<core::Driver>(new KfdVirtioDriver(""));
if (tmp_driver->Open() == HSA_STATUS_SUCCESS) {
driver = std::move(tmp_driver);
return HSA_STATUS_SUCCESS;
}
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::Open() {
return vhsaKmtOpenKFD() == HSAKMT_STATUS_SUCCESS ? HSA_STATUS_SUCCESS : HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::Close() {
return vhsaKmtCloseKFD() == HSAKMT_STATUS_SUCCESS ? HSA_STATUS_SUCCESS : HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::Init() {
HSAKMT_STATUS ret =
vhsaKmtRuntimeEnable(&_amdgpu_r_debug, core::Runtime::runtime_singleton_->flag().debug());
uint32_t caps_mask = 0;
if (ret != HSAKMT_STATUS_SUCCESS && ret != HSAKMT_STATUS_NOT_SUPPORTED) return HSA_STATUS_ERROR;
if (vhsaKmtGetRuntimeCapabilities(&caps_mask) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
core::Runtime::runtime_singleton_->KfdVersion(
ret != HSAKMT_STATUS_NOT_SUPPORTED,
!!(caps_mask & HSA_RUNTIME_ENABLE_CAPS_SUPPORTS_CORE_DUMP_MASK));
if (vhsaKmtGetVersion(&version_) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
core::Runtime::runtime_singleton_->KfdVersion(version_);
if (version_.KernelInterfaceMajorVersion == 1 && version_.KernelInterfaceMinorVersion == 0)
core::g_use_interrupt_wait = false;
/* Force disable interrupt wait in VIRTIO driver temporarily */
core::g_use_interrupt_wait = false;
/* Force disable XNACK in VIRTIO driver temporarily */
core::Runtime::runtime_singleton_->XnackEnabled(false);
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::ShutDown() {
HSAKMT_STATUS ret = vhsaKmtRuntimeDisable();
if (ret != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
ret = vhsaKmtReleaseSystemProperties();
if (ret != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
return Close();
}
hsa_status_t KfdVirtioDriver::QueryKernelModeDriver(core::DriverQuery query) {
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::GetSystemProperties(HsaSystemProperties& sys_props) const {
if (vhsaKmtAcquireSystemProperties(&sys_props) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::GetNodeProperties(HsaNodeProperties& node_props,
uint32_t node_id) const {
if (vhsaKmtGetNodeProperties(node_id, &node_props) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::GetEdgeProperties(std::vector<HsaIoLinkProperties>& io_link_props,
uint32_t node_id) const {
if (vhsaKmtGetNodeIoLinkProperties(node_id, io_link_props.size(), io_link_props.data()) !=
HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::GetMemoryProperties(
uint32_t node_id, std::vector<HsaMemoryProperties>& mem_props) const {
if (mem_props.empty()) {
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
if (vhsaKmtGetNodeMemoryProperties(node_id, mem_props.size(), mem_props.data()) !=
HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::GetCacheProperties(
uint32_t node_id, uint32_t processor_id, std::vector<HsaCacheProperties>& cache_props) const {
if (cache_props.empty()) {
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
if (vhsaKmtGetNodeCacheProperties(node_id, 0, cache_props.size(), cache_props.data()) !=
HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::GetDeviceHandle(uint32_t node_id, void** device_handle) const {
assert(device_handle != nullptr);
if (vhsaKmtGetAMDGPUDeviceHandle(node_id, device_handle) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::GetClockCounters(uint32_t node_id,
HsaClockCounters* clock_counter) const {
assert(clock_counter != nullptr);
if (vhsaKmtGetClockCounters(node_id, clock_counter) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::SetTrapHandler(uint32_t node_id, const void* base, uint64_t base_size,
const void* buffer_base,
uint64_t buffer_base_size) const {
if (vhsaKmtSetTrapHandler(node_id, const_cast<void*>(base), base_size,
const_cast<void*>(buffer_base),
buffer_base_size) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::AllocateMemory(const core::MemoryRegion& mem_region,
core::MemoryRegion::AllocateFlags alloc_flags,
void** mem, size_t size, uint32_t agent_node_id) {
const MemoryRegion& m_region(static_cast<const MemoryRegion&>(mem_region));
HsaMemFlags kmt_alloc_flags(m_region.mem_flags());
HSAKMT_STATUS ret;
kmt_alloc_flags.ui32.ExecuteAccess =
(alloc_flags & core::MemoryRegion::AllocateExecutable ? 1 : 0);
kmt_alloc_flags.ui32.AQLQueueMemory =
(alloc_flags & core::MemoryRegion::AllocateDoubleMap ? 1 : 0);
if (m_region.IsSystem() && (alloc_flags & core::MemoryRegion::AllocateNonPaged)) {
kmt_alloc_flags.ui32.NonPaged = 1;
}
if (!m_region.IsLocalMemory() && (alloc_flags & core::MemoryRegion::AllocateMemoryOnly)) {
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
// Allocating a memory handle for virtual memory
kmt_alloc_flags.ui32.NoAddress = !!(alloc_flags & core::MemoryRegion::AllocateMemoryOnly);
// Allocate pseudo fine grain memory
kmt_alloc_flags.ui32.CoarseGrain =
(alloc_flags & core::MemoryRegion::AllocatePCIeRW ? 0 : kmt_alloc_flags.ui32.CoarseGrain);
kmt_alloc_flags.ui32.NoSubstitute =
(alloc_flags & core::MemoryRegion::AllocatePinned ? 1 : kmt_alloc_flags.ui32.NoSubstitute);
kmt_alloc_flags.ui32.GTTAccess =
(alloc_flags & core::MemoryRegion::AllocateGTTAccess ? 1 : kmt_alloc_flags.ui32.GTTAccess);
kmt_alloc_flags.ui32.Uncached =
(alloc_flags & core::MemoryRegion::AllocateUncached ? 1 : kmt_alloc_flags.ui32.Uncached);
if (m_region.IsLocalMemory()) {
// Allocate physically contiguous memory. AllocateKfdMemory function call
// will fail if this flag is not supported in KFD.
kmt_alloc_flags.ui32.Contiguous =
(alloc_flags & core::MemoryRegion::AllocateContiguous ? 1
: kmt_alloc_flags.ui32.Contiguous);
}
//// Only allow using the suballocator for ordinary VRAM.
if (m_region.IsLocalMemory() && !kmt_alloc_flags.ui32.NoAddress) {
bool subAllocEnabled = !core::Runtime::runtime_singleton_->flag().disable_fragment_alloc();
// Avoid modifying executable or queue allocations.
bool useSubAlloc = subAllocEnabled;
useSubAlloc &= ((alloc_flags & (~core::MemoryRegion::AllocateRestrict)) == 0);
if (useSubAlloc) {
*mem = m_region.fragment_alloc(size);
if ((alloc_flags & core::MemoryRegion::AllocateAsan)) {
// TODO: Implement ASAN support for VIRTIO driver
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
return HSA_STATUS_SUCCESS;
}
}
const uint32_t node_id = (alloc_flags & core::MemoryRegion::AllocateGTTAccess)
? agent_node_id
: m_region.owner()->node_id();
//// Allocate memory.
//// If it fails attempt to release memory from the block allocator and retry.
ret = vhsaKmtAllocMemory(node_id, size, kmt_alloc_flags, mem);
if (ret != HSAKMT_STATUS_SUCCESS) {
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
if (*mem == nullptr) {
m_region.owner()->Trim();
ret = vhsaKmtAllocMemory(node_id, size, kmt_alloc_flags, mem);
if (ret != HSAKMT_STATUS_SUCCESS) {
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
}
if (*mem != nullptr) {
if (kmt_alloc_flags.ui32.NoAddress) return HSA_STATUS_SUCCESS;
// Commit the memory.
// For system memory, on non-restricted allocation, map it to all GPUs. On
// restricted allocation, only CPU is allowed to access by default, so
// no need to map
// For local memory, only map it to the owning GPU. Mapping to other GPU,
// if the access is allowed, is performed on AllowAccess.
HsaMemMapFlags map_flag = m_region.map_flags();
size_t map_node_count = 1;
const uint32_t owner_node_id = m_region.owner()->node_id();
const uint32_t* map_node_id = &owner_node_id;
if (m_region.IsSystem()) {
if ((alloc_flags & core::MemoryRegion::AllocateRestrict) == 0) {
// Map to all GPU agents.
map_node_count = core::Runtime::runtime_singleton_->gpu_ids().size();
if (map_node_count == 0) {
// No need to pin since no GPU in the platform.
return HSA_STATUS_SUCCESS;
}
map_node_id = &core::Runtime::runtime_singleton_->gpu_ids()[0];
} else {
// No need to pin it for CPU exclusive access.
return HSA_STATUS_SUCCESS;
}
}
uint64_t alternate_va = 0;
const bool is_resident =
(MakeMemoryResident(*mem, size, &alternate_va, &map_flag, map_node_count, map_node_id) ==
HSA_STATUS_SUCCESS);
const bool require_pinning =
(!m_region.full_profile() || m_region.IsLocalMemory() || m_region.IsScratch());
if (require_pinning && !is_resident) {
vhsaKmtFreeMemory(*mem, size);
*mem = nullptr;
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
if ((alloc_flags & core::MemoryRegion::AllocateAsan)) {
// TODO: Implement ASAN support for VIRTIO driver
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
return HSA_STATUS_SUCCESS;
}
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
hsa_status_t KfdVirtioDriver::FreeMemory(void* mem, size_t size) {
MakeMemoryUnresident(mem);
return vhsaKmtFreeMemory(mem, size) == HSAKMT_STATUS_SUCCESS ? HSA_STATUS_SUCCESS
: HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::AllocateScratchMemory(uint32_t node_id, uint64_t size,
void** mem) const {
assert(mem != nullptr);
assert(size != 0);
HsaMemFlags flags = {};
flags.ui32.Scratch = 1;
flags.ui32.HostAccess = 1;
void* ptr = nullptr;
HSAKMT_STATUS ret = vhsaKmtAllocMemory(node_id, size, flags, &ptr);
if (ret != HSAKMT_STATUS_SUCCESS || ptr == nullptr) return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
*mem = ptr;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::RegisterMemory(void* ptr, uint64_t size,
HsaMemFlags mem_flags) const {
assert(ptr != nullptr);
assert(size != 0);
if (vhsaKmtRegisterMemoryWithFlags(ptr, size, mem_flags) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::DeregisterMemory(void* ptr) const {
if (vhsaKmtDeregisterMemory(ptr) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::AvailableMemory(uint32_t node_id, uint64_t* available_size) const {
assert(available_size != nullptr);
if (vhsaKmtAvailableMemory(node_id, available_size) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::MakeMemoryResident(const void* mem, size_t size,
uint64_t* alternate_va,
const HsaMemMapFlags* mem_flags,
uint32_t num_nodes, const uint32_t* nodes) const {
assert(mem != nullptr);
assert(size != 0);
if (mem_flags == nullptr && nodes == nullptr) {
if (vhsaKmtMapMemoryToGPU(const_cast<void*>(mem), size, alternate_va) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
} else if (mem_flags != nullptr && nodes != nullptr) {
if (vhsaKmtMapMemoryToGPUNodes(const_cast<void*>(mem), size, alternate_va, *mem_flags,
num_nodes,
const_cast<uint32_t*>(nodes)) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
} else {
debug_print("Invalid memory flags ptr:%p nodes ptr:%p\n", mem_flags, nodes);
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::MakeMemoryUnresident(const void* mem) const {
vhsaKmtUnmapMemoryToGPU(const_cast<void*>(mem));
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::CreateQueue(uint32_t node_id, HSA_QUEUE_TYPE type, uint32_t queue_pct,
HSA_QUEUE_PRIORITY priority, uint32_t sdma_engine_id,
void* queue_addr, uint64_t queue_size_bytes,
HsaEvent* event, HsaQueueResource& queue_resource) const {
if (vhsaKmtCreateQueueExt(node_id, type, queue_pct, priority, sdma_engine_id, queue_addr,
queue_size_bytes, event, &queue_resource) != HSAKMT_STATUS_SUCCESS)
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::DestroyQueue(HSA_QUEUEID queue_id) const {
if (vhsaKmtDestroyQueue(queue_id) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::UpdateQueue(HSA_QUEUEID queue_id, uint32_t queue_percentage,
HSA_QUEUE_PRIORITY priority, void* queue_mem,
uint64_t queue_size, HsaEvent* event) const {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::SetQueueCUMask(HSA_QUEUEID queue_id, uint32_t num_cu_mask,
uint32_t* cu_mask) const {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::AllocQueueGWS(HSA_QUEUEID queue_id, uint32_t num_GWS,
uint32_t* GWS) const {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::ExportDMABuf(void* mem, size_t size, int* dmabuf_fd, size_t* offset) {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::ImportDMABuf(int dmabuf_fd, core::Agent& agent,
core::ShareableHandle& handle) {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::Map(core::ShareableHandle handle, void* mem, size_t offset,
size_t size, hsa_access_permission_t perms) {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::Unmap(core::ShareableHandle handle, void* mem, size_t offset,
size_t size) {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::ReleaseShareableHandle(core::ShareableHandle& handle) {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::GetTileConfig(uint32_t node_id, HsaGpuTileConfig* config) const {
if (vhsaKmtGetTileConfig(node_id, config) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::SPMAcquire(uint32_t node_id) const { return HSA_STATUS_ERROR; }
hsa_status_t KfdVirtioDriver::SPMRelease(uint32_t node_id) const { return HSA_STATUS_ERROR; }
hsa_status_t KfdVirtioDriver::SPMSetDestBuffer(uint32_t node_id, uint32_t size, uint32_t* timeout,
uint32_t* size_copied, void* dest,
bool* is_data_loss) const {
return HSA_STATUS_ERROR;
}
hsa_status_t KfdVirtioDriver::OpenSMI(uint32_t node_id, int* fd) const { return HSA_STATUS_ERROR; }
hsa_status_t KfdVirtioDriver::GetWallclockFrequency(uint32_t node_id, uint64_t* frequency) const {
assert(frequency != nullptr);
amdgpu_gpu_info info;
amdgpu_device_handle handle;
if (GetDeviceHandle(node_id, reinterpret_cast<void**>(&handle)) != HSA_STATUS_SUCCESS)
return HSA_STATUS_ERROR;
if (vamdgpu_query_gpu_info(handle, &info) < 0) return HSA_STATUS_ERROR;
// Reported by libdrm in KHz.
*frequency = uint64_t(info.gpu_counter_freq) * 1000ull;
return HSA_STATUS_SUCCESS;
}
hsa_status_t KfdVirtioDriver::IsModelEnabled(bool* enable) const {
*enable = false;
return HSA_STATUS_SUCCESS;
}
} // namespace AMD
} // namespace rocr
@@ -332,10 +332,24 @@ hsa_status_t XdnaDriver::AllocQueueGWS(HSA_QUEUEID queue_id, uint32_t num_gws,
return HSA_STATUS_ERROR_INVALID_QUEUE;
}
hsa_status_t XdnaDriver::ExportDMABuf(void *mem, size_t size, int *dmabuf_fd,
size_t *offset) {
// Not implemented yet.
return HSA_STATUS_ERROR;
hsa_status_t XdnaDriver::ExportDMABuf(void* mem, size_t size, int* dmabuf_fd, size_t* offset) {
auto bo_handle = FindBOHandle(mem);
if (!bo_handle.IsValid()) {
return HSA_STATUS_ERROR_INVALID_ALLOCATION;
}
drm_prime_handle export_params = {};
export_params.handle = bo_handle.handle;
export_params.flags = DRM_RDWR;
export_params.fd = -1;
if (ioctl(fd_, DRM_IOCTL_PRIME_HANDLE_TO_FD, &export_params) < 0) {
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
*dmabuf_fd = export_params.fd;
*offset = reinterpret_cast<uintptr_t>(mem) - reinterpret_cast<uintptr_t>(bo_handle.vaddr);
return HSA_STATUS_SUCCESS;
}
hsa_status_t XdnaDriver::ImportDMABuf(int dmabuf_fd, core::Agent &agent,
@@ -73,11 +73,7 @@ class BlitSdmaBase : public core::Blit {
core::Signal& out_signal) = 0;
};
// RingIndexTy: 32/64-bit monotonic ring index, counting in bytes.
// HwIndexMonotonic: true if SDMA HW index is monotonic, false if it wraps at end of ring.
// SizeToCountOffset: value added to size (in bytes) to form SDMA command count field.
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
class BlitSdma : public BlitSdmaBase {
template <bool useGCR> class BlitSdma : public BlitSdmaBase {
public:
BlitSdma();
@@ -163,9 +159,9 @@ class BlitSdma : public BlitSdmaBase {
/// could be written. NULL if input size is greater than the size of queue
/// buffer.
char* AcquireWriteAddress(uint32_t cmd_size, RingIndexTy& curr_index);
char* AcquireWriteAddress(uint32_t cmd_size, uint64_t& curr_index);
void UpdateWriteAndDoorbellRegister(RingIndexTy curr_index, RingIndexTy new_index);
void UpdateWriteAndDoorbellRegister(uint64_t curr_index, uint64_t new_index);
/// @brief Updates the Write Register of compute device to the end of
/// SDMA packet written into queue buffer. The update to Write Register
@@ -178,16 +174,16 @@ class BlitSdma : public BlitSdmaBase {
/// @param curr_index Index passed back from AcquireWriteAddress.
///
/// @param cmd_size Command packet size in bytes.
void ReleaseWriteAddress(RingIndexTy curr_index, uint32_t cmd_size);
void ReleaseWriteAddress(uint64_t curr_index, uint32_t cmd_size);
/// @brief Writes NO-OP words into queue buffer in case writing a command
/// causes the queue buffer to wrap.
///
/// @param curr_index Index to begin padding from.
void PadRingToEnd(RingIndexTy curr_index);
void PadRingToEnd(uint64_t curr_index);
uint32_t WrapIntoRing(RingIndexTy index);
bool CanWriteUpto(RingIndexTy upto_index);
uint32_t WrapIntoRing(uint64_t index);
bool CanWriteUpto(uint64_t upto_index);
/// @brief Build fence command
void BuildFenceCommand(char* fence_command_addr, uint32_t* fence,
@@ -265,8 +261,8 @@ class BlitSdma : public BlitSdmaBase {
HsaQueueResource queue_resource_;
// Monotonic ring indices, in bytes, tracking written and submitted commands.
RingIndexTy cached_reserve_index_;
RingIndexTy cached_commit_index_;
uint64_t cached_reserve_index_;
uint64_t cached_commit_index_;
static const uint32_t linear_copy_command_size_;
@@ -314,21 +310,11 @@ class BlitSdma : public BlitSdmaBase {
size_t min_submission_size_;
};
// Ring indices are 32-bit.
// HW ring indices are not monotonic (wrap at end of ring).
// Count fields of SDMA commands are 0-based.
typedef BlitSdma<uint32_t, false, 0, false> BlitSdmaV2V3;
// Ring indices are 64-bit.
// HW ring indices are monotonic (do not wrap at end of ring).
// Count fields of SDMA commands are 1-based.
typedef BlitSdma<uint64_t, true, -1, false> BlitSdmaV4;
typedef BlitSdma<false> BlitSdmaV4;
// Ring indices are 64-bit.
// HW ring indices are monotonic (do not wrap at end of ring).
// Count fields of SDMA commands are 1-based.
// SDMA is connected to gL2.
typedef BlitSdma<uint64_t, true, -1, true> BlitSdmaV5;
typedef BlitSdma<true> BlitSdmaV5;
} // namespace amd
} // namespace rocr
@@ -51,6 +51,7 @@
#include "core/inc/agent.h"
#include "core/inc/queue.h"
#include "core/inc/cache.h"
#include "core/inc/driver.h"
namespace rocr {
namespace AMD {
@@ -62,7 +63,9 @@ class CpuAgent : public core::Agent {
// @param [in] node Node id. Each CPU in different socket will get distinct
// id.
// @param [in] node_props Node property.
CpuAgent(HSAuint32 node, const HsaNodeProperties& node_props);
// @param [in] driver_type Driver type. Default is KFD.
CpuAgent(HSAuint32 node, const HsaNodeProperties& node_props,
core::DriverType driver_type = core::DriverType::KFD);
// @brief CpuAgent destructor.
~CpuAgent();
@@ -73,10 +73,11 @@ typedef ScratchCache::ScratchInfo ScratchInfo;
class GpuAgentInt : public core::Agent {
public:
// @brief Constructor
GpuAgentInt(uint32_t node_id)
: core::Agent(core::Runtime::runtime_singleton_->AgentDriver(
core::DriverType::KFD),
node_id, core::Agent::DeviceType::kAmdGpuDevice) {}
// @param [in] node_id Node id.
// @param [in] driver_type Driver type. Default is KFD.
GpuAgentInt(uint32_t node_id, core::DriverType driver_type)
: core::Agent(core::Runtime::runtime_singleton_->AgentDriver(driver_type), node_id,
core::Agent::DeviceType::kAmdGpuDevice) {}
// @brief Ensure blits are ready (performance hint).
virtual void PreloadBlits() {}
@@ -231,7 +232,10 @@ class GpuAgent : public GpuAgentInt {
// id.
// @param [in] node_props Node property.
// @param [in] xnack_mode XNACK mode of device.
GpuAgent(HSAuint32 node, const HsaNodeProperties& node_props, bool xnack_mode, uint32_t index);
// @param [in] index Index of the GPU device.
// @param [in] driver_type Driver type. Default is KFD.
GpuAgent(HSAuint32 node, const HsaNodeProperties& node_props, bool xnack_mode, uint32_t index,
core::DriverType driver_type = core::DriverType::KFD);
// @brief GPU agent destructor.
~GpuAgent();
@@ -721,9 +725,6 @@ class GpuAgent : public GpuAgentInt {
// @brief Alternative aperture base address. Only on KV.
uintptr_t ape1_base_;
// @brief Alternative aperture size. Only on KV.
size_t ape1_size_;
// @brief Queue with GWS access.
struct {
lazy_ptr<core::Queue> queue_;
@@ -89,7 +89,7 @@
# define PM4_ACQUIRE_MEM_GCR_CNTL_GLV_INV (1 << 8)
# define PM4_ACQUIRE_MEM_GCR_CNTL_GL1_INV (1 << 9)
# define PM4_ACQUIRE_MEM_GCR_CNTL_GL2_INV (1 << 14)
# define PM4_ACQUIRE_MEM_GCR_CNTL_GL2_WB (1 << 15)
#define PM4_RELEASE_MEM_DW1_EVENT_INDEX(x) (((x) & 0xF) << 8)
# define PM4_RELEASE_MEM_EVENT_INDEX_AQL 0x7
@@ -0,0 +1,124 @@
////////////////////////////////////////////////////////////////////////////////
//
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2024-2025, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
// AMD Research and AMD HSA Software Development
//
// Advanced Micro Devices, Inc.
//
// www.amd.com
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to
// deal with the Software without restriction, including without limitation
// the rights to use, copy, modify, merge, publish, distribute, sublicense,
// and/or sell copies of the Software, and to permit persons to whom the
// Software is furnished to do so, subject to the following conditions:
//
// - Redistributions of source code must retain the above copyright notice,
// this list of conditions and the following disclaimers.
// - Redistributions in binary form must reproduce the above copyright
// notice, this list of conditions and the following disclaimers in
// the documentation and/or other materials provided with the distribution.
// - Neither the names of Advanced Micro Devices, Inc,
// nor the names of its contributors may be used to endorse or promote
// products derived from this Software without specific prior written
// permission.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
// THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
// OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
// ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
// DEALINGS WITH THE SOFTWARE.
//
////////////////////////////////////////////////////////////////////////////////
#ifndef HSA_RUNTIME_CORE_INC_AMD_VIRTIO_DRIVER_H_
#define HSA_RUNTIME_CORE_INC_AMD_VIRTIO_DRIVER_H_
#include <memory>
#include <string>
#include "hsakmt/hsakmt.h"
#include "core/inc/driver.h"
#include "core/inc/memory_region.h"
namespace rocr {
namespace AMD {
class KfdVirtioDriver final : public core::Driver {
public:
KfdVirtioDriver(std::string devnode_name);
static hsa_status_t DiscoverDriver(std::unique_ptr<core::Driver>& driver);
hsa_status_t Init() override;
hsa_status_t ShutDown() override;
hsa_status_t QueryKernelModeDriver(core::DriverQuery query) override;
hsa_status_t Open() override;
hsa_status_t Close() override;
hsa_status_t GetSystemProperties(HsaSystemProperties& sys_props) const override;
hsa_status_t GetNodeProperties(HsaNodeProperties& node_props, uint32_t node_id) const override;
hsa_status_t GetEdgeProperties(std::vector<HsaIoLinkProperties>& io_link_props,
uint32_t node_id) const override;
hsa_status_t GetMemoryProperties(uint32_t node_id,
std::vector<HsaMemoryProperties>& mem_props) const override;
hsa_status_t GetCacheProperties(uint32_t node_id, uint32_t processor_id,
std::vector<HsaCacheProperties>& cache_props) const override;
hsa_status_t GetDeviceHandle(uint32_t node_id, void** device_handle) const;
hsa_status_t GetClockCounters(uint32_t node_id, HsaClockCounters* clock_counter) const;
hsa_status_t SetTrapHandler(uint32_t node_id, const void* base, uint64_t base_size,
const void* buffer_base, uint64_t buffer_base_size) const;
hsa_status_t AllocateMemory(const core::MemoryRegion& mem_region,
core::MemoryRegion::AllocateFlags alloc_flags, void** mem,
size_t size, uint32_t agent_node_id) override;
hsa_status_t FreeMemory(void* mem, size_t size) override;
hsa_status_t AllocateScratchMemory(uint32_t node_id, uint64_t size, void** mem) const;
hsa_status_t RegisterMemory(void* ptr, uint64_t size, HsaMemFlags mem_flags) const override;
hsa_status_t DeregisterMemory(void* ptr) const override;
hsa_status_t AvailableMemory(uint32_t node_id, uint64_t* available_size) const;
hsa_status_t MakeMemoryResident(const void* mem, size_t size, uint64_t* alternate_va,
const HsaMemMapFlags* mem_flags, uint32_t num_nodes,
const uint32_t* nodes) const override;
hsa_status_t MakeMemoryUnresident(const void* mem) const override;
hsa_status_t CreateQueue(uint32_t node_id, HSA_QUEUE_TYPE type, uint32_t queue_pct,
HSA_QUEUE_PRIORITY priority, uint32_t sdma_engine_id, void* queue_addr,
uint64_t queue_size_bytes, HsaEvent* event,
HsaQueueResource& queue_resource) const override;
hsa_status_t DestroyQueue(HSA_QUEUEID queue_id) const override;
hsa_status_t UpdateQueue(HSA_QUEUEID queue_id, uint32_t queue_percentage,
HSA_QUEUE_PRIORITY priority, void* queue_mem, uint64_t queue_size,
HsaEvent* event) const override;
hsa_status_t SetQueueCUMask(HSA_QUEUEID queue_id, uint32_t num_cu_mask,
uint32_t* cu_mask) const override;
hsa_status_t AllocQueueGWS(HSA_QUEUEID queue_id, uint32_t num_GWS, uint32_t* GWS) const override;
hsa_status_t ExportDMABuf(void* mem, size_t size, int* dmabuf_fd, size_t* offset) override;
hsa_status_t ImportDMABuf(int dmabuf_fd, core::Agent& agent,
core::ShareableHandle& handle) override;
hsa_status_t Map(core::ShareableHandle handle, void* mem, size_t offset, size_t size,
hsa_access_permission_t perms) override;
hsa_status_t Unmap(core::ShareableHandle handle, void* mem, size_t offset, size_t size) override;
hsa_status_t ReleaseShareableHandle(core::ShareableHandle& handle) override;
hsa_status_t GetTileConfig(uint32_t node_id, HsaGpuTileConfig* config) const;
hsa_status_t SPMAcquire(uint32_t node_id) const override;
hsa_status_t SPMRelease(uint32_t node_id) const override;
hsa_status_t SPMSetDestBuffer(uint32_t node_id, uint32_t size, uint32_t* timeout,
uint32_t* size_copied, void* dest,
bool* is_data_loss) const override;
hsa_status_t OpenSMI(uint32_t node_id, int* fd) const override;
hsa_status_t GetWallclockFrequency(uint32_t node_id, uint64_t* frequency) const;
hsa_status_t IsModelEnabled(bool* enable) const override;
};
} // namespace AMD
} // namespace rocr
#endif // HSA_RUNTIME_CORE_INC_AMD_VIRTIO_DRIVER_H_
@@ -58,7 +58,14 @@ class Queue;
enum class DriverQuery { GET_DRIVER_VERSION };
enum class DriverType { XDNA = 0, KFD, NUM_DRIVER_TYPES };
enum class DriverType {
XDNA = 0,
KFD,
#ifdef HSAKMT_VIRTIO_ENABLED
KFD_VIRTIO,
#endif
NUM_DRIVER_TYPES
};
/// @brief Handle for exported / imported memory.
struct ShareableHandle {
@@ -510,6 +510,14 @@ class Runtime {
std::vector<std::unique_ptr<Driver>>& AgentDrivers() { return agent_drivers_; }
static bool IsGPUDriver(DriverType driver_type) {
return driver_type == core::DriverType::KFD
#ifdef HSAKMT_VIRTIO_ENABLED
|| driver_type == core::DriverType::KFD_VIRTIO
#endif
;
}
protected:
static void AsyncEventsLoop(void*);
static void AsyncIPCSockServerConnLoop(void*);
@@ -814,7 +822,6 @@ class Runtime {
std::map<const void*, AddressHandle> reserved_address_map_; // Indexed by VA
struct MemoryHandle {
MemoryHandle() : region(NULL), size(0), ref_count(0), thunk_handle(NULL), alloc_flag(0) {}
MemoryHandle(const MemoryRegion* region, size_t size, uint64_t flags_unused,
ThunkHandle thunk_handle, MemoryRegion::AllocateFlags alloc_flag)
: region(region),
@@ -824,19 +831,23 @@ class Runtime {
thunk_handle(thunk_handle),
alloc_flag(alloc_flag) {}
static __forceinline hsa_amd_vmem_alloc_handle_t Convert(void* handle) {
static __forceinline hsa_amd_vmem_alloc_handle_t Convert(ThunkHandle handle) {
hsa_amd_vmem_alloc_handle_t ret_handle = {
static_cast<uint64_t>(reinterpret_cast<uintptr_t>(handle))};
return ret_handle;
}
static __forceinline ThunkHandle Convert(hsa_amd_vmem_alloc_handle_t handle) {
return reinterpret_cast<void*>(handle.handle);
}
__forceinline core::Agent* agentOwner() const { return region->owner(); }
const MemoryRegion* region;
size_t size;
int ref_count;
int use_count;
ThunkHandle thunk_handle; // handle returned by hsaKmtAllocMemory(NoAddress = 1)
ThunkHandle thunk_handle; // handle returned by Driver::Allocate(NoAddress = 1)
MemoryRegion::AllocateFlags alloc_flag;
};
std::map<ThunkHandle, MemoryHandle> memory_handle_map_;
@@ -888,11 +899,6 @@ class Runtime {
const hsa_amd_memory_access_desc_t *desc,
const size_t desc_cnt);
// Frees runtime memory when the runtime library is unloaded if safe to do so.
// Failure to release the runtime indicates an incorrect application but is
// common (example: calls library routines at process exit).
friend class RuntimeCleanup;
void InitIPCDmaBufSupport();
bool ipc_dmabuf_supported_;
int IPCClientImport(uint32_t conn_handle, uint64_t dmabuf_fd_handle,
@@ -77,44 +77,33 @@ const size_t BlitSdmaBase::kMaxSingleCopySize = SDMA_PKT_COPY_LINEAR::kMaxSize_;
const size_t BlitSdmaBase::kMaxSingleFillSize = SDMA_PKT_CONSTANT_FILL::kMaxSize_;
// Initialize size of various sDMA commands use by this module
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::linear_copy_command_size_ = sizeof(SDMA_PKT_COPY_LINEAR);
template <bool useGCR>
const uint32_t BlitSdma<useGCR>::linear_copy_command_size_ = sizeof(SDMA_PKT_COPY_LINEAR);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::fill_command_size_ = sizeof(SDMA_PKT_CONSTANT_FILL);
template <bool useGCR>
const uint32_t BlitSdma<useGCR>::fill_command_size_ = sizeof(SDMA_PKT_CONSTANT_FILL);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::fence_command_size_ = sizeof(SDMA_PKT_FENCE);
template <bool useGCR>
const uint32_t BlitSdma<useGCR>::fence_command_size_ = sizeof(SDMA_PKT_FENCE);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::poll_command_size_ = sizeof(SDMA_PKT_POLL_REGMEM);
template <bool useGCR>
const uint32_t BlitSdma<useGCR>::poll_command_size_ = sizeof(SDMA_PKT_POLL_REGMEM);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::flush_command_size_ = sizeof(SDMA_PKT_POLL_REGMEM);
template <bool useGCR>
const uint32_t BlitSdma<useGCR>::flush_command_size_ = sizeof(SDMA_PKT_POLL_REGMEM);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::atomic_command_size_ = sizeof(SDMA_PKT_ATOMIC);
template <bool useGCR>
const uint32_t BlitSdma<useGCR>::atomic_command_size_ = sizeof(SDMA_PKT_ATOMIC);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::timestamp_command_size_ = sizeof(SDMA_PKT_TIMESTAMP);
template <bool useGCR>
const uint32_t BlitSdma<useGCR>::timestamp_command_size_ = sizeof(SDMA_PKT_TIMESTAMP);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::trap_command_size_ = sizeof(SDMA_PKT_TRAP);
template <bool useGCR> const uint32_t BlitSdma<useGCR>::trap_command_size_ = sizeof(SDMA_PKT_TRAP);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
const uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::gcr_command_size_ = sizeof(SDMA_PKT_GCR);
template <bool useGCR> const uint32_t BlitSdma<useGCR>::gcr_command_size_ = sizeof(SDMA_PKT_GCR);
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BlitSdma()
template <bool useGCR>
BlitSdma<useGCR>::BlitSdma()
: agent_(NULL),
queue_start_addr_(NULL),
bytes_queued_(0),
@@ -129,12 +118,11 @@ BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BlitSdma()
std::memset(&queue_resource_, 0, sizeof(queue_resource_));
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::~BlitSdma() {}
template <bool useGCR> BlitSdma<useGCR>::~BlitSdma() {}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::Initialize(
const core::Agent& agent, bool use_xgmi, size_t linear_copy_size_override, int rec_eng) {
template <bool useGCR>
hsa_status_t BlitSdma<useGCR>::Initialize(const core::Agent& agent, bool use_xgmi,
size_t linear_copy_size_override, int rec_eng) {
if (queue_start_addr_ != NULL) {
// Already initialized.
return HSA_STATUS_SUCCESS;
@@ -201,7 +189,7 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>:
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
cached_reserve_index_ = *reinterpret_cast<RingIndexTy*>(queue_resource_.Queue_write_ptr);
cached_reserve_index_ = *reinterpret_cast<uint64_t*>(queue_resource_.Queue_write_ptr);
cached_commit_index_ = cached_reserve_index_;
if (core::g_use_interrupt_wait) {
@@ -218,9 +206,7 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>:
return HSA_STATUS_SUCCESS;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::Destroy(
const core::Agent& agent) {
template <bool useGCR> hsa_status_t BlitSdma<useGCR>::Destroy(const core::Agent& agent) {
// Release all allocated resources and reset them to zero.
if (queue_resource_.QueueId != 0) {
@@ -245,9 +231,8 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>:
return HSA_STATUS_SUCCESS;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::SubmitBlockingCommand(const void* cmd, size_t cmd_size,
template <bool useGCR>
hsa_status_t BlitSdma<useGCR>::SubmitBlockingCommand(const void* cmd, size_t cmd_size,
uint64_t size) {
ScopedAcquire<KernelMutex> lock(&lock_);
@@ -278,11 +263,11 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
return ret;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::SubmitCommand(
const void* cmd, size_t cmd_size, uint64_t size, const std::vector<core::Signal*>& dep_signals,
core::Signal& out_signal, std::vector<core::Signal*>& gang_signals) {
template <bool useGCR>
hsa_status_t BlitSdma<useGCR>::SubmitCommand(const void* cmd, size_t cmd_size, uint64_t size,
const std::vector<core::Signal*>& dep_signals,
core::Signal& out_signal,
std::vector<core::Signal*>& gang_signals) {
uint32_t num_poll_command = 0;
// Cached copy of dep_signals[i]->LoadRelaxed
@@ -355,7 +340,7 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>:
// Add space for acquire or release Hdp flush command
uint32_t flush_cmd_size = 0;
if (core::Runtime::runtime_singleton_->flag().enable_sdma_hdp_flush()) {
if ((HwIndexMonotonic) && (hdp_flush_support_)) {
if (hdp_flush_support_) {
flush_cmd_size = flush_command_size_;
}
}
@@ -368,7 +353,7 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>:
const uint32_t pad_size = total_command_size < min_submission_size_ ?
min_submission_size_ - total_command_size : 0;
RingIndexTy curr_index;
uint64_t curr_index;
char* command_addr;
uint64_t prior_bytes, post_bytes;
{
@@ -426,7 +411,7 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>:
// Issue a Hdp flush cmd
if (core::Runtime::runtime_singleton_->flag().enable_sdma_hdp_flush()) {
if ((HwIndexMonotonic) && (hdp_flush_support_)) {
if (hdp_flush_support_) {
BuildHdpFlushCommand(command_addr);
command_addr += flush_command_size_;
bytes_written_[wrapped_index] = prior_bytes;
@@ -542,9 +527,8 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>:
return HSA_STATUS_SUCCESS;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::SubmitLinearCopyCommand(void* dst, const void* src, size_t size) {
template <bool useGCR>
hsa_status_t BlitSdma<useGCR>::SubmitLinearCopyCommand(void* dst, const void* src, size_t size) {
// Break the copy into multiple copy operation incase the copy size exceeds
// the SDMA linear copy limit.
const size_t max_copy_size = max_single_linear_copy_size_ ? max_single_linear_copy_size_ :
@@ -557,9 +541,8 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
return SubmitBlockingCommand(&buff[0], buff.size() * sizeof(SDMA_PKT_COPY_LINEAR), size);
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::SubmitLinearCopyCommand(void* dst, const void* src, size_t size,
template <bool useGCR>
hsa_status_t BlitSdma<useGCR>::SubmitLinearCopyCommand(void* dst, const void* src, size_t size,
std::vector<core::Signal*>& dep_signals,
core::Signal& out_signal,
std::vector<core::Signal*>& gang_signals) {
@@ -577,9 +560,8 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
out_signal, gang_signals);
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t
BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::SubmitCopyRectCommand(
template <bool useGCR>
hsa_status_t BlitSdma<useGCR>::SubmitCopyRectCommand(
const hsa_pitched_ptr_t* dst, const hsa_dim3_t* dst_offset, const hsa_pitched_ptr_t* src,
const hsa_dim3_t* src_offset, const hsa_dim3_t* range, std::vector<core::Signal*>& dep_signals,
core::Signal& out_signal) {
@@ -653,9 +635,8 @@ BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::SubmitCopyRe
out_signal, gang_signals);
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::SubmitLinearFillCommand(void* ptr, uint32_t value, size_t count) {
template <bool useGCR>
hsa_status_t BlitSdma<useGCR>::SubmitLinearFillCommand(void* ptr, uint32_t value, size_t count) {
const size_t size = count * sizeof(uint32_t);
const uint32_t num_fill_command = (size + kMaxSingleFillSize - 1) / kMaxSingleFillSize;
@@ -666,15 +647,12 @@ hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
return SubmitBlockingCommand(&buff[0], buff.size() * sizeof(SDMA_PKT_CONSTANT_FILL), size);
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
hsa_status_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::EnableProfiling(
bool enable) {
template <bool useGCR> hsa_status_t BlitSdma<useGCR>::EnableProfiling(bool enable) {
return HSA_STATUS_SUCCESS;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
char* BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::AcquireWriteAddress(
uint32_t cmd_size, RingIndexTy& curr_index) {
template <bool useGCR>
char* BlitSdma<useGCR>::AcquireWriteAddress(uint32_t cmd_size, uint64_t& curr_index) {
// Ring is full when all but one byte is written.
if (cmd_size >= kQueueSize) {
return nullptr;
@@ -692,7 +670,7 @@ char* BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::Acquir
}
// Check whether the engine has finished using this region.
const RingIndexTy new_index = curr_index + cmd_size;
const uint64_t new_index = curr_index + cmd_size;
if (CanWriteUpto(new_index) == false) {
// Wait for read index to move and try again.
@@ -713,10 +691,8 @@ char* BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::Acquir
return nullptr;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::UpdateWriteAndDoorbellRegister(RingIndexTy curr_index,
RingIndexTy new_index) {
template <bool useGCR>
void BlitSdma<useGCR>::UpdateWriteAndDoorbellRegister(uint64_t curr_index, uint64_t new_index) {
while (true) {
// Make sure that the address before ::curr_index is already released.
// Otherwise the CP may read invalid packets.
@@ -725,21 +701,19 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
// TODO: remove when sdma wpointer issue is resolved.
// Wait until the SDMA engine finish processing all packets before
// updating the wptr and doorbell.
while (WrapIntoRing(*reinterpret_cast<RingIndexTy*>(queue_resource_.Queue_read_ptr)) !=
while (WrapIntoRing(*reinterpret_cast<uint64_t*>(queue_resource_.Queue_read_ptr)) !=
WrapIntoRing(curr_index)) {
os::YieldThread();
}
}
// Update write pointer and doorbell register.
*reinterpret_cast<RingIndexTy*>(queue_resource_.Queue_write_ptr) =
(HwIndexMonotonic ? new_index : WrapIntoRing(new_index));
*reinterpret_cast<uint64_t*>(queue_resource_.Queue_write_ptr) = new_index;
// Ensure write pointer is visible to GPU before doorbell.
std::atomic_thread_fence(std::memory_order_release);
*reinterpret_cast<RingIndexTy*>(queue_resource_.Queue_DoorBell) =
(HwIndexMonotonic ? new_index : WrapIntoRing(new_index));
*reinterpret_cast<uint64_t*>(queue_resource_.Queue_DoorBell) = new_index;
atomic::Store(&cached_commit_index_, new_index, std::memory_order_release);
break;
@@ -750,9 +724,8 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
}
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::ReleaseWriteAddress(
RingIndexTy curr_index, uint32_t cmd_size) {
template <bool useGCR>
void BlitSdma<useGCR>::ReleaseWriteAddress(uint64_t curr_index, uint32_t cmd_size) {
if (cmd_size > kQueueSize) {
assert(false && "cmd_addr is outside the queue buffer range");
return;
@@ -761,11 +734,9 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::Release
UpdateWriteAndDoorbellRegister(curr_index, curr_index + cmd_size);
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::PadRingToEnd(
RingIndexTy curr_index) {
template <bool useGCR> void BlitSdma<useGCR>::PadRingToEnd(uint64_t curr_index) {
// Reserve region from here to the end of the ring.
RingIndexTy new_index = curr_index + (kQueueSize - WrapIntoRing(curr_index));
uint64_t new_index = curr_index + (kQueueSize - WrapIntoRing(curr_index));
// Check whether the engine has finished using this region.
if (CanWriteUpto(new_index) == false) {
@@ -786,37 +757,22 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::PadRing
}
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
uint32_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::WrapIntoRing(
RingIndexTy index) {
template <bool useGCR> uint32_t BlitSdma<useGCR>::WrapIntoRing(uint64_t index) {
return index & (kQueueSize - 1);
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
bool BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::CanWriteUpto(
RingIndexTy upto_index) {
template <bool useGCR> bool BlitSdma<useGCR>::CanWriteUpto(uint64_t upto_index) {
// Get/calculate the monotonic read index.
RingIndexTy hw_read_index = *reinterpret_cast<RingIndexTy*>(queue_resource_.Queue_read_ptr);
RingIndexTy read_index;
if (HwIndexMonotonic) {
read_index = hw_read_index;
} else {
// Calculate distance from commit index to HW read index.
// Commit index is always < kQueueSize away from HW read index.
RingIndexTy commit_index = atomic::Load(&cached_commit_index_, std::memory_order_relaxed);
RingIndexTy dist_to_read_index = WrapIntoRing(commit_index - hw_read_index);
read_index = commit_index - dist_to_read_index;
}
uint64_t hw_read_index = *reinterpret_cast<uint64_t*>(queue_resource_.Queue_read_ptr);
// Check whether the read pointer has passed the given index.
// At most we can submit (kQueueSize - 1) bytes at a time.
return (upto_index - read_index) < kQueueSize;
return (upto_index - hw_read_index) < kQueueSize;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildFenceCommand(
char* fence_command_addr, uint32_t* fence, uint32_t fence_value) {
template <bool useGCR>
void BlitSdma<useGCR>::BuildFenceCommand(char* fence_command_addr, uint32_t* fence,
uint32_t fence_value) {
assert(fence_command_addr != NULL);
SDMA_PKT_FENCE* packet_addr =
reinterpret_cast<SDMA_PKT_FENCE*>(fence_command_addr);
@@ -836,9 +792,9 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildFe
packet_addr->DATA_UNION.data = fence_value;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildCopyCommand(
char* cmd_addr, uint32_t num_copy_command, void* dst, const void* src, size_t size) {
template <bool useGCR>
void BlitSdma<useGCR>::BuildCopyCommand(char* cmd_addr, uint32_t num_copy_command, void* dst,
const void* src, size_t size) {
size_t cur_size = 0;
const size_t max_copy_size = max_single_linear_copy_size_ ? max_single_linear_copy_size_ :
kMaxSingleCopySize;
@@ -858,9 +814,9 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildCo
packet_addr->HEADER_UNION.sub_op = SDMA_SUBOP_COPY_LINEAR;
if (max_copy_size == (1 << 30) -1)
packet_addr->COUNT_UNION.count_ext.count = copy_size + SizeToCountOffset;
packet_addr->COUNT_UNION.count_ext.count = copy_size - 1; /* count is 1-based */
else
packet_addr->COUNT_UNION.count.count = copy_size + SizeToCountOffset;
packet_addr->COUNT_UNION.count.count = copy_size - 1; /* count is 1-based */
packet_addr->SRC_ADDR_LO_UNION.src_addr_31_0 = ptrlow32(cur_src);
packet_addr->SRC_ADDR_HI_UNION.src_addr_63_32 = ptrhigh32(cur_src);
@@ -881,11 +837,12 @@ Elements are coded by the log2 of the element size in bytes (ie. element 0=1 byt
This routine breaks a large rect into tiles that can be handled by hardware. Pitches and offsets
must be representable in terms of elements in all tiles of the copy.
*/
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildCopyRectCommand(
const std::function<void*(size_t)>& append, const hsa_pitched_ptr_t* dst,
const hsa_dim3_t* dst_offset, const hsa_pitched_ptr_t* src, const hsa_dim3_t* src_offset,
const hsa_dim3_t* range) {
template <bool useGCR>
void BlitSdma<useGCR>::BuildCopyRectCommand(const std::function<void*(size_t)>& append,
const hsa_pitched_ptr_t* dst,
const hsa_dim3_t* dst_offset,
const hsa_pitched_ptr_t* src,
const hsa_dim3_t* src_offset, const hsa_dim3_t* range) {
// Returns the index of the first set bit (ie log2 of the largest power of 2 that evenly divides
// width), the largest element that perfectly covers width.
// width | 16 ensures that we don't return a higher element than is supported and avoids
@@ -1029,9 +986,9 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildCo
}
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildFillCommand(
char* cmd_addr, uint32_t num_fill_command, void* ptr, uint32_t value, size_t count) {
template <bool useGCR>
void BlitSdma<useGCR>::BuildFillCommand(char* cmd_addr, uint32_t num_fill_command, void* ptr,
uint32_t value, size_t count) {
char* cur_ptr = reinterpret_cast<char*>(ptr);
const uint32_t maxDwordCount = kMaxSingleFillSize / sizeof(uint32_t);
SDMA_PKT_CONSTANT_FILL* packet_addr = reinterpret_cast<SDMA_PKT_CONSTANT_FILL*>(cmd_addr);
@@ -1050,7 +1007,8 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildFi
packet_addr->DATA_UNION.src_data_31_0 = value;
packet_addr->COUNT_UNION.count = (fill_count + SizeToCountOffset) * sizeof(uint32_t);
/* count is 1-based */
packet_addr->COUNT_UNION.count = (fill_count - 1) * sizeof(uint32_t);
packet_addr++;
cur_ptr += fill_count * sizeof(uint32_t);
@@ -1059,9 +1017,8 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildFi
assert(count == 0 && "SDMA fill command count error.");
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildPollCommand(
char* cmd_addr, void* addr, uint32_t reference) {
template <bool useGCR>
void BlitSdma<useGCR>::BuildPollCommand(char* cmd_addr, void* addr, uint32_t reference) {
SDMA_PKT_POLL_REGMEM* packet_addr =
reinterpret_cast<SDMA_PKT_POLL_REGMEM*>(cmd_addr);
@@ -1081,9 +1038,8 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildPo
packet_addr->DW5_UNION.retry_count = 0xfff; // Retry forever.
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::BuildAtomicDecrementCommand(char* cmd_addr, void* addr) {
template <bool useGCR>
void BlitSdma<useGCR>::BuildAtomicDecrementCommand(char* cmd_addr, void* addr) {
SDMA_PKT_ATOMIC* packet_addr = reinterpret_cast<SDMA_PKT_ATOMIC*>(cmd_addr);
memset(packet_addr, 0, sizeof(SDMA_PKT_ATOMIC));
@@ -1098,9 +1054,8 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
packet_addr->SRC_DATA_HI_UNION.src_data_63_32 = 0xffffffff;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
useGCR>::BuildGetGlobalTimestampCommand(char* cmd_addr, void* write_address) {
template <bool useGCR>
void BlitSdma<useGCR>::BuildGetGlobalTimestampCommand(char* cmd_addr, void* write_address) {
SDMA_PKT_TIMESTAMP* packet_addr =
reinterpret_cast<SDMA_PKT_TIMESTAMP*>(cmd_addr);
@@ -1113,9 +1068,7 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset,
packet_addr->ADDR_HI_UNION.addr_63_32 = ptrhigh32(write_address);
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildTrapCommand(
char* cmd_addr, uint32_t event_id) {
template <bool useGCR> void BlitSdma<useGCR>::BuildTrapCommand(char* cmd_addr, uint32_t event_id) {
SDMA_PKT_TRAP* packet_addr =
reinterpret_cast<SDMA_PKT_TRAP*>(cmd_addr);
@@ -1125,17 +1078,13 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildTr
packet_addr->INT_CONTEXT_UNION.int_ctx = event_id;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildHdpFlushCommand(
char* cmd_addr) {
template <bool useGCR> void BlitSdma<useGCR>::BuildHdpFlushCommand(char* cmd_addr) {
assert(cmd_addr != NULL);
SDMA_PKT_POLL_REGMEM* addr = reinterpret_cast<SDMA_PKT_POLL_REGMEM*>(cmd_addr);
memcpy(addr, &hdp_flush_cmd, flush_command_size_);
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildGCRCommand(
char* cmd_addr, bool invalidate) {
template <bool useGCR> void BlitSdma<useGCR>::BuildGCRCommand(char* cmd_addr, bool invalidate) {
assert(cmd_addr != NULL);
assert(useGCR && "Unsupported SDMA command - GCR.");
SDMA_PKT_GCR* addr = reinterpret_cast<SDMA_PKT_GCR*>(cmd_addr);
@@ -1154,25 +1103,16 @@ void BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::BuildGC
addr->WORD2_UNION.GCR_CONTROL_GL2_RANGE = 0;
}
template <typename RingIndexTy, bool HwIndexMonotonic, int SizeToCountOffset, bool useGCR>
uint64_t BlitSdma<RingIndexTy, HwIndexMonotonic, SizeToCountOffset, useGCR>::PendingBytes() {
RingIndexTy commit = atomic::Load(&cached_commit_index_, std::memory_order_acquire);
RingIndexTy hw_read_index = *reinterpret_cast<RingIndexTy*>(queue_resource_.Queue_read_ptr);
RingIndexTy read;
if (HwIndexMonotonic) {
read = hw_read_index;
} else {
RingIndexTy dist_to_read_index = WrapIntoRing(commit - hw_read_index);
read = commit - dist_to_read_index;
}
template <bool useGCR> uint64_t BlitSdma<useGCR>::PendingBytes() {
uint64_t commit = atomic::Load(&cached_commit_index_, std::memory_order_acquire);
uint64_t hw_read_index = *reinterpret_cast<uint64_t*>(queue_resource_.Queue_read_ptr);
if (commit == read) return 0;
return bytes_queued_ - bytes_written_[WrapIntoRing(read)];
if (commit == hw_read_index) return 0;
return bytes_queued_ - bytes_written_[WrapIntoRing(hw_read_index)];
}
template class BlitSdma<uint32_t, false, 0, false>;
template class BlitSdma<uint64_t, true, -1, false>;
template class BlitSdma<uint64_t, true, -1, true>;
template class BlitSdma<false>;
template class BlitSdma<true>;
} // namespace amd
} // namespace rocr
@@ -54,10 +54,9 @@
namespace rocr {
namespace AMD {
CpuAgent::CpuAgent(HSAuint32 node, const HsaNodeProperties &node_props)
: core::Agent(
core::Runtime::runtime_singleton_->AgentDriver(core::DriverType::KFD),
node, kAmdCpuDevice),
CpuAgent::CpuAgent(HSAuint32 node, const HsaNodeProperties& node_props,
core::DriverType driver_type)
: core::Agent(core::Runtime::runtime_singleton_->AgentDriver(driver_type), node, kAmdCpuDevice),
properties_(node_props) {
InitRegionList();
@@ -93,8 +93,8 @@ namespace AMD {
const uint64_t CP_DMA_DATA_TRANSFER_CNT_MAX = (1 << 26);
GpuAgent::GpuAgent(HSAuint32 node, const HsaNodeProperties& node_props, bool xnack_mode,
uint32_t index)
: GpuAgentInt(node),
uint32_t index, core::DriverType driver_type)
: GpuAgentInt(node, driver_type),
properties_(node_props),
current_coherency_type_(HSA_AMD_COHERENCY_TYPE_COHERENT),
scratch_used_large_(0),
@@ -106,7 +106,6 @@ GpuAgent::GpuAgent(HSAuint32 node, const HsaNodeProperties& node_props, bool xna
memory_max_frequency_(0),
enum_index_(index),
ape1_base_(0),
ape1_size_(0),
pending_copy_req_ref_(0),
pending_copy_stat_check_ref_(0),
sdma_blit_used_mask_(0),
@@ -712,10 +711,6 @@ core::Blit* GpuAgent::CreateBlitSdma(bool use_xgmi, int rec_eng) {
const size_t copy_size_overrides[2] = {0x3fffff, 0x3fffffff};
switch (isa_->GetMajorVersion()) {
case 7:
case 8:
sdma = new BlitSdmaV2V3();
break;
case 9:
sdma = new BlitSdmaV4();
copy_size_override = (isa_->GetMinorVersion() == 0 && isa_->GetStepping() == 10) ?
@@ -2568,6 +2563,10 @@ hsa_status_t GpuAgent::PcSamplingIterateConfig(hsa_ven_amd_pcs_iterate_configura
if (ret != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
for (uint32_t i = 0; i < size; i++) {
if ((isa_->GetMajorVersion() == 12 && (isa_->GetMinorVersion() == 0)) &&
sampleInfoList[i].method == HSA_PC_SAMPLING_METHOD_KIND_STOCHASTIC_V1) {
continue;
}
hsa_ven_amd_pcs_configuration_t hsaPcSampling;
if (ConvertHsaKmtPcSamplingInfoToHsa(&sampleInfoList[i], &hsaPcSampling) == HSA_STATUS_SUCCESS
&& cb(&hsaPcSampling, cb_data) == HSA_STATUS_INFO_BREAK)
@@ -2614,6 +2613,10 @@ hsa_status_t GpuAgent::PcSamplingCreateFromId(HsaPcSamplingTraceId ioctlId,
if (sampling_method == HSA_VEN_AMD_PCS_METHOD_HOSTTRAP_V1) {
pcs_data = &pcs_hosttrap_data_;
} else if (sampling_method == HSA_VEN_AMD_PCS_METHOD_STOCHASTIC_V1) {
if (isa_->GetMajorVersion() == 12 && (isa_->GetMinorVersion() == 0)) {
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
pcs_data = &pcs_stochastic_data_;
} else {
// Unsupported sampling method
@@ -3093,6 +3096,7 @@ hsa_status_t GpuAgent::PcSamplingFlushDeviceBuffers(
const uint32_t atomic_ex_cmd_sz = 9;
const uint32_t wait_reg_mem_cmd_sz = 7;
const uint32_t acquire_mem_cmd_sz = 8;
const uint32_t dma_data_cmd_sz = 7;
const uint32_t copy_data_cmd_sz = 6;
const uint32_t write_data_cmd_sz = 5;
@@ -3225,6 +3229,20 @@ hsa_status_t GpuAgent::PcSamplingFlushDeviceBuffers(
cmd_data[i++] = PM4_WAIT_REG_MEM_DW6(PM4_WAIT_REG_MEM_POLL_INTERVAL(4) |
PM4_WAIT_REG_MEM_OPTIMIZE_ACE_OFFLOAD_MODE);
// For GFX1200 and GFX1201 only - add an ACQUIRE_MEM packet to flush L2 cache before DMA.
// This ensures that any data written by the trap handler is visible to the DMA engine.
if ((isa_->GetMajorVersion() == 12) && (isa_->GetMinorVersion() == 0)) {
cmd_data[i++] =
PM4_HDR(PM4_HDR_IT_OPCODE_ACQUIRE_MEM, acquire_mem_cmd_sz, isa_->GetMajorVersion());
cmd_data[i++] = 0; // DW1: COHER_CNTL
cmd_data[i++] = 0; // DW2: COHER_SIZE
cmd_data[i++] = 0; // DW3: COHER_SIZE_HI
cmd_data[i++] = 0; // DW4: COHER_BASE_LO
cmd_data[i++] = 0; // DW5: COHER_BASE_HI
cmd_data[i++] = 4; // DW6: POLL_INTERVAL
cmd_data[i++] = PM4_ACQUIRE_MEM_GCR_CNTL_GL2_WB; // DW7: GCR_CNTL (GL2_WB=1, RANGE=ALL)
}
uint8_t* buffer_temp = buffer[which_buffer];
for (copy_bytes = std::min(to_copy, (uint32_t)CP_DMA_DATA_TRANSFER_CNT_MAX); 0 < to_copy;
@@ -68,6 +68,9 @@
#include "core/inc/amd_memory_region.h"
#include "core/inc/runtime.h"
#include "core/util/utils.h"
#ifdef HSAKMT_VIRTIO_ENABLED
#include "core/inc/amd_virtio_driver.h"
#endif
extern r_debug _amdgpu_r_debug;
@@ -75,16 +78,21 @@ namespace rocr {
namespace AMD {
// Anonymous namespace.
namespace {
#if _WIN32
constexpr size_t num_drivers = 0;
#elif __linux__
constexpr size_t num_drivers = 2;
#endif
const std::array<std::function<hsa_status_t(std::unique_ptr<core::Driver>&)>, num_drivers>
const std::array<std::function<hsa_status_t(std::unique_ptr<core::Driver>&)>,
#if _WIN32
0
#elif __linux__
static_cast<size_t>(core::DriverType::NUM_DRIVER_TYPES)
#endif
>
discover_driver_funcs = {
#ifdef __linux__
KfdDriver::DiscoverDriver, XdnaDriver::DiscoverDriver
KfdDriver::DiscoverDriver,
XdnaDriver::DiscoverDriver,
#ifdef HSAKMT_VIRTIO_ENABLED
KfdVirtioDriver::DiscoverDriver,
#endif
#endif
};
@@ -110,14 +118,14 @@ bool InitializeDriver(std::unique_ptr<core::Driver>& driver) {
return true;
}
void DiscoverCpu(HSAuint32 node_id, HsaNodeProperties& node_prop) {
CpuAgent* cpu = new CpuAgent(node_id, node_prop);
void DiscoverCpu(HSAuint32 node_id, HsaNodeProperties& node_prop, core::DriverType driver_type) {
CpuAgent* cpu = new CpuAgent(node_id, node_prop, driver_type);
cpu->Enable();
core::Runtime::runtime_singleton_->RegisterAgent(cpu, true);
}
GpuAgent* DiscoverGpu(HSAuint32 node_id, HsaNodeProperties& node_prop, bool xnack_mode,
bool enabled) {
bool enabled, core::DriverType driver_type) {
GpuAgent* gpu = nullptr;
if (node_prop.NumFComputeCores == 0) {
// Ignore non GPUs.
@@ -125,7 +133,7 @@ GpuAgent* DiscoverGpu(HSAuint32 node_id, HsaNodeProperties& node_prop, bool xnac
}
try {
gpu = new GpuAgent(node_id, node_prop, xnack_mode,
core::Runtime::runtime_singleton_->gpu_agents().size());
core::Runtime::runtime_singleton_->gpu_agents().size(), driver_type);
const HsaVersionInfo& kfd_version = core::Runtime::runtime_singleton_->KfdVersion().version;
@@ -152,7 +160,7 @@ GpuAgent* DiscoverGpu(HSAuint32 node_id, HsaNodeProperties& node_prop, bool xnac
node_prop.Capability.ui32.SRAM_EDCSupport = 1;
delete gpu;
gpu = new GpuAgent(node_id, node_prop, xnack_mode,
core::Runtime::runtime_singleton_->gpu_agents().size());
core::Runtime::runtime_singleton_->gpu_agents().size(), driver_type);
}
}
} catch (const hsa_exception& e) {
@@ -257,24 +265,29 @@ void SurfaceGpuList(std::vector<int32_t>& gpu_list, bool xnack_mode, bool enable
const int32_t invalidIdx = -1;
int32_t list_sz = gpu_list.size();
HsaNodeProperties node_prop = {0};
const auto& gpu_driver = core::Runtime::runtime_singleton_->AgentDriver(core::DriverType::KFD);
for (int32_t idx = 0; idx < list_sz; idx++) {
if (gpu_list[idx] == invalidIdx) {
break;
for (const auto& gpu_driver : core::Runtime::runtime_singleton_->AgentDrivers()) {
if (!core::Runtime::IsGPUDriver(gpu_driver->kernel_driver_type_)) {
continue;
}
// Obtain properties of the node
hsa_status_t ret = gpu_driver.GetNodeProperties(node_prop, gpu_list[idx]);
assert(ret == HSA_STATUS_SUCCESS && "Error in getting Node Properties");
for (int32_t idx = 0; idx < list_sz; idx++) {
if (gpu_list[idx] == invalidIdx) {
break;
}
// disable interrupt signal for DTIF platform
if (core::Runtime::runtime_singleton_->flag().enable_dtif())
core::g_use_interrupt_wait = false;
// Obtain properties of the node
hsa_status_t ret = gpu_driver->GetNodeProperties(node_prop, gpu_list[idx]);
assert(ret == HSA_STATUS_SUCCESS && "Error in getting Node Properties");
// Instantiate a Gpu device. The IO links
// of this node have already been registered
assert((node_prop.NumFComputeCores != 0) && "Improper node used for GPU device discovery.");
DiscoverGpu(gpu_list[idx], node_prop, xnack_mode, enabled);
// disable interrupt signal for DTIF platform
if (core::Runtime::runtime_singleton_->flag().enable_dtif())
core::g_use_interrupt_wait = false;
// Instantiate a Gpu device. The IO links
// of this node have already been registered
assert((node_prop.NumFComputeCores != 0) && "Improper node used for GPU device discovery.");
DiscoverGpu(gpu_list[idx], node_prop, xnack_mode, enabled, gpu_driver->kernel_driver_type_);
}
}
}
@@ -335,7 +348,7 @@ bool BuildTopology() {
/// @todo: Add support for AIEs.
// Query if env ROCR_VISIBLE_DEVICES is defined. If defined
// determine number and order of GPU devices to be surfaced.
if (filter && driver->kernel_driver_type_ == core::DriverType::KFD) {
if (filter && (core::Runtime::IsGPUDriver(driver->kernel_driver_type_))) {
rvdFilter.BuildRvdTokenList();
rvdFilter.BuildDeviceUuidList(node_props_vec);
visibleCnt = rvdFilter.BuildUsrDeviceList();
@@ -350,7 +363,7 @@ bool BuildTopology() {
for (auto& node_props : node_props_vec) {
if (node_props.NumCPUCores) {
// Node has CPU cores so instantiate a CPU agent.
DiscoverCpu(node_id, node_props);
DiscoverCpu(node_id, node_props, driver->kernel_driver_type_);
}
if (node_props.NumNeuralCores) {
@@ -54,9 +54,7 @@
#include <sys/mman.h>
#include <sys/socket.h>
#include <sys/un.h>
#include <iostream>
#include <thread>
#include <chrono>
#include "core/inc/runtime.h"
#include "core/inc/hsa_table_interface.h"
@@ -3078,8 +3076,8 @@ Agent* Runtime::GetSVMPrefetchAgent(void* ptr, size_t size) {
return agents_by_node_[prefetch_node][0];
}
hsa_status_t Runtime::DmaBufExport(const void* ptr, size_t size, int* dmabuf,
uint64_t* offset, uint64_t flags) {
hsa_status_t Runtime::DmaBufExport(const void* ptr, size_t size, int* dmabuf, uint64_t* offset,
uint64_t flags) {
#ifdef __linux__
ScopedAcquire<KernelSharedMutex::Shared> lock(memory_lock_.shared());
// Lookup containing allocation.
@@ -3090,18 +3088,23 @@ hsa_status_t Runtime::DmaBufExport(const void* ptr, size_t size, int* dmabuf,
(ptr < reinterpret_cast<const uint8_t*>(mem->first) + mem->second.size)) {
// Check size is in bounds.
if (uintptr_t(ptr) - uintptr_t(mem->first) + size <= mem->second.size) {
// Check allocation is on GPU
if (mem->second.region->owner()->device_type() != Agent::kAmdGpuDevice)
return HSA_STATUS_ERROR_INVALID_AGENT;
switch (mem->second.region->owner()->device_type()) {
case Agent::kAmdGpuDevice: {
auto* owner = static_cast<AMD::GpuAgent*>(mem->second.region->owner());
rocr::AMD::GpuAgent* owner =
static_cast<AMD::GpuAgent*>(mem->second.region->owner());
if (flags & HSA_AMD_DMABUF_MAPPING_TYPE_PCIE &&
!owner->is_xgmi_cpu_gpu() &&
!owner->LargeBarEnabled()) {
return (hsa_status_t)HSA_STATUS_ERROR_NOT_SUPPORTED;
if (flags & HSA_AMD_DMABUF_MAPPING_TYPE_PCIE && !owner->is_xgmi_cpu_gpu() &&
!owner->LargeBarEnabled()) {
return static_cast<hsa_status_t>(HSA_STATUS_ERROR_NOT_SUPPORTED);
}
} break;
case Agent::kAmdCpuDevice:
return HSA_STATUS_ERROR_INVALID_AGENT;
case Agent::kAmdAieDevice:
break;
case Agent::kUnknownDevice:
return HSA_STATUS_ERROR_INVALID_AGENT;
}
int fd;
uint64_t off;
hsa_status_t err = mem->second.region->owner()->driver().ExportDMABuf(
@@ -3194,10 +3197,11 @@ hsa_status_t Runtime::VMemoryAddressFree(void* va, size_t size) {
if (it->second.use_count > 0) return HSA_STATUS_ERROR_RESOURCE_FREE;
if (it->second.registered)
if (it->second.registered) {
if (HSAKMT_CALL(hsaKmtFreeMemory(it->second.os_addr, size)) != HSAKMT_STATUS_SUCCESS) return HSA_STATUS_ERROR;
else
} else {
if (munmap(it->second.os_addr, size)) return HSA_STATUS_ERROR;
}
reserved_address_map_.erase(it);
return HSA_STATUS_SUCCESS;
@@ -3213,7 +3217,7 @@ hsa_status_t Runtime::VMemoryHandleCreate(const MemoryRegion* region, size_t siz
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
ScopedAcquire<KernelSharedMutex> lock(&memory_lock_);
void *user_mode_driver_handle;
ThunkHandle user_mode_driver_handle;
hsa_status_t status =
region->Allocate(size, alloc_flags, &user_mode_driver_handle, 0);
if (status == HSA_STATUS_SUCCESS) {
@@ -3230,7 +3234,7 @@ hsa_status_t Runtime::VMemoryHandleCreate(const MemoryRegion* region, size_t siz
hsa_status_t Runtime::VMemoryHandleRelease(hsa_amd_vmem_alloc_handle_t memoryOnlyHandle) {
ScopedAcquire<KernelSharedMutex> lock(&memory_lock_);
auto memoryHandleIt = memory_handle_map_.find(reinterpret_cast<void*>(memoryOnlyHandle.handle));
auto memoryHandleIt = memory_handle_map_.find(MemoryHandle::Convert(memoryOnlyHandle));
if (memoryHandleIt == memory_handle_map_.end()) {
debug_warning(false && "Can't find memory handle");
@@ -3285,7 +3289,7 @@ hsa_status_t Runtime::VMemoryHandleMap(void* va, size_t size, size_t in_offset,
if (reinterpret_cast<uint8_t*>(va) + size > lowerMappedHandleIt->first) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
auto memoryHandleIt = memory_handle_map_.find(reinterpret_cast<void*>(memoryOnlyHandle.handle));
auto memoryHandleIt = memory_handle_map_.find(MemoryHandle::Convert(memoryOnlyHandle));
if (memoryHandleIt == memory_handle_map_.end()) {
debug_warning(false && "Can't find memory handle");
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
@@ -3648,7 +3652,7 @@ hsa_status_t Runtime::VMemoryExportShareableHandle(int* dmabuf_fd,
hsa_amd_vmem_alloc_handle_t handle,
uint64_t flags) {
*dmabuf_fd = -1;
auto memoryHandle = memory_handle_map_.find((void*)handle.handle);
auto memoryHandle = memory_handle_map_.find(MemoryHandle::Convert(handle));
if (memoryHandle == memory_handle_map_.end()) {
debug_warning(false && "Can't find memory handle");
return HSA_STATUS_ERROR_INVALID_ALLOCATION;
@@ -3744,7 +3748,7 @@ hsa_status_t Runtime::VMemoryRetainAllocHandle(hsa_amd_vmem_alloc_handle_t* mapp
hsa_status_t Runtime::VMemoryGetAllocPropertiesFromHandle(hsa_amd_vmem_alloc_handle_t allocHandle,
const core::MemoryRegion** mem_region,
hsa_amd_memory_type_t* type) {
auto memoryHandleIt = memory_handle_map_.find(reinterpret_cast<void*>(allocHandle.handle));
auto memoryHandleIt = memory_handle_map_.find(MemoryHandle::Convert(allocHandle));
if (memoryHandleIt == memory_handle_map_.end()) return HSA_STATUS_ERROR_INVALID_ALLOCATION;
*mem_region = memoryHandleIt->second.region;
@@ -56,6 +56,7 @@
.set SQ_WAVE_EXCP_FLAG_PRIV_HT_SHIFT , 7
.set SQ_WAVE_EXCP_FLAG_PRIV_WAVE_START_SHIFT , 8
.set SQ_WAVE_EXCP_FLAG_PRIV_WAVE_END_SHIFT , 9
.set SQ_WAVE_EXCP_FLAG_PRIV_PERF_SNAPSHOT , 10
.set SQ_WAVE_EXCP_FLAG_PRIV_TRAP_AFTER_INST_SHIFT , 11
.set SQ_WAVE_EXCP_FLAG_PRIV_XNACK_ERROR_SHIFT , 12
@@ -74,6 +75,7 @@
.set SQ_WAVE_STATE_PRIV_HALT_BFE , (SQ_WAVE_STATE_PRIV_HALT_SHIFT | (1 << 16))
.set SQ_WAVE_STATE_PRIV_HALT_SHIFT , 14
.set SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT , 2
.set TRAP_ID_ABORT , 2
.set TRAP_ID_DEBUGTRAP , 3
.set TTMP6_SAVED_STATUS_HALT_MASK , (1 << TTMP6_SAVED_STATUS_HALT_SHIFT)
@@ -87,140 +89,273 @@
.set TTMP11_DEBUG_ENABLED_SHIFT , 23
.set TTMP_PC_HI_SHIFT , 7
// ABI between first and second level trap handler:
// { ttmp1, ttmp0 } = TrapID[3:0], zeros, PC[47:0]
// ttmp11 = 0[7:0], DebugEnabled[0], 0[15:0], NoScratch[0], 0[5:0]
// ttmp12 = SQ_WAVE_STATE_PRIV
// ttmp14 = TMA[31:0]
// ttmp15 = TMA[63:32]
.set TTMP13_HT_FLAG_BIT , 22 // TTMP13 bit for hosttrap
.set TTMP13_STOCH_FLAG_BIT , 21 // TTMP13 bit for stochastic
.set TTMP13_BUF_FULL_BIT , 31 // TTMP13 bit buf full mark
.set TTMP8_DISPATCH_ID_MASK , 0X1FFFFFF
// Per-sample data layout within the device buffer. Each sample is 64 bytes.
// These are offsets from the start of a specific sample slot in the device buffer.
trap_entry:
// Clear ttmp3 as it will contain the exception code.
s_mov_b32 ttmp3, 0
.set SAMPLE_OFF_BYTES_PER_SAMPLE , 0x40 // bytes per sample slot
.set SAMPLE_OFF_PC_HOST , 0x00 // original PC (host only)
.set SAMPLE_OFF_EXEC_LOHI , 0x08 // saved EXEC low/high
.set SAMPLE_OFF_WGID_XY , 0x10 // WG id X / Y
.set SAMPLE_OFF_WGID_Z_WAVE , 0x18 // WG id Z
.set SAMPLE_OFF_TIMESTAMP , 0x30 // 64 bit realtime counter
.set SAMPLE_OFF_HW_ID , 0x20 // HW_ID (values combined from the HW_ID1 + HW_ID2)
.set SAMPLE_OFF_SNAPSHOT_DATA , 0x24
.set SAMPLE_OFF_CORRELATION , 0x38 // doorbell + dispatch id
.set SAMPLE_OFF_BUF_WRITTEN_VAL , 0x10 // Offset to buf_written_val0/1 in pcs_sampling_data_t
.set SAMPLE_OFF_BUF_SIZE , 0x8 // Offset to buf_size in pcs_sampling_data_t
.set SAMPLE_OFF_DONE_SIG0 , 0x18 // Offset for done_sig0 (hsa_signal_t handle for buffer 0)
.set SAMPLE_OFF_DONE_SIG1 , 0x28 // Offset for done_sig1 (hsa_signal_t handle for buffer 1)
.set SAMPLE_OFF_SIGNAL_VALUE , 0x8 // Offset within signal structure to value field
.set SAMPLE_OFF_EVENT_MAILBOX0 , 0x10 // Offset for event mailbox pointer for buffer 0
.set SAMPLE_OFF_EVENT_MAILBOX1 , 0x20 // Offset for event mailbox pointer for buffer 1
.set WAVE_ID_MASK , 0x1f // Mask to extract Wave ID from TTMP register.
.set BUF_INDEX_MASK , 0x7fffffff // strip bit31 from add_x2
.set SAMPLE_OFF_BUF_WRITTEN_VAL , 0x10 // Offset to buf_written_val0/1 in pcs_sampling_data_t
.set SAMPLE_INDEX_WIDTH , 31 // The sample index is 63 bits; the high part is 31 bits.
.set HW_REG_SHADER_HW_ID1 , 0xf817
.set HW_REG_SHADER_HW_ID2 , 0xf818
.set HW_REG_SQ_PERF_SNAPSHOT_PC_LO , 0xf80b
.set HW_REG_SQ_PERF_SNAPSHOT_PC_HI , 0xf80c
.set HW_REG_SQ_PERF_SNAPSHOT_DATA1 , 0xf80f
.set HW_REG_SQ_PERF_SNAPSHOT_DATA2 , 0xf810
.set HW_REG_SQ_PERF_SNAPSHOT_DATA , 0xf81b
// Macro to store the Correlation ID (Dispatch ID and Doorbell ID) into the current sample slot
//
// Assumes the following registers are set before it is called:
// v[0:1]:Must contain the 64-bit base address of the target sample slot
// ttmp8 :Must contain the dispatch ID in bits [24:0]
// exec :Must be set to 0x1 to ensure operations apply only to lane 0
//
// Clobbers the following registers:
// v[2:3]:Used for [dispatch_id, doorbell_id]
// ttmp6 :Used as scratch register
.macro STORE_CORRELATION_ID
s_sendmsg_rtn_b32 ttmp6, sendmsg(MSG_RTN_GET_DOORBELL) // Gets current queue's doorbell ID into ttmp6.
s_wait_kmcnt 0
s_and_b32 ttmp6, ttmp6, DOORBELL_ID_MASK // Mask to get actual doorbell ID.
v_writelane_b32 v3, ttmp6, 0 // Store doorbell ID into high part of v[2:3] (via v3).
s_and_b32 ttmp6, ttmp8, TTMP8_DISPATCH_ID_MASK // Get dispatch ID from ttmp8 into ttmp6
v_writelane_b32 v2, ttmp6, 0 // Store dispatch ID into low part of v[2:3] (via v2)
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_CORRELATION, scope:SCOPE_SYS // Store {dispatch_id, doorbell_id} into sample slot.
// v[0:1] = sample slot base address.
// v[2] = dispatch_id, v[3] = doorbell_id.
.endm
// Macro to store the HW_ID registers into the current sample slot
//
// Assumes the following registers are set before it is called:
// v[0:1]: Must contain the 64-bit base address of the target sample slot.
// exec : Must be set to 0x1 to ensure operations apply only to lane 0.
//
// Clobbers the following registers:
// v[2:3]: Used to stage the data for the global store.
// ttmp6 : Used as scratch registers.
.macro STORE_HW_ID
// Current ROCr API determines single dword for HW_ID, while this information is scattered accross two
// dword registers HW_ID1 and HW_ID2 on GFX10+ architectures.
// Thus, we combine values from HW_ID1 and HW_ID2 into a single dword HW_ID with the following layout:
// WAVE_ID[4:0]
// QUEUE_ID[8:5]
// RESERVED [9]
// WGP_ID[13:10]
// SIMD_ID[15:14]
// SA_ID[16]
// ME_ID[17]
// SE_ID[19:18]
// PIPE_ID[21:20]
// RESERVED [22]
// WG_ID[27:23]
// VM_ID[31:28]
// Note: We don't show DP_RATE and STATE_ID that are useless for compute kernels
// Also, we reduced SE_ID to 2 bits as there's only a maximum of 4 SEs on existing gfx12.0 parts
// Finally, ME_ID is reduced to 1 bit as wavefronts are dispatched from either ME0 or ME1 in gfx12.
// Bits 9 and 22 are reserved for a future use.
s_getreg_b32 ttmp6, HW_REG_SHADER_HW_ID1 // Put HW_ID1 in ttmp6
v_and_b32 v2, ttmp6, 0x1feffcff // Mask DP_RATE, SE_ID[2] and SIMD_ID
v_and_b32 v3, ttmp6, 0x300 // Put SIMD_ID into ttmp6[8:9]
v_lshl_or_b32 v2, v3, 6, v2 // Put SIMD_ID into v2[15:14]
s_getreg_b32 ttmp6, HW_REG_SHADER_HW_ID2 // Put HW_ID2 in ttmp6
v_and_b32 v3, ttmp6, 0xf000000 // v3 = VM_ID in bits 27:24
v_lshl_or_b32 v2, v3, 4, v2 // Put VM_ID into v2[31:28]
v_and_b32 v3, ttmp6, 0x1f0000 // v3 = WG_ID in bits 20:16
v_lshl_or_b32 v2, v3, 7, v2 // Put WG_ID in v2[27:23]
v_and_b32 v3, ttmp6, 0x100 // v3 = ME_ID[0] in bit 8
v_lshl_or_b32 v2, v3, 9, v2 // Put ME_ID in v2[17]
v_and_b32 v3, ttmp6, 0x30 // v3 = PIPE_ID in bits 5:4
v_lshl_or_b32 v2, v3, 16, v2 // Put PIPE_ID in v2[21:20]
v_and_b32 v3, ttmp6, 0xf // v3 = QUEUE_ID in bits 3:0
v_lshl_or_b32 v2, v3, 5, v2 // Put QUEUE_ID in v2[8:5]
global_store_b32 v[0:1], v2, off, offset:SAMPLE_OFF_HW_ID, scope:SCOPE_SYS // store HW_ID
.endm
// ABI (Application Binary Interface) between first and second-level trap handler:
// ttmp0: PC_LO[31:0] (Program Counter Low)
// ttmp1: PC_HI[15:0] (Program Counter High, bits 0-15), TrapID[3:0] (in bits 28-31 of original PC_HI)
// ttmp11: 0[7:0], DebugEnabled[0], 0[15:0], NoScratch[0], 0[5:0]
// ttmp12: SQ_WAVE_STATE_PRIV (Private wave state register value).
// ttmp14: TMA[31:0] - TMA_LO (Trap Memory Argument Low - base address for trap handler data, low 32 bits).
// ttmp15: TTMA[63:32] - TMA_HI (Trap Memory Argument High - base address for trap handler data, high 32 bits).
// For PC Sampling, this points to pcs_hosttrap_data_ or pcs_stochastic_data_
trap_entry:
s_mov_b32 ttmp3, 0
.check_hosttrap:
// ttmp[14:15] points to TMA.
// Available: ttmp[2:3], ttmp[4:5], ttmp6, ttmp[10:11]
s_getreg_b32 ttmp2, hwreg(HW_REG_EXCP_FLAG_PRIV) // On gfx12, EXCP_FLAG_PRIV.b7
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_HT_SHIFT
s_cbranch_scc0 .check_stochastic
// It's a Host Trap event.
s_load_b64 ttmp[14:15], ttmp[14:15], 0x0, scope:SCOPE_CU // ttmp[14:15]=*host_trap_buffers
s_bitset1_b32 ttmp13, TTMP13_HT_FLAG_BIT // set bit 22 in TTMP13
// Clear the Host Trap flag in the hardware register to acknowledge the event
s_setreg_imm32_b32 hwreg(HW_REG_EXCP_FLAG_PRIV, SQ_WAVE_EXCP_FLAG_PRIV_HT_SHIFT,1), 0
s_wait_kmcnt 0 // Ensure previous load is complete.
s_branch .profile_trap_handlers
.check_stochastic:
s_getreg_b32 ttmp2, hwreg(HW_REG_EXCP_FLAG_PRIV) // EXCP_FLAG_PRIV.b10=stochastic_sample_trap
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_PERF_SNAPSHOT // Test Performance Snapshot bit.
s_cbranch_scc0 .check_exceptions // If not Stochastic, check for other exceptions.
s_load_b64 ttmp[14:15], ttmp[14:15], 0x8, scope:SCOPE_CU // ttmp[14:15]=*stoch_trap_buf
s_wait_kmcnt 0
s_bitset1_b32 ttmp13, TTMP13_STOCH_FLAG_BIT // set bit 21 in TTMP13
s_setreg_imm32_b32 hwreg(HW_REG_EXCP_FLAG_PRIV, SQ_WAVE_EXCP_FLAG_PRIV_PERF_SNAPSHOT,1), 0 // Clear the perf_snapshot flag
s_branch .profile_trap_handlers
// Check if this is a trap (s_trap instruction) or a hardware exception.
// Extract TrapID from ttmp1 (which contains PC_HI).
// Branch if not a trap (an exception instead).
s_bfe_u32 ttmp2, ttmp1, SQ_WAVE_PC_HI_TRAP_ID_BFE
s_cbranch_scc0 .check_exceptions
s_bfe_u32 ttmp2, ttmp1, SQ_WAVE_PC_HI_TRAP_ID_BFE // ttmp2 = TrapID
s_cbranch_scc0 .check_exceptions // If TrapID is 0, it's an exception, so branch.
// If caused by s_trap then advance PC, then figure out the trap ID:
// - if trapID is DEBUGTRAP and debugger is attach, report WAVE_TRAP,
// - if trapID is ABORTTRAP, report WAVE_ABORT,
// - report WAVE_TRAP for any other trap ID.
s_add_u32 ttmp0, ttmp0, 0x4
s_addc_u32 ttmp1, ttmp1, 0x0
s_add_u32 ttmp0, ttmp0, 0x4 // PC_LO += 4
s_addc_u32 ttmp1, ttmp1, 0x0 // PC_HI += carry.
// If llvm.debugtrap and debugger is not attached.
s_cmp_eq_u32 ttmp2, TRAP_ID_DEBUGTRAP
s_cbranch_scc0 .not_debug_trap
s_cmp_eq_u32 ttmp2, TRAP_ID_DEBUGTRAP
s_cbranch_scc0 .not_debug_trap
s_bitcmp1_b32 ttmp11, TTMP11_DEBUG_ENABLED_SHIFT
s_cbranch_scc0 .check_exceptions
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
s_bitcmp1_b32 ttmp11, TTMP11_DEBUG_ENABLED_SHIFT
s_cbranch_scc0 .check_exceptions
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
.not_debug_trap:
s_cmp_eq_u32 ttmp2, TRAP_ID_ABORT
s_cbranch_scc0 .not_abort_trap
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_ABORT_M0
s_branch .check_exceptions
s_cmp_eq_u32 ttmp2, TRAP_ID_ABORT
s_cbranch_scc0 .not_abort_trap
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_ABORT_M0
s_branch .check_exceptions
.not_abort_trap:
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
s_bitcmp1_b32 ttmp8, TTMP8_DEBUG_FLAG_SHIFT
s_cbranch_scc0 .check_exceptions
// We need to explititly look for all exceptions we want to report to the
// host:
// - EXCP_FLAG_PRIV.XNACK_ERROR (&& EXCP_FLAG_PRIV.MEMVIOL)
// -> WAVE_MEMORY_VIOLATION
// - EXCP_FLAG_PRIV.MEMVIOL (and !EXCP_FLAG_PRIV.XNACK_ERROR)
// -> WAVE_APERTURE_VIOLATION
// - EXCP_FLAG_PRIV.ILLEGAL_INST -> WAVE_ILLEGAL_INSTRUCTION
// - EXCP_FLAG_PRIV.WAVE_START -> WAVE_TRAP
// - EXCP_FLAG_PRIV.WAVE_END && TRAP_CTRL.WAVE_END -> WAVE_TRAP
// - TRAP_CTRL.TRAP_AFTER_INST -> WAVE_TRAP
// - EXCP_FLAG_PRIV.ADDR_WATCH && TRAP_CTL.WATCH -> WAVE_TRAP
// - (EXCP_FLAG_USER[ALU] & TRAP_CTRL[ALU]) != 0 -> WAVE_MATH_ERROR
.check_exceptions:
s_getreg_b32 ttmp2, hwreg(HW_REG_EXCP_FLAG_PRIV)
s_getreg_b32 ttmp13, hwreg(HW_REG_TRAP_CTRL)
s_getreg_b32 ttmp2, hwreg(HW_REG_EXCP_FLAG_PRIV)
s_getreg_b32 ttmp13, hwreg(HW_REG_TRAP_CTRL)
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_XNACK_ERROR_SHIFT
s_cbranch_scc0 .not_memory_violation
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_MEMORY_VIOLATION_M0
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_XNACK_ERROR_SHIFT
s_cbranch_scc0 .not_memory_violation
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_MEMORY_VIOLATION_M0
// Aperture violation requires XNACK_ERROR == 0.
s_branch .not_aperture_violation
s_branch .not_aperture_violation
.not_memory_violation:
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_MEMVIOL_SHIFT
s_cbranch_scc0 .not_aperture_violation
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_APERTURE_VIOLATION_M0
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_MEMVIOL_SHIFT
s_cbranch_scc0 .not_aperture_violation
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_APERTURE_VIOLATION_M0
.not_aperture_violation:
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_ILLEGAL_INST_SHIFT
s_cbranch_scc0 .not_illegal_instruction
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_ILLEGAL_INSTRUCTION_M0
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_ILLEGAL_INST_SHIFT
s_cbranch_scc0 .not_illegal_instruction
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_ILLEGAL_INSTRUCTION_M0
.not_illegal_instruction:
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_WAVE_START_SHIFT
s_cbranch_scc0 .not_wave_end
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_WAVE_START_SHIFT
s_cbranch_scc0 .not_wave_end
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
.not_wave_start:
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_WAVE_END_SHIFT
s_cbranch_scc0 .not_wave_end
s_bitcmp1_b32 ttmp13, SQ_WAVE_TRAP_CTRL_WAVE_END_SHIFT
s_cbranch_scc0 .not_wave_end
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
s_bitcmp1_b32 ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_WAVE_END_SHIFT
s_cbranch_scc0 .not_wave_end
s_bitcmp1_b32 ttmp13, SQ_WAVE_TRAP_CTRL_WAVE_END_SHIFT
s_cbranch_scc0 .not_wave_end
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
.not_wave_end:
s_bitcmp1_b32 ttmp13, SQ_WAVE_TRAP_CTRL_TRAP_AFTER_INST
s_cbranch_scc0 .not_trap_after_inst
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
s_bitcmp1_b32 ttmp13, SQ_WAVE_TRAP_CTRL_TRAP_AFTER_INST
s_cbranch_scc0 .not_trap_after_inst
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
.not_trap_after_inst:
s_and_b32 ttmp2, ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_ADDR_WATCH_MASK
s_cbranch_scc0 .not_addr_watch
s_bitcmp1_b32 ttmp13, SQ_WAVE_TRAP_CTRL_ADDR_WATCH_SHIFT
s_cbranch_scc0 .not_addr_watch
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
s_and_b32 ttmp2, ttmp2, SQ_WAVE_EXCP_FLAG_PRIV_ADDR_WATCH_MASK
s_cbranch_scc0 .not_addr_watch
s_bitcmp1_b32 ttmp13, SQ_WAVE_TRAP_CTRL_ADDR_WATCH_SHIFT
s_cbranch_scc0 .not_addr_watch
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_TRAP_M0
.not_addr_watch:
s_getreg_b32 ttmp2, hwreg(HW_REG_EXCP_FLAG_USER, SQ_WAVE_EXCP_FLAG_USER_MATH_EXCP_SHIFT, SQ_WAVE_EXCP_FLAG_USER_MATH_EXCP_SIZE)
s_and_b32 ttmp13, ttmp13, SQ_WAVE_TRAP_CTRL_MATH_EXCP_MASK
s_and_b32 ttmp2, ttmp2, ttmp13
s_cbranch_scc0 .not_math_exception
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_MATH_ERROR_M0
s_getreg_b32 ttmp2, hwreg(HW_REG_EXCP_FLAG_USER, SQ_WAVE_EXCP_FLAG_USER_MATH_EXCP_SHIFT, SQ_WAVE_EXCP_FLAG_USER_MATH_EXCP_SIZE)
s_and_b32 ttmp13, ttmp13, SQ_WAVE_TRAP_CTRL_MATH_EXCP_MASK
s_and_b32 ttmp2, ttmp2, ttmp13
s_cbranch_scc0 .not_math_exception
s_or_b32 ttmp3, ttmp3, EC_QUEUE_WAVE_MATH_ERROR_M0
.not_math_exception:
s_cmp_eq_u32 ttmp3, 0
s_cmp_eq_u32 ttmp3, 0
// This was not a s_trap we are interested in or an exception, return to
// the user code.
s_cbranch_scc1 .exit_trap
s_cbranch_scc1 .exit_trap
.send_interrupt:
// Fetch doorbell id for our queue.
s_sendmsg_rtn_b32 ttmp2, sendmsg(MSG_RTN_GET_DOORBELL)
s_wait_kmcnt 0
s_and_b32 ttmp2, ttmp2, DOORBELL_ID_MASK
s_or_b32 ttmp3, ttmp2, ttmp3
s_sendmsg_rtn_b32 ttmp2, sendmsg(MSG_RTN_GET_DOORBELL)
s_wait_kmcnt 0
s_and_b32 ttmp2, ttmp2, DOORBELL_ID_MASK
s_or_b32 ttmp3, ttmp2, ttmp3
// Save trap id and halt status in ttmp6.
s_andn2_b32 ttmp6, ttmp6, (TTMP6_SAVED_TRAP_ID_MASK | TTMP6_SAVED_STATUS_HALT_MASK)
s_bfe_u32 ttmp2, ttmp1, SQ_WAVE_PC_HI_TRAP_ID_BFE
s_min_u32 ttmp2, ttmp2, 0xF
s_lshl_b32 ttmp2, ttmp2, TTMP6_SAVED_TRAP_ID_SHIFT
s_or_b32 ttmp6, ttmp6, ttmp2
s_bfe_u32 ttmp2, ttmp12, SQ_WAVE_STATE_PRIV_HALT_BFE
s_lshl_b32 ttmp2, ttmp2, TTMP6_SAVED_STATUS_HALT_SHIFT
s_or_b32 ttmp6, ttmp6, ttmp2
s_andn2_b32 ttmp6, ttmp6, (TTMP6_SAVED_TRAP_ID_MASK | TTMP6_SAVED_STATUS_HALT_MASK)
s_bfe_u32 ttmp2, ttmp1, SQ_WAVE_PC_HI_TRAP_ID_BFE
s_min_u32 ttmp2, ttmp2, 0xF
s_lshl_b32 ttmp2, ttmp2, TTMP6_SAVED_TRAP_ID_SHIFT
s_or_b32 ttmp6, ttmp6, ttmp2
s_bfe_u32 ttmp2, ttmp12, SQ_WAVE_STATE_PRIV_HALT_BFE
s_lshl_b32 ttmp2, ttmp2, TTMP6_SAVED_STATUS_HALT_SHIFT
s_or_b32 ttmp6, ttmp6, ttmp2
// m0 = interrupt data = (exception_code << DOORBELL_ID_SIZE) | doorbell_id
s_mov_b32 ttmp2, m0
s_mov_b32 m0, ttmp3
s_nop 0x0 // Manually inserted wait states
s_sendmsg sendmsg(MSG_INTERRUPT)
s_mov_b32 ttmp2, m0
s_mov_b32 m0, ttmp3
s_sendmsg sendmsg(MSG_INTERRUPT)
// Wait for the message to go out.
s_wait_kmcnt 0
s_mov_b32 m0, ttmp2
s_wait_kmcnt 0
s_mov_b32 m0, ttmp2
// Parking the wave requires saving the original pc in the preserved ttmps.
// Register layout before parking the wave:
@@ -234,44 +369,488 @@ trap_entry:
// ttmp11: 1st_level_ttmp11[31:23] pc_hi[15:0] 1st_level_ttmp11[6:0]
//
// Save the PC
s_mov_b32 ttmp10, ttmp0
s_and_b32 ttmp1, ttmp1, SQ_WAVE_PC_HI_ADDRESS_MASK
s_lshl_b32 ttmp1, ttmp1, TTMP_PC_HI_SHIFT
s_andn2_b32 ttmp11, ttmp11, (SQ_WAVE_PC_HI_ADDRESS_MASK << TTMP_PC_HI_SHIFT)
s_or_b32 ttmp11, ttmp11, ttmp1
s_mov_b32 ttmp10, ttmp0
s_and_b32 ttmp1, ttmp1, SQ_WAVE_PC_HI_ADDRESS_MASK
s_lshl_b32 ttmp1, ttmp1, TTMP_PC_HI_SHIFT
s_andn2_b32 ttmp11, ttmp11, (SQ_WAVE_PC_HI_ADDRESS_MASK << TTMP_PC_HI_SHIFT)
s_or_b32 ttmp11, ttmp11, ttmp1
// Park the wave
s_getpc_b64 [ttmp0, ttmp1]
s_add_u32 ttmp0, ttmp0, .parked - .
s_addc_u32 ttmp1, ttmp1, 0x0
s_getpc_b64 [ttmp0, ttmp1]
s_add_u32 ttmp0, ttmp0, .parked - .
s_addc_u32 ttmp1, ttmp1, 0x0
.halt_wave:
// Halt the wavefront upon restoring STATUS below.
s_bitset1_b32 ttmp6, TTMP6_WAVE_STOPPED_SHIFT
s_bitset1_b32 ttmp12, SQ_WAVE_STATE_PRIV_HALT_SHIFT
s_bitset1_b32 ttmp6, TTMP6_WAVE_STOPPED_SHIFT
s_bitset1_b32 ttmp12, SQ_WAVE_STATE_PRIV_HALT_SHIFT
// Initialize TTMP registers
s_bitcmp1_b32 ttmp8, TTMP8_DEBUG_FLAG_SHIFT
s_cbranch_scc1 .ttmps_initialized
s_mov_b32 ttmp4, 0
s_mov_b32 ttmp5, 0
s_bitset1_b32 ttmp8, TTMP8_DEBUG_FLAG_SHIFT
s_bitcmp1_b32 ttmp8, TTMP8_DEBUG_FLAG_SHIFT
s_cbranch_scc1 .ttmps_initialized
s_mov_b32 ttmp4, 0
s_mov_b32 ttmp5, 0
s_bitset1_b32 ttmp8, TTMP8_DEBUG_FLAG_SHIFT
.ttmps_initialized:
s_branch .exit_trap
.profile_trap_handlers:
// Register state at the start of profile_trap_handlers:
//
// ttmp0: PC_LO[31:0] - Contains program counter low bits
// ttmp1: PC_HI[15:0] - Contains program counter high bits
// ttmp2: Contains HW_REG_EXCP_FLAG_PRIV
// ttmp3: Initialized to 0, available for use
// ttmp4: Available - Can be freely used
// ttmp5: Available - Can be freely used
// ttmp6: Initially contains flags - trap ID and halt status - reused after saving
// ttmp7: Contains WGID_Y in high 16 bits, WGID_Z in low 16 bits
// ttmp8: Contains dispatch ID in bits [24:0] and debug flag
// ttmp9: Contains WGID_X
// ttmp10: Available - Used next to save exec_lo
// ttmp11: Contains debug flags - Used next to save exec_hi
// ttmp12: Contains SQ_WAVE_STATE_PRIV
// ttmp13: Contains flag bits for sampling type - HT_FLAG_BIT or STOCH_FLAG_BIT
// ttmp[14:15]: Contains HT or ST buffer base address
//
// v[0:3] contain user shader data that must be preserved/restored
// exec: Contains user's execution mask
s_mov_b64 ttmp[10:11], exec // save exec to ttmp[10:11]
s_mov_b64 exec, 0x1 // turn on lane 0 only
v_readlane_b32 ttmp2, v0, 0
v_readlane_b32 ttmp3, v1, 0 // Save out lane 0s first 2 VGPRs
// At this point, ttmp[4:5], ttmp6 and v[0:1] are free
// Atomically get current sample slot index and select buffer
// pcs_sampling_data_t.buf_write_val (uint64_t) stores:
// Bit 63: current_buffer_id (0 or 1)
// Bits 62-0: current_sample_index_in_buffer
// v0 = 1 (value to add to the low part of buf_write_val)
// v1 = 0 (value to add to the high part of buf_write_val, bit 63 is buffer selector)
v_mov_b32 v0, 1
v_mov_b32 v1, 0
global_atomic_add_u64 v[0:1], v1, v[0:1], ttmp[14:15], scope:SCOPE_SYS th:TH_ATOMIC_RETURN
s_wait_loadcnt 0 // Wait for atomic operation to complete and return value
// At this point, ttmp[4:5] and ttmp6 are free
// v[0:1] (lane 0) now holds the previous value of buf_write_val.
// This previous value gives the slot index for the current sample.
v_readlane_b32 ttmp6, v1, 0x0 // previous buf_write_val[63:32]
s_lshr_b32 ttmp6, ttmp6, TTMP13_BUF_FULL_BIT // ttmp6 = previous_buffer_id (0 or 1, from bit 63 of original uint64_t)
// This ttmp6 is used to select which buffer's metadata (size, watermark, signal) to use.
// It's also used to calculate the base address of the sample buffer.
s_bitset0_b32 ttmp13, TTMP13_BUF_FULL_BIT // Clear our local buffer full flag for now
s_cmp_eq_u32 ttmp6, 0 // store off buf_to_use
s_cbranch_scc1 .skip_bufbit_set // into bit31 of ttmp13
s_bitset1_b32 ttmp13, TTMP13_BUF_FULL_BIT
.skip_bufbit_set:
// ttmp[2:3]=v[0:1]-backup, ttmp[4:5]=free, ttmp6=buf_to_use (also in ttmp13.b31)
// ttmp[10:11]=EXEC backup. ttmp[14:15]=tma
// v[0:1].lane0=local_entry, v[2:3]=original, EXEC=0x1
v_bfe_u32 v1, v1, 0, SAMPLE_INDEX_WIDTH // v[0:1] = new local_entry
// removes bit 31 from v1, returning v1 & 0x7FFFFFFF.
v_readlane_b32 ttmp5, v1, 0 // ttmp5 = high 31 bits of sample index (if index > 2^32-1).
s_cmp_lg_u32 ttmp5, 0 // Check if sample index is very large (overflowed 32 bits).
s_cbranch_scc1 .lost_sample // If ttmp5 > 0, index is too large, treat as lost sample.
s_load_b32 ttmp5, ttmp[14:15], SAMPLE_OFF_BUF_SIZE, scope:SCOPE_CU // ttmp5 = pcs_sampling_data_t.buf_size
v_readlane_b32 ttmp4, v0, 0 // ttmp4 = sample_index_for_current_sample (from v0)
s_wait_kmcnt 0 // Wait for buf_size load.
s_cmp_ge_u32 ttmp4, ttmp5 // if local_entry >= buf_size
s_cbranch_scc1 .lost_sample // If index >= buf_size, buffer is full, sample is lost.
// This also sets TTMP13_BUF_FULL_BIT implicitly by branching.
// Register state before calculating the sample buffer address:
// ttmp2 = backup of original shader's v0
// ttmp3 = backup of original shader's v1
// ttmp4 = sample_index_for_current_sample (from v0)
// ttmp5 = buf_size
// ttmp6 = buffer_id (0 or 1)
// ttmp[10:11] = original shader's [exec_lo, exec_hi]
// ttmp[14:15] = base_address_of_pcs_sampling_data_t (TMA)
// ttmp13.b31 = buffer_id (0 or 1, same as ttmp6)
// v[0:1].lane0 = sample index value from atomic
// v[2:3] = original user shader's v[2:3] values
// exec = backup of user shader's v[0:1]
s_mov_b64 exec, ttmp[2:3] // stash into EXEC to free up ttmp
// Calculate the base address of the correct sample buffer (buffer0 or buffer1).
// The buffers are located after the pcs_sampling_data_t struct header.
// Address = (TMA + SAMPLE_OFF_BYTES_PER_SAMPLE) + (buffer_id * buf_size * 64)
s_mul_i32 ttmp2, ttmp5, ttmp6 // low 32 bits
s_mul_hi_u32 ttmp3, ttmp5, ttmp6 // high 32 bits
// Multiply by 64 bytes per sample slot (shift left by 6 bits)
// This converts from units of samples to units of bytes
s_lshl_b64 ttmp[2:3], ttmp[2:3], 6
s_add_u32 ttmp2, ttmp2, SAMPLE_OFF_BYTES_PER_SAMPLE
s_addc_u32 ttmp3, ttmp3, 0
s_add_u32 ttmp4, ttmp14, ttmp2 // ttmp4 = TMA_base_lo + total_offset_lo. This is low part of &bufferX
s_addc_u32 ttmp5, ttmp15, ttmp3 // ttmp5 = TMA_base_hi + total_offset_hi + carry. This is high part of &bufferX
// ttmp[4:5] now correctly points to the base of the selected sample buffer array
s_bitcmp1_b32 ttmp13, TTMP13_HT_FLAG_BIT // if ttmp13.b22==1, this is hosttrap
s_cbranch_scc1 .fill_sample_ht
s_bitcmp1_b32 ttmp13, TTMP13_STOCH_FLAG_BIT
s_cbranch_scc1 .fill_sample_stoch
s_mov_b64 ttmp[2:3], exec // Restore user v[0:1] backup to ttmp[2:3]
v_readlane_b32 ttmp4, v2, 0 // Backup user v[2:3] to ttmp[4:5] for restore.
v_readlane_b32 ttmp5, v3, 0
s_branch .restore_vector_before_exit_trap
.fill_sample_ht:
// At this point, v[0:1] is local_entry (but v1 is 0)
// v[2:3] is original user-data
// ttmp[2:3] is free
// ttmp[4:5] holds &buffer
// ttmp6 holds buf_to_use
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// [ttmp14:15]=tma, ttmp13.b31 = buf_to_use
// EXEC holds holds backup of original shaders v[0:1]
v_readlane_b32 ttmp6, v0, 0 // ttmp6=local_entry
s_mul_i32 ttmp2, ttmp6, SAMPLE_OFF_BYTES_PER_SAMPLE // into buffer for 64B objects
s_mul_hi_u32 ttmp3, ttmp6, SAMPLE_OFF_BYTES_PER_SAMPLE // ttmp[2:3] now holds the offset
s_add_u32 ttmp2, ttmp2, ttmp4
s_addc_u32 ttmp3, ttmp3, ttmp5 // ttmp[2:3]=&bufferX[local_entry]
v_readlane_b32 ttmp4, v2, 0x0 // ttmp[4:5] now holds backup of
v_readlane_b32 ttmp5, v3, 0x0 // user-data from v[2:3]
v_writelane_b32 v0, ttmp2, 0x0
v_writelane_b32 v1, ttmp3, 0x0 // v[0:1]=&buffer[local_entry]
s_sendmsg_rtn_b64 ttmp[2:3], sendmsg(MSG_RTN_GET_REALTIME)
s_wait_kmcnt 0 // Wait for timestamp
// v[0:1] = &buffer[local_entry]
// v[2:3] = free
// ttmp[2:3] holds the thing we want to store
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma, ttmp13.b31 = buf_to_use
// EXEC holds backup of original shaders v[0:1]
v_writelane_b32 v2, ttmp2, 0 // bring output data to v[2:3]
v_writelane_b32 v3, ttmp3, 0
s_mov_b64 ttmp[2:3], exec // vector stores need EXEC set
s_mov_b64 exec, 1 // so ttmp[2:3] holds it for now
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_TIMESTAMP, scope:SCOPE_SYS // store out timestamp
// v[0:1] = &buffer[local_entry]
// v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma, ttmp13.b31 = buf_to_use
// EXEC is 0x1
s_and_b32 ttmp1, ttmp1, SQ_WAVE_PC_HI_ADDRESS_MASK // Clear out extra data from PC_HI
v_writelane_b32 v2, ttmp0, 0
v_writelane_b32 v3, ttmp1, 0
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_PC_HOST, scope:SCOPE_SYS // store out PC
v_writelane_b32 v2, ttmp10, 0
v_writelane_b32 v3, ttmp11, 0
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_EXEC_LOHI, scope:SCOPE_SYS // store out original EXEC
// Store Workgroup ID X and Y at offset SAMPLE_OFF_WGID_XY (0x10).
// ttmp9 = WGID_X (from first-level handler).
// ttmp7 contains WGID_Y in high 16 bits.
v_writelane_b32 v2, ttmp9, 0 // wg_id_x
s_bfe_u32 ttmp6, ttmp7, (16<<16) // extract bits 15:0, wg_id_y
v_writelane_b32 v3, ttmp6, 0
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_WGID_XY, scope:SCOPE_SYS // store wg_id_x and wg_id_y
// Store Workgroup ID Z and Wave ID at offset SAMPLE_OFF_WGID_Z_WAVE (0x18).
// ttmp7 contains WGID_Z in low 16 bits.
// ttmp11 contains Wave ID in low 6 bits (from EXEC_hi).
s_bfe_u32 ttmp6, ttmp7, (16|16<<16) // extract bits 31:16, wg_id_z
v_writelane_b32 v2, ttmp6, 0
v_writelane_b32 v3, ttmp8, 0x0 // wave_in_wg is bits 29:25
v_lshrrev_b32 v3, 25, v3 // Shift wave_in_wg to 4:0
v_and_b32 v3, v3, WAVE_ID_MASK // put (ttmp8>>25)&0x1f into v3
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_WGID_Z_WAVE, scope:SCOPE_SYS // store wg_id_z and wave_id
// v[0:1] = &buffer[local_entry]
// v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma, ttmp13.b31 = buf_to_use
// EXEC is 0x1
// Get HW_ID1 & 2 with S_GETREG_B32 with size=32 (F8 in upper bits), offset=0, and:
// HW_ID1 = 23 (0x17), HW_ID2 = 24 (0x18)
STORE_HW_ID
// The following is still true as we get ready to jump to correlation ID check
// v[0:1] = &buffer[local_entry]
// v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15=tma, ttmp13.b31 = buf_to_use
// EXEC is 0x1
STORE_CORRELATION_ID
// Ensure all stores have completed before returning and incrementing written_val
s_wait_storecnt 0
// Still true after returning back from correlation ID check
// v[0:1] = &buffer[local_entry], but we no longer need it
// v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma, ttmp13.b31 = buf_to_use
// EXEC is 0x1
//
s_branch .ret_from_fill_sample
.fill_sample_stoch:
// v0 contains local_entry, v1 is free
// v[2:3] is original user-data
// ttmp[2:3] is free
// ttmp[4:5] holds &buffer
// ttmp6 holds buf_to_use
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// [ttmp14:15]=tma, ttmp13.b31 = buf_to_use
// EXEC holds holds backup of original shaders v[0:1]
v_readlane_b32 ttmp6, v0, 0x0 // ttmp2=local_entry
s_mul_i32 ttmp2, ttmp6, SAMPLE_OFF_BYTES_PER_SAMPLE // into buffer for 64B objects
s_mul_hi_u32 ttmp3, ttmp6, SAMPLE_OFF_BYTES_PER_SAMPLE // ttmp[2:3] now holds the offset
s_add_u32 ttmp2, ttmp2, ttmp4
s_addc_u32 ttmp3, ttmp3, ttmp5 // ttmp[2:3]=&bufferX[local_entry]
v_readlane_b32 ttmp4, v2, 0x0 // ttmp[4:5] now holds backup of
v_readlane_b32 ttmp5, v3, 0x0 // user-data from v[2:3]
v_writelane_b32 v0, ttmp2, 0x0
v_writelane_b32 v1, ttmp3, 0x0 // v[0:1]=&buffer[local_entry]
s_sendmsg_rtn_b64 ttmp[2:3], sendmsg(MSG_RTN_GET_REALTIME)
s_wait_kmcnt 0 // Wait for timestamp
// v[0:1] = &buffer[local_entry]
// v[2:3] = free
// ttmp[2:3] holds the thing we want to store
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma, ttmp13.b31 = buf_to_use
// EXEC holds backup of original shaders v[0:1]
v_writelane_b32 v2, ttmp2, 0 // bring output data to v[2:3]
v_writelane_b32 v3, ttmp3, 0
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_TIMESTAMP, scope:SCOPE_SYS // store out timestamp
// v[0:1] = &buffer[local_entry]
// v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma, ttmp13.b31 = buf_to_use
// EXEC is 0x1
v_writelane_b32 v2, ttmp10, 0
v_writelane_b32 v3, ttmp11, 0
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_EXEC_LOHI, scope:SCOPE_SYS // store out original EXEC
v_writelane_b32 v2, ttmp9, 0 // wg_id_x
s_bfe_u32 ttmp6, ttmp7, (0 | (16 << 16)) // extract bits 15:0, wg_id_y
v_writelane_b32 v3, ttmp6, 0
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_WGID_XY, scope:SCOPE_SYS // store wg_id_x and wg_id_y
s_bfe_u32 ttmp6, ttmp7, (16|16<<16) // extract bits 31:16, wg_id_z
v_writelane_b32 v2, ttmp6, 0 // put wg_id_z in v2
v_writelane_b32 v3, ttmp8, 0x0 // wave_in_wg is bits 29:25
v_lshrrev_b32 v3, 25, v3 // Shift wave_in_wg to 4:0
v_and_b32 v3, v3, WAVE_ID_MASK // put (ttmp8>>25)&0x1f into v3
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_WGID_Z_WAVE, scope:SCOPE_SYS // store wg_id_z and wave_id
STORE_HW_ID
//Read SNAPSHOT Data
s_getreg_b32 ttmp6, HW_REG_SQ_PERF_SNAPSHOT_DATA1
v_writelane_b32 v2, ttmp6, 0x0
s_getreg_b32 ttmp6, HW_REG_SQ_PERF_SNAPSHOT_DATA2
v_writelane_b32 v3, ttmp6, 0x0
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_SNAPSHOT_DATA + 4, scope:SCOPE_SYS // store snapshot DATA1 and DATA2
s_getreg_b32 ttmp2, HW_REG_SQ_PERF_SNAPSHOT_DATA
v_writelane_b32 v2, ttmp2, 0
global_store_b32 v[0:1], v2, off, offset:SAMPLE_OFF_SNAPSHOT_DATA, scope:SCOPE_SYS // store perf snapshot DATA
s_getreg_b32 ttmp6, HW_REG_SQ_PERF_SNAPSHOT_PC_LO
v_writelane_b32 v2, ttmp6, 0x0
s_getreg_b32 ttmp6, HW_REG_SQ_PERF_SNAPSHOT_PC_HI
v_writelane_b32 v3, ttmp6, 0x0
global_store_b64 v[0:1], v[2:3], off, offset:SAMPLE_OFF_PC_HOST, scope:SCOPE_SYS // store PC_HI:PC_LO
// The following is still true as we get ready to jump to correlation ID check
// v[0:1] = &buffer[local_entry]
// v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma, ttmp13.b31 tells us buf_to_use
// EXEC is 0x1
STORE_CORRELATION_ID
// Ensure all stores have completed before returning and incrementing written_val
s_wait_storecnt 0
.ret_from_fill_sample:
// v[0:1] = free
// v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma, ttmp13.b31 tells us buf_to_use
// EXEC is 0x1
// Sample data has been written to the device buffer.
// Now, atomically increment the count of written samples for the current buffer.
// This is pcs_sampling_data_t.buf_written_val0 or buf_written_val1.
s_lshr_b32 ttmp6, ttmp13, 31 // ttmp6 is buf_to_use
s_mulk_i32 ttmp6, 0x10 // ttmp6=offset from
// written_val0 to written_val_X
s_add_u32 ttmp14, ttmp14, ttmp6 // now ttmp[14:15] points to base for
s_addc_u32 ttmp15, ttmp15, 0 // buf_written_valX atomic operation
// Atomically increment the chosen buf_written_val.
// v0 = 0 (value to add - low part), v1 = 1 (value to add - high part, effectively just adding 1 to uint32_t)
v_mov_b32 v0, 0 // want to atomic increment
v_mov_b32 v1, 1 // buf_written_valX
global_atomic_add_u32 v0, v0, v1, ttmp[14:15], offset:SAMPLE_OFF_BUF_WRITTEN_VAL, scope:SCOPE_SYS th:TH_ATOMIC_RETURN
s_wait_loadcnt 0
// v0 = done, v1 = free, v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=buf_written_valX-0x10, EXEC=0x1
// Check Watermark and Signal Host
s_mov_b64 exec, ttmp[4:5] // stash users v[2:3] in EXEC
s_load_b32 ttmp5, ttmp[14:15], 0x14, scope:SCOPE_CU // load watermark into ttmp5
v_readlane_b32 ttmp4, v0, 0 // put done into ttmp4
s_wait_kmcnt 0 // wait for watermark to load
s_cmp_lg_u32 ttmp4, ttmp5 // if done != watermark, exit
s_add_u32 ttmp4, ttmp4, 1 // ttmp4 is now current_sample_count (count_before_inc + 1)
s_cmp_lt_u32 ttmp4, ttmp5 // if (current_sample_count < watermark), don't signal
s_mov_b64 ttmp[4:5], exec // restore users v[2:3]
s_mov_b64 exec, 1
s_cbranch_scc1 .restore_vector_before_exit_trap
.send_signal:
// v[0:3] = free, ttmp[2:5] = backups of original v[0:3], ttmp6=free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=buf_written_valX-0x10, EXEC=old copy of original shader v[2:3]
// write done-signal and optional interrupt
// Watermark reached or exceeded. Signal the host.
// Load the hsa_signal_t handle for the current buffer.
// done_sig0 is at offset 0x18. done_sig1 is at 0x28.
// addr = ttmp[14:15] + 0x18 + (buffer_id * 0x10).
// ttmp0 still holds buffer_id * 0x10.
s_load_b64 ttmp[14:15], ttmp[14:15], SAMPLE_OFF_DONE_SIG0, scope:SCOPE_CU // load done_sig into ttmp[14:15]
s_mov_b64 exec, 1
s_wait_kmcnt 0
v_mov_b32 v0, 0
v_mov_b32 v1, 0 // value to store into v[0:1]
v_writelane_b32 v2, ttmp14, 0
v_writelane_b32 v3, ttmp15, 0 // Put signal address into v[2:3]
global_store_b64 v[2:3], v[0:1], off, offset:SAMPLE_OFF_SIGNAL_VALUE, scope:SCOPE_SYS // zero out signal value
s_load_b32 ttmp6, ttmp[14:15], 0x18, scope:SCOPE_CU // load event_id into ttmp6
s_load_b64 ttmp[14:15], ttmp[14:15], SAMPLE_OFF_EVENT_MAILBOX0, scope:SCOPE_CU // load event mailbox ptr into 14:15
s_wait_kmcnt 0
s_cmp_eq_u64 ttmp[14:15], 0 // null mailbox means no interrupt
s_cbranch_scc1 .restore_vector_before_exit_trap
s_cmp_eq_u32 ttmp6, 0 // event_id zero means no interrupt
s_cbranch_scc1 .restore_vector_before_exit_trap
v_writelane_b32 v2, ttmp14, 0
v_writelane_b32 v3, ttmp15, 0 // Put mailbox address into v[2:3]
s_wait_storecnt 0
v_writelane_b32 v0, ttmp6, 0x0 // put event_id into v0
global_store_b32 v[2:3], v0, off, offset:0x0, scope:SCOPE_SYS // Send event ID to the mailbox
s_wait_storecnt 0
s_mov_b32 ttmp14, m0 // save off m0
v_readlane_b32 ttmp15, v0, 0 // Put ID into message payload
s_mov_b32 m0, ttmp15
s_sendmsg sendmsg(MSG_INTERRUPT) // send interrupt message
s_wait_kmcnt 0
s_mov_b32 m0, ttmp14 // restore m0
// v[0:1] = free
// v[2:3] = free
// ttmp[2:3] holds backup of original shaders v[0:1]
// ttmp[4:5] holds backup of original shaders v[2:3]
// ttmp6 = free
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=somewhere in tma region, EXEC is junk
.restore_vector_before_exit_trap:
v_writelane_b32 v2, ttmp4, 0
v_writelane_b32 v3, ttmp5, 0
.lost_sample:
// v0 contains local_entry, v1 is free
// v[2:3] is original user-data
// ttmp[2:3] [local_entry, buf_size]
// ttmp[4:5] = free
// ttmp6=buf_to_use (also in ttmp13.b31)
// ttmp[10:11] holds original shaders [exec_lo,exec_hi]
// ttmp[14:15]=tma
// EXEC=0x1
// Restore vector registers before exiting
s_bitcmp1_b32 ttmp13, TTMP13_STOCH_FLAG_BIT // Check if stochastic sampling
s_cbranch_scc0 .lost_sample_restore // If not, just restore and exit
s_getreg_b32 ttmp6, HW_REG_SQ_PERF_SNAPSHOT_PC_HI // Read PC_HI to release lock
.lost_sample_restore:
v_writelane_b32 v0, ttmp2, 0 // restore v[0:1] to user data
v_writelane_b32 v1, ttmp3, 0
s_mov_b64 exec, ttmp[10:11] // restore exec mask
.exit_trap:
// Restore SQ_WAVE_STATUS.
s_and_b64 exec, exec, exec // Restore STATUS.EXECZ, not writable by s_setreg_b32
s_and_b64 vcc, vcc, vcc // Restore STATUS.VCCZ, not writable by s_setreg_b32
s_setreg_b32 hwreg(HW_REG_STATE_PRIV, 0, SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT), ttmp12
s_lshr_b32 ttmp12, ttmp12, (SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT + 1)
s_setreg_b32 hwreg(HW_REG_STATE_PRIV, SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT + 1, 32 - SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT - 1), ttmp12
s_and_b64 exec, exec, exec // Restore STATUS.EXECZ, not writable by s_setreg_b32
s_and_b64 vcc, vcc, vcc // Restore STATUS.VCCZ, not writable by s_setreg_b32
s_setreg_b32 hwreg(HW_REG_STATE_PRIV, 0, SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT), ttmp12
s_lshr_b32 ttmp12, ttmp12, (SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT + 1)
s_setreg_b32 hwreg(HW_REG_STATE_PRIV, SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT + 1, 32 - SQ_WAVE_STATE_PRIV_BARRIER_COMPLETE_SHIFT - 1), ttmp12
// Return to original (possibly modified) PC.
s_rfe_b64 [ttmp0, ttmp1]
s_rfe_b64 [ttmp0, ttmp1]
.parked:
s_trap 0x2
s_branch .parked
s_trap 0x2
s_branch .parked
// Add s_code_end padding so instruction prefetch always has something to read.
.rept (256 - ((. - trap_entry) % 64)) / 4