From 6f37386eb22d469d61785634619707ea1c33b09a Mon Sep 17 00:00:00 2001 From: Tony Gutierrez Date: Wed, 15 Jan 2025 12:13:55 -0800 Subject: [PATCH] rocr: Flags to alloc queue buf/struct in dev mem This builds on a prior change that allowed for allocating a user-mode queue's packet buffer in device memory to also allocate the queue struct in device memory. This provides additional latency benefits particularly for cases where dispatches are performed from the GPU itself. Flags are added to support the various use cases. [ROCm/ROCR-Runtime commit: 6e3c375bf151d2b96ced44e9004979ee52229739] --- projects/rocr-runtime/libhsakmt/src/memory.c | 1 - .../runtime/hsa-runtime/core/inc/agent.h | 44 +++++++++--------- .../hsa-runtime/core/inc/amd_aie_agent.h | 9 ++-- .../hsa-runtime/core/inc/amd_aie_aql_queue.h | 3 +- .../hsa-runtime/core/inc/amd_aql_queue.h | 10 ++-- .../hsa-runtime/core/inc/amd_cpu_agent.h | 5 +- .../hsa-runtime/core/inc/amd_gpu_agent.h | 11 ++--- .../hsa-runtime/core/inc/amd_memory_region.h | 7 +-- .../runtime/hsa-runtime/core/inc/host_queue.h | 4 +- .../hsa-runtime/core/inc/intercept_queue.h | 10 +++- .../runtime/hsa-runtime/core/inc/queue.h | 46 ++++++++++--------- .../core/runtime/amd_aie_agent.cpp | 28 ++++++++--- .../core/runtime/amd_aie_aql_queue.cpp | 18 +++++--- .../core/runtime/amd_aql_queue.cpp | 38 +++++++++------ .../core/runtime/amd_blit_kernel.cpp | 2 +- .../core/runtime/amd_cpu_agent.cpp | 7 ++- .../core/runtime/amd_gpu_agent.cpp | 31 ++++++++++--- .../core/runtime/amd_memory_region.cpp | 17 +++---- .../hsa-runtime/core/runtime/host_queue.cpp | 7 +-- .../runtime/hsa-runtime/core/runtime/hsa.cpp | 15 +++++- .../core/runtime/intercept_queue.cpp | 6 +-- .../runtime/hsa-runtime/core/util/flag.h | 6 +-- .../runtime/hsa-runtime/inc/hsa_ext_amd.h | 26 +++++++++++ 23 files changed, 220 insertions(+), 131 deletions(-) diff --git a/projects/rocr-runtime/libhsakmt/src/memory.c b/projects/rocr-runtime/libhsakmt/src/memory.c index 6d36b77d59..e7230c1015 100644 --- a/projects/rocr-runtime/libhsakmt/src/memory.c +++ b/projects/rocr-runtime/libhsakmt/src/memory.c @@ -150,7 +150,6 @@ HSAKMT_STATUS HSAKMTAPI hsaKmtAllocMemoryAlign(HSAuint32 PreferredNode, *MemoryAddress = NULL; if ((MemFlags.ui32.CoarseGrain && MemFlags.ui32.ExtendedCoherent) || - (MemFlags.ui32.CoarseGrain && MemFlags.ui32.Uncached) || (MemFlags.ui32.ExtendedCoherent && MemFlags.ui32.Uncached)) return HSAKMT_STATUS_INVALID_PARAMETER; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/agent.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/agent.h index 491b4a6d95..a157fbf252 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/agent.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/agent.h @@ -255,29 +255,29 @@ class Agent : public Checked<0xF6BC25EB17E6F917> { virtual hsa_status_t IterateCache(hsa_status_t (*callback)(hsa_cache_t cache, void* data), void* data) const = 0; - // @brief Create queue. - // - // @param [in] size Number of packets the queue is expected to hold. Must be a - // power of 2 greater than 0. - // @param [in] queue_type Queue type. - // @param [in] event_callback Callback invoked for every - // asynchronous event related to the newly created queue. May be NULL.The HSA - // runtime passes three arguments to the callback : a code identifying the - // event that triggered the invocation, a pointer to the queue where the event - // originated, and the application data. - // @param [in] data Application data that is passed to @p callback. - // @param [in] private_segment_size A hint to indicate the maximum expected - // private segment usage per work-item, in bytes. - // @param [in] group_segment_size A hint to indicate the maximum expected - // group segment usage per work-group, in bytes. - // @param[out] queue Memory location where the HSA runtime stores a pointer - // to the newly created queue. - // - // @retval HSA_STATUS_SUCCESS The queue has been created successfully. - virtual hsa_status_t QueueCreate(size_t size, hsa_queue_type32_t queue_type, + /// @brief Create queue. + /// + /// @param [in] size Number of packets the queue is expected to hold. Must be a + /// power of 2 greater than 0. + /// @param [in] queue_type Queue type. + /// @param [in] flags Flags to specify queue attributes on creation. + /// @param [in] event_callback Callback invoked for every + /// asynchronous event related to the newly created queue. May be NULL.The HSA + /// runtime passes three arguments to the callback : a code identifying the + /// event that triggered the invocation, a pointer to the queue where the event + /// originated, and the application data. + /// @param [in] data Application data that is passed to @p callback. + /// @param [in] private_segment_size A hint to indicate the maximum expected + /// private segment usage per work-item, in bytes. + /// @param [in] group_segment_size A hint to indicate the maximum expected + /// group segment usage per work-group, in bytes. + /// @param[out] queue Memory location where the HSA runtime stores a pointer + /// to the newly created queue. + /// + /// @retval HSA_STATUS_SUCCESS The queue has been created successfully. + virtual hsa_status_t QueueCreate(size_t size, hsa_queue_type32_t queue_type, uint64_t flags, HsaEventCallback event_callback, void* data, - uint32_t private_segment_size, - uint32_t group_segment_size, + uint32_t private_segment_size, uint32_t group_segment_size, Queue** queue) = 0; // @brief Query the value of an attribute. diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aie_agent.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aie_agent.h index 798022754c..938d6182fb 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aie_agent.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aie_agent.h @@ -78,11 +78,10 @@ public: hsa_status_t GetInfo(hsa_agent_info_t attribute, void *value) const override; - hsa_status_t QueueCreate(size_t size, hsa_queue_type32_t queue_type, - core::HsaEventCallback event_callback, void *data, - uint32_t private_segment_size, - uint32_t group_segment_size, - core::Queue **queue) override; + hsa_status_t QueueCreate(size_t size, hsa_queue_type32_t queue_type, uint64_t flags, + core::HsaEventCallback event_callback, void* data, + uint32_t private_segment_size, uint32_t group_segment_size, + core::Queue** queue) override; // @brief Override from core::Agent. const std::vector& supported_isas() const override { diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aie_aql_queue.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aie_aql_queue.h index cc5bfb8fe3..8e6739722f 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aie_aql_queue.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aie_aql_queue.h @@ -68,7 +68,8 @@ class AieAqlQueue : public core::Queue, return queue->IsType(&rtti_id()); } - AieAqlQueue(AieAgent *agent, size_t req_size_pkts, uint32_t node_id); + AieAqlQueue(core::SharedQueue* shared_queue, AieAgent* agent, size_t req_size_pkts, + uint32_t node_id, uint64_t flags); ~AieAqlQueue(); hsa_status_t Inactivate() override; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aql_queue.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aql_queue.h index ff040a8392..33136756d5 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aql_queue.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_aql_queue.h @@ -63,9 +63,9 @@ class AqlQueue : public core::Queue, private core::LocalSignal, public core::Doo static __forceinline bool IsType(core::Queue* queue) { return queue->IsType(&rtti_id()); } // Acquires/releases queue resources and requests HW schedule/deschedule. - AqlQueue(GpuAgent* agent, size_t req_size_pkts, HSAuint32 node_id, - ScratchInfo& scratch, core::HsaEventCallback callback, - void* err_data, bool is_kv = false); + AqlQueue(core::SharedQueue* shared_queue, GpuAgent* agent, size_t req_size_pkts, + HSAuint32 node_id, ScratchInfo& scratch, core::HsaEventCallback callback, void* err_data, + uint64_t flags, bool is_kv = false); ~AqlQueue(); @@ -236,7 +236,9 @@ class AqlQueue : public core::Queue, private core::LocalSignal, public core::Doo // (De)allocates and (de)registers ring_buf_. void AllocRegisteredRingBuffer(uint32_t queue_size_pkts); - void FreeRegisteredRingBuffer(); + + /// @brief Frees the queue's packet ring buffer and its queue struct. + void FreeQueueMemory(); /// @brief Abstracts the file handle use for double mapping queues. void CloseRingBufferFD(const char* ring_buf_shm_path, int fd) const; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_cpu_agent.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_cpu_agent.h index 21fb66b2d7..2c3f9afa4d 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_cpu_agent.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_cpu_agent.h @@ -100,10 +100,9 @@ class CpuAgent : public core::Agent { hsa_status_t GetInfo(hsa_agent_info_t attribute, void* value) const override; // @brief Override from core::Agent. - hsa_status_t QueueCreate(size_t size, hsa_queue_type32_t queue_type, + hsa_status_t QueueCreate(size_t size, hsa_queue_type32_t queue_type, uint64_t flags, core::HsaEventCallback event_callback, void* data, - uint32_t private_segment_size, - uint32_t group_segment_size, + uint32_t private_segment_size, uint32_t group_segment_size, core::Queue** queue) override; // @brief Override from core::Agent. diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_gpu_agent.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_gpu_agent.h index d2dae376f7..6e071719c8 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_gpu_agent.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_gpu_agent.h @@ -328,10 +328,9 @@ class GpuAgent : public GpuAgentInt { hsa_status_t GetInfo(hsa_agent_info_t attribute, void* value) const override; // @brief Override from core::Agent. - hsa_status_t QueueCreate(size_t size, hsa_queue_type32_t queue_type, + hsa_status_t QueueCreate(size_t size, hsa_queue_type32_t queue_type, uint64_t flags, core::HsaEventCallback event_callback, void* data, - uint32_t private_segment_size, - uint32_t group_segment_size, + uint32_t private_segment_size, uint32_t group_segment_size, core::Queue** queue) override; // @brief Decrement GWS ref count. @@ -436,7 +435,7 @@ class GpuAgent : public GpuAgentInt { if (t0_.GPUClockCounter == t1_.GPUClockCounter) SyncClocks(); } - // @brief Override from AMD::GpuAgentInt. + /// @brief Override from AMD::GpuAgentInt. __forceinline bool is_xgmi_cpu_gpu() const { return xgmi_cpu_gpu_; } const size_t MAX_SCRATCH_APERTURE_PER_XCC = (1ULL << 32); @@ -830,8 +829,8 @@ class GpuAgent : public GpuAgentInt { // structure for stochastic sampling pcs_data_t pcs_stochastic_data_; - // @bried XGMI CPU<->GPU - bool xgmi_cpu_gpu_; + /// @brief XGMI CPU<->GPU + bool xgmi_cpu_gpu_ = false; }; } // namespace amd diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_memory_region.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_memory_region.h index 79f6631550..68a161641d 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_memory_region.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_memory_region.h @@ -171,7 +171,7 @@ class MemoryRegion : public core::MemoryRegion { return static_cast(mem_props_.MemoryClockMax); } - __forceinline size_t GetPageSize() const { return kPageSize(); } + __forceinline static size_t GetPageSize() { return kPageSize_; } __forceinline const HsaMemFlags &mem_flags() const { return mem_flag_; } __forceinline const HsaMemMapFlags &map_flags() const { return map_flag_; } @@ -199,10 +199,7 @@ private: // fragments of the block routing to the same MemoryRegion. mutable KernelMutex access_lock_; - static __forceinline const size_t& kPageSize() { - static size_t kPageSize_ = sysconf(_SC_PAGESIZE); - return kPageSize_; - } + static const size_t kPageSize_; // Determine access type allowed to requesting device hsa_amd_memory_pool_access_t GetAccessInfo(const core::Agent& agent, diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/host_queue.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/host_queue.h index 4219068afb..5464c53a91 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/host_queue.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/host_queue.h @@ -54,8 +54,8 @@ class HostQueue : public Queue { public: static __forceinline bool IsType(core::Queue* queue) { return queue->IsType(&rtti_id()); } - HostQueue(hsa_region_t region, uint32_t ring_size, hsa_queue_type32_t type, - uint32_t features, hsa_signal_t doorbell_signal); + HostQueue(core::SharedQueue* shared_queue, hsa_region_t region, uint32_t ring_size, + hsa_queue_type32_t type, uint32_t features, hsa_signal_t doorbell_signal); ~HostQueue(); diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/intercept_queue.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/intercept_queue.h index 72fc4aaf64..f9b8df0f8e 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/intercept_queue.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/intercept_queue.h @@ -63,12 +63,18 @@ class QueueWrapper : public Queue { public: std::unique_ptr wrapped; - explicit QueueWrapper(std::unique_ptr queue) : Queue(), wrapped(std::move(queue)) { + explicit QueueWrapper(std::unique_ptr queue) + : Queue(static_cast(core::Runtime::runtime_singleton_->system_allocator()( + sizeof(core::SharedQueue), 4096, 0, 0)), + 0), + wrapped(std::move(queue)) { memcpy(&amd_queue_, &wrapped->amd_queue_, sizeof(amd_queue_)); wrapped->set_public_handle(wrapped.get(), public_handle_); } - ~QueueWrapper() {} + ~QueueWrapper() { + if (shared_queue_) core::Runtime::runtime_singleton_->system_deallocator()(shared_queue_); + } hsa_status_t Inactivate() override { return wrapped->Inactivate(); } hsa_status_t SetPriority(HSA_QUEUE_PRIORITY priority) override { diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/queue.h b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/queue.h index 185943e466..2a3c551be1 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/queue.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/queue.h @@ -158,16 +158,6 @@ struct SharedQueue { Queue* core_queue; }; -class LocalQueue { - public: - LocalQueue(int mem_flags) : local_queue_(mem_flags) {} - LocalQueue(int agent_node_id, int mem_flags) : local_queue_(agent_node_id, mem_flags) {} - SharedQueue* queue() const { return local_queue_.shared_object(); } - - private: - Shared local_queue_; -}; - /// @brief Class Queue which encapsulate user mode queues and /// provides Api to access its Read, Write indices using Acquire, /// Release and Relaxed semantics. @@ -176,18 +166,18 @@ Queue is intended to be an pure interface class and may be wrapped or replaced by tools. All funtions other than Convert and public_handle must be virtual. */ -class Queue : public Checked<0xFA3906A679F9DB49>, private LocalQueue { +class Queue : public Checked<0xFA3906A679F9DB49> { public: - Queue(int mem_flags = 0) : LocalQueue(mem_flags), amd_queue_(queue()->amd_queue) { - queue()->core_queue = this; - public_handle_ = Convert(this); - pcie_write_ordering_ = false; - } + Queue(SharedQueue* shared_queue, uint64_t queue_flags) + : Queue(shared_queue, queue_flags, false) {} - Queue(int agent_node_id, int mem_flags) : LocalQueue(agent_node_id, mem_flags), amd_queue_(queue()->amd_queue) { - queue()->core_queue = this; + Queue(SharedQueue* shared_queue, uint64_t queue_flags, bool pcie_write_ordering) + : amd_queue_(shared_queue->amd_queue), + shared_queue_(shared_queue), + flags_(queue_flags), + pcie_write_ordering_(pcie_write_ordering) { public_handle_ = Convert(this); - pcie_write_ordering_ = false; + shared_queue->core_queue = this; } virtual ~Queue() {} @@ -386,9 +376,18 @@ class Queue : public Checked<0xFA3906A679F9DB49>, private LocalQueue { bool IsType(rtti_t id) { return _IsA(id); } - bool needsPcieOrdering() const { return pcie_write_ordering_; } + /// @brief Used to determine if the queue's packet buffer was allocated + /// in the agent's local device memory. + bool IsDeviceMemRingBuf() const { + return (flags_ & HSA_AMD_QUEUE_CREATE_DEVICE_MEM_RING_BUF) != 0; + } + /// @brief Used to determine if the queue descriptor was allocated in + /// the agent's local device memory. + bool IsDeviceMemQueueDescriptor() const { + return (flags_ & HSA_AMD_QUEUE_CREATE_DEVICE_MEM_QUEUE_DESCRIPTOR) != 0; + } - void setPcieOrdering(bool val) { pcie_write_ordering_ = val; } + bool needsPcieOrdering() const { return pcie_write_ordering_; } protected: static void set_public_handle(Queue* ptr, hsa_queue_t* handle) { @@ -400,6 +399,8 @@ class Queue : public Checked<0xFA3906A679F9DB49>, private LocalQueue { virtual bool _IsA(rtti_t id) const = 0; + SharedQueue* shared_queue_; + hsa_queue_t* public_handle_; /// Next available queue id. @@ -410,7 +411,8 @@ class Queue : public Checked<0xFA3906A679F9DB49>, private LocalQueue { // HSA Queue ID - used to bind a unique ID static std::atomic hsa_queue_counter_; - bool pcie_write_ordering_; + const uint64_t flags_; + bool pcie_write_ordering_ = false; DISALLOW_COPY_AND_ASSIGN(Queue); }; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aie_agent.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aie_agent.cpp index 3a9b972542..cb3428ce92 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aie_agent.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aie_agent.cpp @@ -261,11 +261,16 @@ hsa_status_t AieAgent::GetInfo(hsa_agent_info_t attribute, void *value) const { return HSA_STATUS_SUCCESS; } -hsa_status_t AieAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type, - core::HsaEventCallback event_callback, - void *data, uint32_t private_segment_size, - uint32_t group_segment_size, - core::Queue **queue) { +hsa_status_t AieAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type, uint64_t flags, + core::HsaEventCallback event_callback, void* data, + uint32_t private_segment_size, uint32_t group_segment_size, + core::Queue** queue) { + if ((flags & HSA_AMD_QUEUE_CREATE_DEVICE_MEM_RING_BUF) != 0 || + (flags & HSA_AMD_QUEUE_CREATE_DEVICE_MEM_QUEUE_DESCRIPTOR) != 0) { + // AIE agents do not currently support queue creation in device memory. + return HSA_STATUS_ERROR_INVALID_ARGUMENT; + } + if (!IsPowerOfTwo(size)) { return HSA_STATUS_ERROR_INVALID_ARGUMENT; } @@ -274,7 +279,18 @@ hsa_status_t AieAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type, return HSA_STATUS_ERROR_INVALID_ARGUMENT; } - auto aql_queue(new AieAqlQueue(this, size, node_id())); + core::SharedQueue* shared_queue = + static_cast(core::Runtime::runtime_singleton_->system_allocator()( + sizeof(core::SharedQueue), MemoryRegion::GetPageSize(), 0, node_id())); + + if (!shared_queue) return HSA_STATUS_ERROR_OUT_OF_RESOURCES; + + auto aql_queue(new AieAqlQueue(shared_queue, this, size, node_id(), flags)); + if (aql_queue == nullptr) { + core::Runtime::runtime_singleton_->system_deallocator()(shared_queue); + return HSA_STATUS_ERROR_OUT_OF_RESOURCES; + } + *queue = aql_queue; return HSA_STATUS_SUCCESS; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aie_aql_queue.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aie_aql_queue.cpp index afee21c629..e182ffaa39 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aie_aql_queue.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aie_aql_queue.cpp @@ -65,10 +65,13 @@ namespace rocr { namespace AMD { -AieAqlQueue::AieAqlQueue(AieAgent *agent, size_t req_size_pkts, - uint32_t node_id) - : Queue(0, 0), LocalSignal(0, false), DoorbellSignal(signal()), - agent_(*agent), active_(false) { +AieAqlQueue::AieAqlQueue(core::SharedQueue* shared_queue, AieAgent* agent, size_t req_size_pkts, + uint32_t node_id, uint64_t flags) + : Queue(shared_queue, flags), + LocalSignal(0, false), + DoorbellSignal(signal()), + agent_(*agent), + active_(false) { if (agent_.device_type() != core::Agent::DeviceType::kAmdAieDevice) { throw AMD::hsa_exception( HSA_STATUS_ERROR_INVALID_AGENT, @@ -105,9 +108,10 @@ AieAqlQueue::AieAqlQueue(AieAgent *agent, size_t req_size_pkts, AieAqlQueue::~AieAqlQueue() { AieAqlQueue::Inactivate(); - if (ring_buf_) { - agent_.system_deallocator()(ring_buf_); - } + + if (ring_buf_) agent_.system_deallocator()(ring_buf_); + + if (shared_queue_) core::Runtime::runtime_singleton_->system_deallocator()(shared_queue_); } hsa_status_t AieAqlQueue::Inactivate() { diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aql_queue.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aql_queue.cpp index 47a6f3876b..1be77c6063 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aql_queue.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aql_queue.cpp @@ -73,12 +73,12 @@ namespace rocr { namespace AMD { - #define SCRATCH_ALT_RATIO 4 -AqlQueue::AqlQueue(GpuAgent* agent, size_t req_size_pkts, HSAuint32 node_id, ScratchInfo& scratch, - core::HsaEventCallback callback, void* err_data, bool is_kv) - : Queue(agent->node_id(), agent->isMES() ? (MemoryRegion::AllocateGTTAccess | MemoryRegion::AllocateNonPaged) : 0), +AqlQueue::AqlQueue(core::SharedQueue* shared_queue, GpuAgent* agent, size_t req_size_pkts, + HSAuint32 node_id, ScratchInfo& scratch, core::HsaEventCallback callback, + void* err_data, uint64_t flags, bool is_kv) + : Queue(shared_queue, flags, !agent->is_xgmi_cpu_gpu()), LocalSignal(0, false), DoorbellSignal(signal()), ring_buf_(nullptr), @@ -129,7 +129,7 @@ AqlQueue::AqlQueue(GpuAgent* agent, size_t req_size_pkts, HSAuint32 node_id, Scr // Allocate the AQL packet ring buffer. AllocRegisteredRingBuffer(queue_size_pkts); if (ring_buf_ == nullptr) throw std::bad_alloc(); - MAKE_NAMED_SCOPE_GUARD(RingGuard, [&]() { FreeRegisteredRingBuffer(); }); + MAKE_NAMED_SCOPE_GUARD(RingGuard, [&]() { FreeQueueMemory(); }); // Fill the ring buffer with invalid packet headers. // Leave packet content uninitialized to help track errors. @@ -350,7 +350,6 @@ AqlQueue::AqlQueue(GpuAgent* agent, size_t req_size_pkts, HSAuint32 node_id, Scr if (!core::Runtime::runtime_singleton_->flag().cu_mask_skip_init()) SetCUMasking(0, nullptr); active_ = true; - setPcieOrdering(agent->is_xgmi_cpu_gpu()); PM4IBGuard.Dismiss(); RingGuard.Dismiss(); @@ -386,10 +385,11 @@ AqlQueue::~AqlQueue() { if (queue_scratch_.main_queue_base) agent_->ReleaseQueueMainScratch(queue_scratch_); if (queue_scratch_.alt_queue_base) agent_->ReleaseQueueAltScratch(queue_scratch_); - FreeRegisteredRingBuffer(); exception_signal_->WaitingDec(); exception_signal_->DestroySignal(); HSA::hsa_signal_destroy(amd_queue_.queue_inactive_signal); + FreeQueueMemory(); + if (core::g_use_interrupt_wait) { ScopedAcquire lock(&queue_lock()); queue_count()--; @@ -737,9 +737,10 @@ void AqlQueue::AllocRegisteredRingBuffer(uint32_t queue_size_pkts) { ring_buf_alloc_bytes_ = queue_size_pkts * sizeof(core::AqlPacket); assert(IsMultipleOf(ring_buf_alloc_bytes_, 4096) && "Ring buffer sizes must be 4KiB aligned."); - if (core::Runtime::runtime_singleton_->flag().dev_mem_queue()) { - ring_buf_ = agent_->finegrain_allocator()(ring_buf_alloc_bytes_, - core::MemoryRegion::AllocateUncached); + if (IsDeviceMemRingBuf()) { + ring_buf_ = agent_->coarsegrain_allocator()( + ring_buf_alloc_bytes_, + core::MemoryRegion::AllocateExecutable | core::MemoryRegion::AllocateUncached); } else { ring_buf_ = agent_->system_allocator()( ring_buf_alloc_bytes_, 0x1000, @@ -755,7 +756,16 @@ void AqlQueue::AllocRegisteredRingBuffer(uint32_t queue_size_pkts) { } } -void AqlQueue::FreeRegisteredRingBuffer() { +void AqlQueue::FreeQueueMemory() { + if (shared_queue_) { + if (IsDeviceMemQueueDescriptor()) + agent_->coarsegrain_deallocator()(shared_queue_); + else + core::Runtime::runtime_singleton_->system_deallocator()(shared_queue_); + + shared_queue_ = nullptr; + } + if ((agent_->profile() == HSA_PROFILE_FULL) && queue_full_workaround_) { #ifdef __linux__ munmap(ring_buf_, ring_buf_alloc_bytes_); @@ -767,8 +777,8 @@ void AqlQueue::FreeRegisteredRingBuffer() { #endif } else { if (ring_buf_) { - if (core::Runtime::runtime_singleton_->flag().dev_mem_queue()) { - agent_->finegrain_deallocator()(ring_buf_); + if (IsDeviceMemRingBuf()) { + agent_->coarsegrain_deallocator()(ring_buf_); } else { agent_->system_deallocator()(ring_buf_); } @@ -1664,7 +1674,7 @@ void AqlQueue::ExecutePM4(uint32_t* cmd_data, size_t cmd_size_b, hsa_fence_scope // Overwrite the AQL invalid header (first dword) last. // This prevents the slot from being read until it's fully written. memcpy(&queue_slot[1], &slot_data[1], slot_size_b - sizeof(uint32_t)); - if (core::Runtime::runtime_singleton_->flag().dev_mem_queue() && !agent_->is_xgmi_cpu_gpu()) { + if (IsDeviceMemRingBuf() && needsPcieOrdering()) { // Ensure the packet body is written as header may get reordered when writing over PCIE _mm_sfence(); } diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_blit_kernel.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_blit_kernel.cpp index 3286f9107e..c92ac2f710 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_blit_kernel.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_blit_kernel.cpp @@ -890,7 +890,7 @@ void BlitKernel::PopulateQueue(uint64_t index, uint64_t code_handle, void* args, std::atomic_thread_fence(std::memory_order_acquire); queue_buffer[index & queue_bitmask_] = packet; std::atomic_thread_fence(std::memory_order_release); - if (core::Runtime::runtime_singleton_->flag().dev_mem_queue() && !queue_->needsPcieOrdering()) { + if (queue_->IsDeviceMemRingBuf() && queue_->needsPcieOrdering()) { // Ensure the packet body is written as header may get reordered when writing over PCIE _mm_sfence(); } diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_cpu_agent.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_cpu_agent.cpp index 8486abe24e..369aace289 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_cpu_agent.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_cpu_agent.cpp @@ -424,10 +424,9 @@ hsa_status_t CpuAgent::GetInfo(hsa_agent_info_t attribute, void* value) const { return HSA_STATUS_SUCCESS; } -hsa_status_t CpuAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type, - core::HsaEventCallback event_callback, - void* data, uint32_t private_segment_size, - uint32_t group_segment_size, +hsa_status_t CpuAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type, uint64_t flags, + core::HsaEventCallback event_callback, void* data, + uint32_t private_segment_size, uint32_t group_segment_size, core::Queue** queue) { // No HW AQL packet processor on CPU device. return HSA_STATUS_ERROR; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_gpu_agent.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_gpu_agent.cpp index d1548ded38..63560a14a4 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_gpu_agent.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_gpu_agent.cpp @@ -706,7 +706,8 @@ core::Queue* GpuAgent::CreateInterceptibleQueue(void (*callback)(hsa_status_t st uint32_t size = std::max(in_size, minAqlSize_); size = std::min(size, maxAqlSize_); - QueueCreate(size, HSA_QUEUE_TYPE_MULTI, callback, data, 0, 0, &queue); + QueueCreate(size, HSA_QUEUE_TYPE_MULTI, HSA_AMD_QUEUE_CREATE_SYSTEM_MEM, callback, data, 0, 0, + &queue); if (queue != nullptr) core::Runtime::runtime_singleton_->InternalQueueCreateNotify(core::Queue::Convert(queue), this->public_handle()); @@ -1666,10 +1667,9 @@ hsa_status_t GpuAgent::GetInfo(hsa_agent_info_t attribute, void* value) const { return HSA_STATUS_SUCCESS; } -hsa_status_t GpuAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type, - core::HsaEventCallback event_callback, - void* data, uint32_t private_segment_size, - uint32_t group_segment_size, +hsa_status_t GpuAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type, uint64_t flags, + core::HsaEventCallback event_callback, void* data, + uint32_t private_segment_size, uint32_t group_segment_size, core::Queue** queue) { // Handle GWS queues. if (queue_type == HSA_QUEUE_TYPE_COOPERATIVE) { @@ -1739,9 +1739,26 @@ hsa_status_t GpuAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type, // ensured. queues_[QueueUtility].touch(); + bool dev_mem_queue_descriptor = (flags & HSA_AMD_QUEUE_CREATE_DEVICE_MEM_QUEUE_DESCRIPTOR) != 0; + // Create an HW AQL queue - auto aql_queue = - new AqlQueue(this, size, node_id(), scratch, event_callback, data, is_kv_device_); + core::SharedQueue* shared_queue = nullptr; + + if (dev_mem_queue_descriptor) { + shared_queue = static_cast( + finegrain_allocator()(sizeof(core::SharedQueue), core::MemoryRegion::AllocateUncached)); + } else { + shared_queue = + static_cast(core::Runtime::runtime_singleton_->system_allocator()( + sizeof(core::SharedQueue), MemoryRegion::GetPageSize(), + isMES() ? (MemoryRegion::AllocateGTTAccess | MemoryRegion::AllocateNonPaged) : 0, + node_id())); + } + + if (!shared_queue) return HSA_STATUS_ERROR_OUT_OF_RESOURCES; + + auto aql_queue = new AqlQueue(shared_queue, this, size, node_id(), scratch, event_callback, data, + flags, is_kv_device_); *queue = aql_queue; aql_queues_.push_back(aql_queue); diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_memory_region.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_memory_region.cpp index 485239b1dd..a19670f153 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_memory_region.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_memory_region.cpp @@ -56,6 +56,7 @@ namespace AMD { // Tracks aggregate size of system memory available on platform size_t MemoryRegion::max_sysmem_alloc_size_ = 0; +const size_t MemoryRegion::kPageSize_ = sysconf(_SC_PAGESIZE); bool MemoryRegion::RegisterMemory(void* ptr, size_t size, const HsaMemFlags& MemFlags) { assert(ptr != NULL); @@ -123,7 +124,7 @@ MemoryRegion::MemoryRegion(bool fine_grain, bool kernarg, bool full_profile, virtual_size_ = kGpuVmSize; } else if (IsSystem()) { - mem_flag_.ui32.PageSize = MemoryRegion::kPageSize(); + mem_flag_.ui32.PageSize = GetPageSize(); mem_flag_.ui32.NoSubstitute = 0; mem_flag_.ui32.HostAccess = 1; mem_flag_.ui32.CachePolicy = HSA_CACHING_CACHED; @@ -136,7 +137,7 @@ MemoryRegion::MemoryRegion(bool fine_grain, bool kernarg, bool full_profile, // Adjust allocatable size per page align - max_single_alloc_size_ = AlignDown(static_cast(GetPhysicalSize()), kPageSize()); + max_single_alloc_size_ = AlignDown(static_cast(GetPhysicalSize()), GetPageSize()); // Keep track of total system memory available // @note: System memory is surfaced as both coarse @@ -148,7 +149,7 @@ MemoryRegion::MemoryRegion(bool fine_grain, bool kernarg, bool full_profile, } assert(GetVirtualSize() != 0); - assert(IsMultipleOf(max_single_alloc_size_, kPageSize())); + assert(IsMultipleOf(max_single_alloc_size_, GetPageSize())); } MemoryRegion::~MemoryRegion() {} @@ -175,7 +176,7 @@ hsa_status_t MemoryRegion::AllocateImpl(size_t& size, AllocateFlags alloc_flags, return HSA_STATUS_ERROR_INVALID_ALLOCATION; } - size = AlignUp(size, kPageSize()); + size = AlignUp(size, GetPageSize()); return owner()->driver().AllocateMemory(*this, alloc_flags, address, size, agent_node_id); @@ -276,7 +277,7 @@ hsa_status_t MemoryRegion::GetInfo(hsa_region_info_t attribute, case HSA_HEAPTYPE_DEVICE_SVM: case HSA_HEAPTYPE_FRAME_BUFFER_PRIVATE: case HSA_HEAPTYPE_FRAME_BUFFER_PUBLIC: - *((size_t*)value) = kPageSize(); + *((size_t*)value) = GetPageSize(); break; default: *((size_t*)value) = 0; @@ -289,7 +290,7 @@ hsa_status_t MemoryRegion::GetInfo(hsa_region_info_t attribute, case HSA_HEAPTYPE_DEVICE_SVM: case HSA_HEAPTYPE_FRAME_BUFFER_PRIVATE: case HSA_HEAPTYPE_FRAME_BUFFER_PUBLIC: - *((size_t*)value) = kPageSize(); + *((size_t*)value) = GetPageSize(); break; default: *((size_t*)value) = 0; @@ -358,12 +359,12 @@ hsa_status_t MemoryRegion::GetPoolInfo(hsa_amd_memory_pool_info_t attribute, case HSA_AMD_MEMORY_POOL_INFO_RUNTIME_ALLOC_REC_GRANULE: switch (mem_props_.HeapType) { case HSA_HEAPTYPE_SYSTEM: - *((size_t*)value) = kPageSize(); + *((size_t*)value) = GetPageSize(); break; case HSA_HEAPTYPE_FRAME_BUFFER_PRIVATE: case HSA_HEAPTYPE_FRAME_BUFFER_PUBLIC: *((size_t*)value) = core::Runtime::runtime_singleton_->flag().disable_fragment_alloc() - ? kPageSize() + ? GetPageSize() : fragment_allocator_.default_block_size(); break; default: diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/host_queue.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/host_queue.cpp index 43de8baca3..dcd52224d1 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/host_queue.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/host_queue.cpp @@ -48,9 +48,9 @@ namespace rocr { namespace core { -HostQueue::HostQueue(hsa_region_t region, uint32_t ring_size, hsa_queue_type32_t type, - uint32_t features, hsa_signal_t doorbell_signal) - : Queue(), size_(ring_size) { +HostQueue::HostQueue(core::SharedQueue* shared_queue, hsa_region_t region, uint32_t ring_size, + hsa_queue_type32_t type, uint32_t features, hsa_signal_t doorbell_signal) + : Queue(shared_queue, 0), size_(ring_size) { HSA::hsa_memory_register(this, sizeof(HostQueue)); MAKE_NAMED_SCOPE_GUARD(registerGuard, [&]() { HSA::hsa_memory_deregister(this, sizeof(HostQueue)); }); @@ -93,6 +93,7 @@ HostQueue::HostQueue(hsa_region_t region, uint32_t ring_size, hsa_queue_type32_t } HostQueue::~HostQueue() { + HSA::hsa_memory_free(shared_queue_); HSA::hsa_memory_free(ring_); HSA::hsa_memory_deregister(this, sizeof(HostQueue)); } diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/hsa.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/hsa.cpp index e5be19b3b4..9152f070db 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/hsa.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/hsa.cpp @@ -742,8 +742,13 @@ hsa_status_t hsa_queue_create( if (callback == nullptr) callback = core::Queue::DefaultErrorHandler; + uint64_t queue_create_flags = 0; + + if (core::Runtime::runtime_singleton_->flag().dev_mem_queue_buf()) + queue_create_flags = HSA_AMD_QUEUE_CREATE_DEVICE_MEM_RING_BUF; + core::Queue* cmd_queue = nullptr; - status = agent->QueueCreate(size, type, callback, data, private_segment_size, + status = agent->QueueCreate(size, type, queue_create_flags, callback, data, private_segment_size, group_segment_size, &cmd_queue); if (status != HSA_STATUS_SUCCESS) return status; @@ -774,7 +779,13 @@ hsa_status_t hsa_soft_queue_create(hsa_region_t region, uint32_t size, const core::Signal* signal = core::Signal::Convert(doorbell_signal); IS_VALID(signal); - core::HostQueue* host_queue = new core::HostQueue(region, size, type, features, doorbell_signal); + void* shared_queue = nullptr; + hsa_status_t err = HSA::hsa_memory_allocate(region, sizeof(core::SharedQueue), &shared_queue); + if (err != HSA_STATUS_SUCCESS) return err; + assert(shared_queue && "Queue struct is NULL when creating host queue."); + + core::HostQueue* host_queue = new core::HostQueue(static_cast(shared_queue), + region, size, type, features, doorbell_signal); *queue = core::Queue::Convert(host_queue); diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/intercept_queue.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/intercept_queue.cpp index a86dabb3c2..258aed5d6f 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/intercept_queue.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/intercept_queue.cpp @@ -256,7 +256,7 @@ uint64_t InterceptQueue::Submit(const AqlPacket* packets, uint64_t count) { // Submit barrier which will wake async queue processing. ring[barrier & mask].packet.body = {}; ring[barrier & mask].barrier_and.completion_signal = Signal::Convert(async_doorbell_); - if (Runtime::runtime_singleton_->flag().dev_mem_queue() && !needsPcieOrdering()) { + if (wrapped->IsDeviceMemRingBuf() && needsPcieOrdering()) { // Ensure the packet body is written as header may get reordered when writing over PCIE _mm_sfence(); } @@ -303,7 +303,7 @@ uint64_t InterceptQueue::Submit(const AqlPacket* packets, uint64_t count) { ++packets_index; } if (write_index != 0) { - if (Runtime::runtime_singleton_->flag().dev_mem_queue() && !needsPcieOrdering()) { + if (wrapped->IsDeviceMemRingBuf() && needsPcieOrdering()) { // Ensure the packet body is written as header may get reordered when writing over PCIE _mm_sfence(); } @@ -372,7 +372,7 @@ void InterceptQueue::StoreRelaxed(hsa_signal_value_t value) { Cursor.pkt_index = i; auto& handler = interceptors[Cursor.interceptor_index]; handler.first(&ring[i & mask], 1, i, handler.second, PacketWriter); - if (Runtime::runtime_singleton_->flag().dev_mem_queue() && !needsPcieOrdering()) { + if (IsDeviceMemRingBuf() && needsPcieOrdering()) { // Ensure the packet body is written as header may get reordered when writing over PCIE _mm_sfence(); } diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/util/flag.h b/projects/rocr-runtime/runtime/hsa-runtime/core/util/flag.h index e4ed38c615..3275b677e8 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/util/flag.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/util/flag.h @@ -257,7 +257,7 @@ class Flag { override_cpu_affinity_ = (var == "0") ? false : true; var = os::GetEnvVar("HSA_ALLOCATE_QUEUE_DEV_MEM"); - dev_mem_queue_ = (var == "1") ? true : false; + dev_mem_queue_buf_ = (var == "1") ? true : false; var = os::GetEnvVar("HSA_WAIT_ANY_DEBUG"); wait_any_ = (var == "1") ? true : false; @@ -396,7 +396,7 @@ class Flag { size_t pc_sampling_max_device_buffer_size() const { return pc_sampling_max_device_buffer_size_; } - bool dev_mem_queue() const { return dev_mem_queue_; } + bool dev_mem_queue_buf() const { return dev_mem_queue_buf_; } uint32_t signal_abort_timeout() const { return signal_abort_timeout_; } @@ -433,7 +433,7 @@ class Flag { bool enable_mwaitx_; bool enable_ipc_mode_legacy_; bool wait_any_; - bool dev_mem_queue_; + bool dev_mem_queue_buf_; uint32_t signal_abort_timeout_; int async_events_thread_priority_; bool enable_3d_swizzle_ = false; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/inc/hsa_ext_amd.h b/projects/rocr-runtime/runtime/hsa-runtime/inc/hsa_ext_amd.h index 9eb5a945f3..5dbbbaf788 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/inc/hsa_ext_amd.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/inc/hsa_ext_amd.h @@ -2805,6 +2805,32 @@ typedef enum hsa_amd_queue_priority_s { hsa_status_t HSA_API hsa_amd_queue_set_priority(hsa_queue_t* queue, hsa_amd_queue_priority_t priority); +/** + * @brief Queue creation attributes. + */ +typedef enum { + /** + * The queue's packet buffer and queue descriptor struct should be + * allocated in system memory (default). Mutually exclusive with + * HSA_AMD_QUEUE_CREATE_DEVICE_MEM_RING_BUF and + * HSA_AMD_QUEUE_CREATE_DEVICE_MEM_QUEUE_DESCRIPTOR. + */ + HSA_AMD_QUEUE_CREATE_SYSTEM_MEM = 0, + /** + * The queue's packet buffer should be allocated in the agent's + * fine-grain device memory region. + */ + HSA_AMD_QUEUE_CREATE_DEVICE_MEM_RING_BUF = (1 << 0), + /** + * The queue desciptor struct should be allocated in the agent's + * fine-grain device memory region. Not supported for devices + * connected via PCIe because the CPU's atomic read-modify-write + * operations cannot be promoted to PCIe atomic read-modify-write + * operations. + */ + HSA_AMD_QUEUE_CREATE_DEVICE_MEM_QUEUE_DESCRIPTOR = (1 << 1), +} hsa_amd_queue_create_flag_t; + /** @} */ /** \addtogroup memory Memory