rocr: Allocate AQL queue on device memory

- Use HSA_ALLOCATE_QUEUE_DEV_MEM=1 to create AQL queue in device
memory.
- Before writing AQL packet header to the queue use an SFENCE to ensure
that there is no reodering of the writes over PCIE

Change-Id: I5eacdc35108c4a1e245c75ae349b7495451aa60d
This commit is contained in:
Saleel Kudchadker
2024-08-20 04:44:46 +00:00
förälder fe8d8c15f1
incheckning 3baaa6e9c0
10 ändrade filer med 95 tillägg och 40 borttagningar
+16 -9
Visa fil
@@ -2,24 +2,24 @@
//
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
//
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
//
//
// Developed by:
//
//
// AMD Research and AMD HSA Software Development
//
//
// Advanced Micro Devices, Inc.
//
//
// www.amd.com
//
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to
// deal with the Software without restriction, including without limitation
// the rights to use, copy, modify, merge, publish, distribute, sublicense,
// and/or sell copies of the Software, and to permit persons to whom the
// Software is furnished to do so, subject to the following conditions:
//
//
// - Redistributions of source code must retain the above copyright notice,
// this list of conditions and the following disclaimers.
// - Redistributions in binary form must reproduce the above copyright
@@ -29,7 +29,7 @@
// nor the names of its contributors may be used to endorse or promote
// products derived from this Software without specific prior written
// permission.
//
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
@@ -348,7 +348,7 @@ class GpuAgent : public GpuAgentInt {
}
core::Agent* GetNearestCpuAgent(void) const;
void RegisterGangPeer(core::Agent& gang_peer, unsigned int bandwidth_factor) override;
void RegisterRecSdmaEngIdMaskPeer(core::Agent& gang_peer, uint32_t rec_sdma_eng_id_mask) override;
@@ -417,6 +417,9 @@ class GpuAgent : public GpuAgentInt {
if (t0_.GPUClockCounter == t1_.GPUClockCounter) SyncClocks();
}
// @brief Override from AMD::GpuAgentInt.
__forceinline bool is_xgmi_cpu_gpu() const { return xgmi_cpu_gpu_; }
const size_t MAX_SCRATCH_APERTURE_PER_XCC = (1ULL << 32);
size_t MaxScratchDevice() const { return properties_.NumXcc * MAX_SCRATCH_APERTURE_PER_XCC; }
@@ -624,6 +627,7 @@ class GpuAgent : public GpuAgentInt {
// @brief HDP flush registers
hsa_amd_hdp_flush_t HDP_flush_ = {nullptr, nullptr};
private:
// @brief Query the driver to get the region list owned by this agent.
void InitRegionList();
@@ -782,6 +786,9 @@ class GpuAgent : public GpuAgentInt {
std::map<uint64_t, uint32_t> rec_sdma_eng_id_peers_info_;
bool uses_rec_sdma_eng_id_mask_;
// @bried XGMI CPU<->GPU
bool xgmi_cpu_gpu_;
};
} // namespace amd
+11 -10
Visa fil
@@ -2,24 +2,24 @@
//
// The University of Illinois/NCSA
// Open Source License (NCSA)
//
// Copyright (c) 2014-2020, Advanced Micro Devices, Inc. All rights reserved.
//
//
// Copyright (c) 2014-2024, Advanced Micro Devices, Inc. All rights reserved.
//
// Developed by:
//
//
// AMD Research and AMD HSA Software Development
//
//
// Advanced Micro Devices, Inc.
//
//
// www.amd.com
//
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to
// deal with the Software without restriction, including without limitation
// the rights to use, copy, modify, merge, publish, distribute, sublicense,
// and/or sell copies of the Software, and to permit persons to whom the
// Software is furnished to do so, subject to the following conditions:
//
//
// - Redistributions of source code must retain the above copyright notice,
// this list of conditions and the following disclaimers.
// - Redistributions in binary form must reproduce the above copyright
@@ -29,7 +29,7 @@
// nor the names of its contributors may be used to endorse or promote
// products derived from this Software without specific prior written
// permission.
//
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
@@ -104,7 +104,8 @@ class MemoryRegion : public Checked<0x9C961F19EE175BB3> {
// Note: The node_id needs to be the node_id of the device even though this is allocating
// system memory
AllocateGTTAccess = (1 << 9),
AllocateContiguous = (1 << 10), // Physically contiguous memory
AllocateContiguous = (1 << 10), // Physically contiguous memory
AllocateUncached = (1 << 11), // Uncached memory
};
typedef uint32_t AllocateFlags;
+8
Visa fil
@@ -182,11 +182,13 @@ class Queue : public Checked<0xFA3906A679F9DB49>, private LocalQueue {
Queue(int mem_flags = 0) : LocalQueue(mem_flags), amd_queue_(queue()->amd_queue) {
queue()->core_queue = this;
public_handle_ = Convert(this);
pcie_write_ordering_ = false;
}
Queue(int agent_node_id, int mem_flags) : LocalQueue(agent_node_id, mem_flags), amd_queue_(queue()->amd_queue) {
queue()->core_queue = this;
public_handle_ = Convert(this);
pcie_write_ordering_ = false;
}
virtual ~Queue() {}
@@ -385,6 +387,10 @@ class Queue : public Checked<0xFA3906A679F9DB49>, private LocalQueue {
bool IsType(rtti_t id) { return _IsA(id); }
bool needsPcieOrdering() const { return pcie_write_ordering_; }
void setPcieOrdering(bool val) { pcie_write_ordering_ = val; }
protected:
static void set_public_handle(Queue* ptr, hsa_queue_t* handle) {
ptr->do_set_public_handle(handle);
@@ -405,6 +411,8 @@ class Queue : public Checked<0xFA3906A679F9DB49>, private LocalQueue {
// HSA Queue ID - used to bind a unique ID
static std::atomic<uint64_t> hsa_queue_counter_;
bool pcie_write_ordering_;
DISALLOW_COPY_AND_ASSIGN(Queue);
};
} // namespace core