clr: Implement per-stream SDMA engine affinity for improved copy performance (#2480)

Problem:
The existing SDMA engine selection logic had several issues:
1. Same VirtualGPU/stream could use different SDMA engines for consecutive
   async copies since copy_engine_status may report engines as busy
2. Busy and Preferred engine check for every copy
3. No global tracking of which VirtualGPU uses which engine, leading to
   suboptimal resource allocation

Solution:
Implemented a global SDMA engine allocator with per-stream affinity:

- Added Device::SdmaEngineAllocator to manage VirtualGPU → engine assignments
  * Maintains global map of active assignments
  * Enforces exclusivity: different streams use different engines (except
    inter-GPU copies where preferred engines are prioritized for optimal
    hardware paths like XGMI links)
  * Thread-safe allocation/release with Monitor lock

- Modified VirtualGPU to cache assigned engine locally (assigned_sdma_engine_)
  for fast lookup without map access on hot path

- Refactored rocrCopyBuffer() to:
  1. Check local cached engine first → use if assigned
  2. Call AllocateSdmaEngine() if not assigned → cache result

- Moved HSA API queries (memory_copy_engine_status, memory_get_preferred_copy_engine)
  into AllocateEngine() for cleaner separation of concerns

- Engine release on HostQueue::finish() instead of only VirtualGPU destruction
  * Improves engine utilization by releasing earlier
  * Added virtual ReleaseSdmaEngines() method to device::VirtualDevice

- Added future path for simple round-robin allocation (kUseSimpleRR) for
  next-gen GPUs with uniform SDMA bandwidth (disabled by default)

Cleanup:
- Removed selectSdmaEngine() helper (logic moved to allocator)
- Removed getSdmaRWMasks() (allocator accesses maxSdmaReadMask_/WriteMask_ directly)
- Removed unused sdmaEngineReadMask_/WriteMask_ member variables from DmaBlitManager

Benefits:
- Ensures consistent per-stream SDMA engine usage
- Prevents cross-stream contention and engine thrashing
- Prioritizes hardware-optimal paths for inter-GPU transfers
- Better resource utilization through earlier release
- Cleaner, more maintainable code structure
This commit is contained in:
SaleelK
2026-01-07 19:37:45 -08:00
committed by GitHub
parent be04fa8250
commit 6b28faa532
8 changed files with 223 additions and 50 deletions
+138 -3
View File
@@ -146,6 +146,7 @@ Device::Device(hsa_agent_t bkendDevice)
preferred_numa_node_(0),
maxSdmaReadMask_(0),
maxSdmaWriteMask_(0),
sdma_engine_allocator_(*this),
cpu_agent_info_(nullptr) {
group_segment_.handle = 0;
gpuvm_segment_.handle = 0;
@@ -3509,9 +3510,143 @@ void Device::HiddenHeapInit(const VirtualGPU& gpu) {
}
// ================================================================================================
void Device::getSdmaRWMasks(uint32_t* readMask, uint32_t* writeMask) const {
*readMask = maxSdmaReadMask_;
*writeMask = maxSdmaWriteMask_;
uint32_t Device::SdmaEngineAllocator::AllocateEngine(VirtualGPU* vgpu, HwQueueEngine engine_type,
hsa_agent_t dstAgent, hsa_agent_t srcAgent) {
amd::ScopedLock lock(lock_);
// Get valid engine mask based on operation type (read vs write)
uint32_t validEngineMask = (engine_type == HwQueueEngine::SdmaRead)
? device_.maxSdmaReadMask_
: device_.maxSdmaWriteMask_;
// Simple round-robin path if all engines have equal bandwidth
// Disabled by default - use preferred engine logic for current GPUs
constexpr bool kUseSimpleRR = false;
if (kUseSimpleRR) {
// Simple round-robin: just cycle through valid engines
// This will be enabled for future GPUs where engine selection doesn't matter
if (validEngineMask == 0) {
ClPrint(amd::LOG_WARNING, amd::LOG_COPY,
"No valid SDMA engines for VirtualGPU %p", vgpu);
return 0;
}
// Cycle through bit positions, find next valid engine
uint32_t start_bit = next_rr_engine_.fetch_add(1, std::memory_order_relaxed);
uint32_t selected_mask = 0;
// Try up to 32 positions to find a valid engine
for (uint32_t i = 0; i < 32; ++i) {
uint32_t bit = (start_bit + i) % 32;
uint32_t mask = 1u << bit;
if (validEngineMask & mask) {
selected_mask = mask;
break;
}
}
vgpu_to_engine_[vgpu] = selected_mask;
ClPrint(amd::LOG_INFO, amd::LOG_COPY,
"Assigned SDMA engine (simple RR) to VirtualGPU %p: mask=0x%x, engine_type=%d",
vgpu, selected_mask, engine_type);
return selected_mask;
}
// Current path: Query HSA for engine status and preferences
uint32_t freeEngineMask = 0;
uint32_t preferredMask = 0;
hsa_status_t status = HSA_STATUS_SUCCESS;
// Query current engine status
status = Hsa::memory_copy_engine_status(dstAgent, srcAgent, &freeEngineMask);
if (status == HSA_STATUS_SUCCESS) {
// Query preferred (high-bandwidth) engines
status = Hsa::memory_get_preferred_copy_engine(dstAgent, srcAgent, &preferredMask);
}
// Constrain to valid engines
freeEngineMask &= validEngineMask;
preferredMask &= validEngineMask;
ClPrint(amd::LOG_DEBUG, amd::LOG_COPY,
"Engine query for VirtualGPU %p: status=%x, free_mask=0x%x, preferred_mask=0x%x, "
"valid_mask=0x%x, engine_type=%d",
vgpu, status, freeEngineMask, preferredMask, validEngineMask, engine_type);
uint32_t candidate_mask = 0;
uint32_t allocated_mask = 0;
// For inter-GPU copies, strongly prefer the recommended engines
bool is_inter_gpu = (engine_type == HwQueueEngine::SdmaInter);
if (is_inter_gpu && (preferredMask != 0)) {
// Inter-GPU: prioritize preferredMask, even if engines are already allocated
candidate_mask = validEngineMask & preferredMask;
ClPrint(amd::LOG_DEBUG, amd::LOG_COPY,
"Inter-GPU copy for VirtualGPU %p: prioritizing preferred engines, "
"candidate_mask=0x%x",
vgpu, candidate_mask);
} else {
// Regular read/write/intra: enforce exclusivity (don't share engines)
// Build a mask of engines already allocated to other VirtualGPUs
for (const auto& pair : vgpu_to_engine_) {
allocated_mask |= pair.second;
}
uint32_t available_mask = validEngineMask & ~allocated_mask;
if (available_mask == 0) {
ClPrint(amd::LOG_WARNING, amd::LOG_COPY,
"No unallocated SDMA engines available for VirtualGPU %p, engine_type=%d "
"(valid_mask=0x%x, allocated_mask=0x%x)",
vgpu, engine_type, validEngineMask, allocated_mask);
return 0;
}
// Prefer high-bandwidth (recommended) engines if available
candidate_mask = available_mask & preferredMask;
if (candidate_mask == 0) {
candidate_mask = available_mask;
}
}
if (candidate_mask == 0) {
ClPrint(amd::LOG_WARNING, amd::LOG_COPY,
"No candidate SDMA engines for VirtualGPU %p, engine_type=%d",
vgpu, engine_type);
return 0;
}
// Select the lowest bit (first available engine)
uint32_t selected_mask = candidate_mask & (~candidate_mask + 1);
// Update the map
vgpu_to_engine_[vgpu] = selected_mask;
ClPrint(amd::LOG_INFO, amd::LOG_COPY,
"Assigned SDMA engine to VirtualGPU %p: mask=0x%x, engine_type=%d, "
"valid_mask=0x%x, preferred_mask=0x%x, allocated_mask=0x%x, is_inter_gpu=%d",
vgpu, selected_mask, engine_type, validEngineMask, preferredMask,
allocated_mask, is_inter_gpu);
return selected_mask;
}
// ================================================================================================
void Device::SdmaEngineAllocator::ReleaseEngine(VirtualGPU* vgpu) {
amd::ScopedLock lock(lock_);
auto it = vgpu_to_engine_.find(vgpu);
if (it != vgpu_to_engine_.end()) {
ClPrint(amd::LOG_DEBUG, amd::LOG_COPY,
"Released SDMA engine for VirtualGPU %p: mask=0x%x",
vgpu, it->second);
vgpu_to_engine_.erase(it);
}
}
// ================================================================================================