SWDEV-363074 - Clean-up sync between SDMA and compute

HIP can't rely on the resource tracking, used in OCL and requires different explicit sync.
Make sure ROCCLR syncs compute only when SDMA is used and vise versa.
The new logic will allow to enable CPDMA without unnecessary waits.

Change-Id: Ib9d1788cfd5afa5ea2fec4c96a37d8b9c4d0059d
This commit is contained in:
German
2022-10-27 18:22:38 -04:00
committed by German Andryeyev
parent 3bce4df27d
commit ff6b4db70b
4 changed files with 42 additions and 47 deletions
+10 -12
View File
@@ -61,7 +61,8 @@ uint32_t VirtualGPU::Queue::AllocedQueues(const VirtualGPU& gpu, Pal::EngineType
return allocedQueues;
}
VirtualGPU::Queue* VirtualGPU::Queue::Create(const VirtualGPU& gpu, Pal::QueueType queueType,
// ================================================================================================
VirtualGPU::Queue* VirtualGPU::Queue::Create(VirtualGPU& gpu, Pal::QueueType queueType,
uint engineIdx, Pal::ICmdAllocator* cmdAllocator,
uint rtCU, amd::CommandQueue::Priority priority,
uint64_t residency_limit, uint max_command_buffers) {
@@ -341,6 +342,7 @@ void VirtualGPU::Queue::addCmdDoppRef(Pal::IGpuMemory* iMem, bool lastDoppCmd, b
palDoppRefs_.push_back(doppRef);
}
// ================================================================================================
bool VirtualGPU::Queue::flush() {
if (!gpu_.dev().settings().alwaysResident_ && palMemRefs_.size() != 0) {
if (Pal::Result::Success !=
@@ -381,6 +383,13 @@ bool VirtualGPU::Queue::flush() {
submitInfo.fenceCount = 1;
submitInfo.ppFences = &iCmdFences_[cmdBufIdSlot_];
if (amd::IS_HIP) {
// HIP disables per resource tracking, because the app may embed SVM ptr into other buffers.
// Force CPU sync if there are pending operations on SDMA, until OS fences will be added
if (iQueue_->Type() == Pal::QueueTypeCompute) {
gpu_.WaitForIdleSdma();
}
}
// Submit command buffer to OS
Pal::Result result;
if (gpu_.rgpCaptureEna()) {
@@ -2695,9 +2704,6 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
dev().rgpCaptureMgr()->PostDispatch(this);
}
// Mark the flag indicating if a dispatch is outstanding.
state_.hasPendingDispatch_ = true;
return true;
}
@@ -3953,12 +3959,4 @@ void* VirtualGPU::getOrCreateHostcallBuffer() {
return hostcallBuffer_;
}
void VirtualGPU::releaseGpuMemoryFence() {
if (isPendingDispatch() && amd::IS_HIP) {
WaitForIdleCompute();
// Reset the status.
state_.hasPendingDispatch_ = false;
}
}
} // namespace pal