diff --git a/projects/clr/rocclr/device/rocm/rocblit.cpp b/projects/clr/rocclr/device/rocm/rocblit.cpp
index a51a9163f1..8cefbe93c0 100644
--- a/projects/clr/rocclr/device/rocm/rocblit.cpp
+++ b/projects/clr/rocclr/device/rocm/rocblit.cpp
@@ -338,10 +338,10 @@ bool DmaBlitManager::copyBufferRect(device::Memory& srcMemory, device::Memory& d
ClPrint(amd::LOG_DEBUG, amd::LOG_COPY2,
"HSA Async Copy wait_event=0x%zx, completion_signal=0x%zx",
(wait_events.size() != 0) ? wait_events[0].handle : 0, active.handle);
- hsa_status_t status = Hsa::memory_async_copy(
- (reinterpret_cast
(dst) + dstOffset), dstAgent,
- (reinterpret_cast(src) + srcOffset), srcAgent, size[0],
- wait_events.size(), wait_events.data(), active);
+ hsa_status_t status =
+ Hsa::memory_async_copy((reinterpret_cast(dst) + dstOffset), dstAgent,
+ (reinterpret_cast(src) + srcOffset), srcAgent,
+ size[0], wait_events.size(), wait_events.data(), active);
if (status != HSA_STATUS_SUCCESS) {
gpu().Barriers().ResetCurrentSignal();
LogPrintfError("DMA buffer failed with code %d", status);
@@ -552,9 +552,9 @@ inline bool DmaBlitManager::rocrCopyBuffer(address dst, hsa_agent_t& dstAgent, c
copyEngine, dst, src, size, forceSDMA, engine,
(wait_events.size() != 0) ? wait_events[0].handle : 0, active.handle);
- status = Hsa::memory_async_copy_on_engine(dst, dstAgent, src, srcAgent, size,
- wait_events.size(), wait_events.data(), active,
- copyEngine, forceSDMA);
+ status =
+ Hsa::memory_async_copy_on_engine(dst, dstAgent, src, srcAgent, size, wait_events.size(),
+ wait_events.data(), active, copyEngine, forceSDMA);
} else {
kUseRegularCopyApi = true;
}
@@ -2044,7 +2044,10 @@ bool KernelBlitManager::fillBuffer1D(device::Memory& memory, const void* pattern
size_t globalWorkSize = std::min(dev().settings().limit_blit_wg_ * localWorkSize, kfill_size);
globalWorkSize = amd::alignUp(globalWorkSize, localWorkSize);
- auto constBuf = gpu().allocKernArg(kCBSize, kCBAlignment);
+ bool isGraphPktCapturing =
+ gpu().command() != nullptr && gpu().command()->getPktCapturingState();
+ auto constBuf = isGraphPktCapturing ? gpu().command()->getGraphKernArg(kCBSize, kCBAlignment)
+ : gpu().allocKernArg(kCBSize, kCBAlignment);
// If pattern has been expanded, use the expanded pattern, otherwise use the default pattern.
if (packed_obj.pattern_expanded_) {
@@ -2136,7 +2139,10 @@ bool KernelBlitManager::fillBuffer2D(device::Memory& memory, const void* pattern
}
// Get constant buffer to allow multipel fills
- auto constBuf = gpu().allocKernArg(kCBSize, kCBAlignment);
+ bool isGraphPktCapturing =
+ gpu().command() != nullptr && gpu().command()->getPktCapturingState();
+ auto constBuf = isGraphPktCapturing ? gpu().command()->getGraphKernArg(kCBSize, kCBAlignment)
+ : gpu().allocKernArg(kCBSize, kCBAlignment);
memcpy(constBuf, pattern, patternSize);
constexpr bool kDirectVa = true;
diff --git a/projects/clr/rocclr/device/rocm/rocvirtual.cpp b/projects/clr/rocclr/device/rocm/rocvirtual.cpp
index 339fec50c6..a633c18b94 100644
--- a/projects/clr/rocclr/device/rocm/rocvirtual.cpp
+++ b/projects/clr/rocclr/device/rocm/rocvirtual.cpp
@@ -1056,14 +1056,13 @@ bool VirtualGPU::dispatchGenericAqlPacket(AqlPacket* packet, uint16_t header, ui
// Make sure the slot is free for usage
while ((index - Hsa::queue_load_read_index_scacquire(gpu_queue_)) >= sw_queue_size) {
- // Active spin - no yield
+ amd::Os::yield();
}
// Add blocking command if the original value of read index was behind of the queue size.
// Note: direct dispatch relies on the slot stall above to keep the forward progress
// of the app if a dispatched kernel requires some CPU input for completion
- if (blocking || (!AMD_DIRECT_DISPATCH &&
- (index - hsa_queue_load_read_index_relaxed(gpu_queue_)) >= sw_queue_size)) {
+ if (blocking) {
if (packet->completion_signal.handle == 0) {
packet->completion_signal = Barriers().ActiveSignal();
}
@@ -1202,7 +1201,7 @@ bool VirtualGPU::dispatchGenericAqlPacketBatch(const std::vector& pa
// Make sure the slot is free for usage
while ((startIndex - Hsa::queue_load_read_index_scacquire(gpu_queue_)) >= sw_queue_size) {
- // Active spin - no yield
+ amd::Os::yield();
}
fence_dirty_ = true;
@@ -3691,8 +3690,8 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
if (!kernel.parameters().deviceKernelArgs() || gpuKernel.isInternalKernel()) {
// Allocate buffer to hold kernel arguments
if (isGraphCapture) {
- argBuffer = command_->getKernArgOffset(gpuKernel.KernargSegmentByteSize(),
- gpuKernel.KernargSegmentAlignment());
+ argBuffer = command_->getGraphKernArg(gpuKernel.KernargSegmentByteSize(),
+ gpuKernel.KernargSegmentAlignment());
command_->SetKernelName(gpuKernel.getDemangledName().c_str());
} else {
ClPrint(amd::LOG_DETAIL_DEBUG, amd::LOG_KERN,
diff --git a/projects/clr/rocclr/platform/command.hpp b/projects/clr/rocclr/platform/command.hpp
index 73ed76ea34..f36636eb48 100644
--- a/projects/clr/rocclr/platform/command.hpp
+++ b/projects/clr/rocclr/platform/command.hpp
@@ -341,7 +341,7 @@ class Command : public Event {
return packet;
}
- address getKernArgOffset(int size, int alignment) {
+ address getGraphKernArg(int size, int alignment) {
return graphKernArgMgr_->AllocKernArg(size, alignment);
}