clr: Fix memory corruption for memset nodes (#1068)

* Detect graph capture and use graph kernelarg memory for FillBuffer pattern
This commit is contained in:
SaleelK
2025-09-23 17:17:33 -07:00
committed by GitHub
orang tua aece11079c
melakukan 34b9184686
3 mengubah file dengan 21 tambahan dan 16 penghapusan
+15 -9
Melihat File
@@ -338,10 +338,10 @@ bool DmaBlitManager::copyBufferRect(device::Memory& srcMemory, device::Memory& d
ClPrint(amd::LOG_DEBUG, amd::LOG_COPY2,
"HSA Async Copy wait_event=0x%zx, completion_signal=0x%zx",
(wait_events.size() != 0) ? wait_events[0].handle : 0, active.handle);
hsa_status_t status = Hsa::memory_async_copy(
(reinterpret_cast<address>(dst) + dstOffset), dstAgent,
(reinterpret_cast<const_address>(src) + srcOffset), srcAgent, size[0],
wait_events.size(), wait_events.data(), active);
hsa_status_t status =
Hsa::memory_async_copy((reinterpret_cast<address>(dst) + dstOffset), dstAgent,
(reinterpret_cast<const_address>(src) + srcOffset), srcAgent,
size[0], wait_events.size(), wait_events.data(), active);
if (status != HSA_STATUS_SUCCESS) {
gpu().Barriers().ResetCurrentSignal();
LogPrintfError("DMA buffer failed with code %d", status);
@@ -552,9 +552,9 @@ inline bool DmaBlitManager::rocrCopyBuffer(address dst, hsa_agent_t& dstAgent, c
copyEngine, dst, src, size, forceSDMA, engine,
(wait_events.size() != 0) ? wait_events[0].handle : 0, active.handle);
status = Hsa::memory_async_copy_on_engine(dst, dstAgent, src, srcAgent, size,
wait_events.size(), wait_events.data(), active,
copyEngine, forceSDMA);
status =
Hsa::memory_async_copy_on_engine(dst, dstAgent, src, srcAgent, size, wait_events.size(),
wait_events.data(), active, copyEngine, forceSDMA);
} else {
kUseRegularCopyApi = true;
}
@@ -2044,7 +2044,10 @@ bool KernelBlitManager::fillBuffer1D(device::Memory& memory, const void* pattern
size_t globalWorkSize = std::min(dev().settings().limit_blit_wg_ * localWorkSize, kfill_size);
globalWorkSize = amd::alignUp(globalWorkSize, localWorkSize);
auto constBuf = gpu().allocKernArg(kCBSize, kCBAlignment);
bool isGraphPktCapturing =
gpu().command() != nullptr && gpu().command()->getPktCapturingState();
auto constBuf = isGraphPktCapturing ? gpu().command()->getGraphKernArg(kCBSize, kCBAlignment)
: gpu().allocKernArg(kCBSize, kCBAlignment);
// If pattern has been expanded, use the expanded pattern, otherwise use the default pattern.
if (packed_obj.pattern_expanded_) {
@@ -2136,7 +2139,10 @@ bool KernelBlitManager::fillBuffer2D(device::Memory& memory, const void* pattern
}
// Get constant buffer to allow multipel fills
auto constBuf = gpu().allocKernArg(kCBSize, kCBAlignment);
bool isGraphPktCapturing =
gpu().command() != nullptr && gpu().command()->getPktCapturingState();
auto constBuf = isGraphPktCapturing ? gpu().command()->getGraphKernArg(kCBSize, kCBAlignment)
: gpu().allocKernArg(kCBSize, kCBAlignment);
memcpy(constBuf, pattern, patternSize);
constexpr bool kDirectVa = true;
@@ -1056,14 +1056,13 @@ bool VirtualGPU::dispatchGenericAqlPacket(AqlPacket* packet, uint16_t header, ui
// Make sure the slot is free for usage
while ((index - Hsa::queue_load_read_index_scacquire(gpu_queue_)) >= sw_queue_size) {
// Active spin - no yield
amd::Os::yield();
}
// Add blocking command if the original value of read index was behind of the queue size.
// Note: direct dispatch relies on the slot stall above to keep the forward progress
// of the app if a dispatched kernel requires some CPU input for completion
if (blocking || (!AMD_DIRECT_DISPATCH &&
(index - hsa_queue_load_read_index_relaxed(gpu_queue_)) >= sw_queue_size)) {
if (blocking) {
if (packet->completion_signal.handle == 0) {
packet->completion_signal = Barriers().ActiveSignal();
}
@@ -1202,7 +1201,7 @@ bool VirtualGPU::dispatchGenericAqlPacketBatch(const std::vector<AqlPacket*>& pa
// Make sure the slot is free for usage
while ((startIndex - Hsa::queue_load_read_index_scacquire(gpu_queue_)) >= sw_queue_size) {
// Active spin - no yield
amd::Os::yield();
}
fence_dirty_ = true;
@@ -3691,8 +3690,8 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
if (!kernel.parameters().deviceKernelArgs() || gpuKernel.isInternalKernel()) {
// Allocate buffer to hold kernel arguments
if (isGraphCapture) {
argBuffer = command_->getKernArgOffset(gpuKernel.KernargSegmentByteSize(),
gpuKernel.KernargSegmentAlignment());
argBuffer = command_->getGraphKernArg(gpuKernel.KernargSegmentByteSize(),
gpuKernel.KernargSegmentAlignment());
command_->SetKernelName(gpuKernel.getDemangledName().c_str());
} else {
ClPrint(amd::LOG_DETAIL_DEBUG, amd::LOG_KERN,
+1 -1
Melihat File
@@ -341,7 +341,7 @@ class Command : public Event {
return packet;
}
address getKernArgOffset(int size, int alignment) {
address getGraphKernArg(int size, int alignment) {
return graphKernArgMgr_->AllocKernArg(size, alignment);
}