Add HSA signal global tracking logic.
Implement the global class for signals tracking per device queue.
Switch to the new tracking mechanism.
Change-Id: I3c4dda04b34e6d18d6a95510d84102909633b415
[ROCm/clr commit: 8698aeef0d]
This commit is contained in:
@@ -62,12 +62,14 @@ bool DmaBlitManager::readMemoryStaged(Memory& srcMemory, void* dstHost, Memory&
|
||||
bool DmaBlitManager::readBuffer(device::Memory& srcMemory, void* dstHost,
|
||||
const amd::Coord3D& origin, const amd::Coord3D& size,
|
||||
bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
gpu().releaseGpuMemoryFence();
|
||||
// HSA copy functionality with a possible async operation
|
||||
gpu().releaseGpuMemoryFence(kIgnoreBarrier, kSkipCpuWait);
|
||||
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableReadBuffer_ ||
|
||||
(srcMemory.isHostMemDirectAccess() && !srcMemory.isCpuUncached())) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().Barriers().WaitCurrent();
|
||||
return HostBlitManager::readBuffer(srcMemory, dstHost, origin, size, entire);
|
||||
} else {
|
||||
size_t srcSize = size[0];
|
||||
@@ -149,12 +151,14 @@ bool DmaBlitManager::readBuffer(device::Memory& srcMemory, void* dstHost,
|
||||
bool DmaBlitManager::readBufferRect(device::Memory& srcMemory, void* dstHost,
|
||||
const amd::BufferRect& bufRect, const amd::BufferRect& hostRect,
|
||||
const amd::Coord3D& size, bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
// HSA copy functionality with a possible async operation
|
||||
gpu().releaseGpuMemoryFence();
|
||||
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableReadBufferRect_ ||
|
||||
(srcMemory.isHostMemDirectAccess() && !srcMemory.isCpuUncached())) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().Barriers().WaitCurrent();
|
||||
return HostBlitManager::readBufferRect(srcMemory, dstHost, bufRect, hostRect, size, entire);
|
||||
} else {
|
||||
Memory& xferBuf = dev().xferRead().acquire();
|
||||
@@ -187,7 +191,7 @@ bool DmaBlitManager::readBufferRect(device::Memory& srcMemory, void* dstHost,
|
||||
bool DmaBlitManager::readImage(device::Memory& srcMemory, void* dstHost, const amd::Coord3D& origin,
|
||||
const amd::Coord3D& size, size_t rowPitch, size_t slicePitch,
|
||||
bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
// HSA copy functionality with a possible async operation
|
||||
gpu().releaseGpuMemoryFence();
|
||||
|
||||
if (setup_.disableReadImage_) {
|
||||
@@ -219,14 +223,16 @@ bool DmaBlitManager::writeMemoryStaged(const void* srcHost, Memory& dstMemory, M
|
||||
bool DmaBlitManager::writeBuffer(const void* srcHost, device::Memory& dstMemory,
|
||||
const amd::Coord3D& origin, const amd::Coord3D& size,
|
||||
bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
gpu().releaseGpuMemoryFence();
|
||||
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableWriteBuffer_ || dstMemory.isHostMemDirectAccess() ||
|
||||
gpuMem(dstMemory).IsPersistentDirectMap()) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
return HostBlitManager::writeBuffer(srcHost, dstMemory, origin, size, entire);
|
||||
} else {
|
||||
// HSA copy functionality with a possible async operation
|
||||
gpu().releaseGpuMemoryFence(kIgnoreBarrier, kSkipCpuWait);
|
||||
|
||||
size_t dstSize = size[0];
|
||||
size_t tmpSize = 0;
|
||||
size_t offset = 0;
|
||||
@@ -309,7 +315,7 @@ bool DmaBlitManager::writeBufferRect(const void* srcHost, device::Memory& dstMem
|
||||
const amd::BufferRect& hostRect,
|
||||
const amd::BufferRect& bufRect, const amd::Coord3D& size,
|
||||
bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
// HSA copy functionality with a possible async operation
|
||||
gpu().releaseGpuMemoryFence();
|
||||
|
||||
// Use host copy if memory has direct access
|
||||
@@ -347,7 +353,7 @@ bool DmaBlitManager::writeBufferRect(const void* srcHost, device::Memory& dstMem
|
||||
bool DmaBlitManager::writeImage(const void* srcHost, device::Memory& dstMemory,
|
||||
const amd::Coord3D& origin, const amd::Coord3D& size,
|
||||
size_t rowPitch, size_t slicePitch, bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
// HSA copy functionality with a possible async operation
|
||||
gpu().releaseGpuMemoryFence();
|
||||
|
||||
if (setup_.disableWriteImage_) {
|
||||
@@ -365,12 +371,11 @@ bool DmaBlitManager::writeImage(const void* srcHost, device::Memory& dstMemory,
|
||||
bool DmaBlitManager::copyBuffer(device::Memory& srcMemory, device::Memory& dstMemory,
|
||||
const amd::Coord3D& srcOrigin, const amd::Coord3D& dstOrigin,
|
||||
const amd::Coord3D& size, bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
gpu().releaseGpuMemoryFence();
|
||||
|
||||
if (setup_.disableCopyBuffer_ ||
|
||||
(srcMemory.isHostMemDirectAccess() && !srcMemory.isCpuUncached() &&
|
||||
(dev().agent_profile() != HSA_PROFILE_FULL) && dstMemory.isHostMemDirectAccess())) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
return HostBlitManager::copyBuffer(srcMemory, dstMemory, srcOrigin, dstOrigin, size);
|
||||
} else {
|
||||
return hsaCopy(gpuMem(srcMemory), gpuMem(dstMemory), srcOrigin, dstOrigin, size);
|
||||
@@ -383,14 +388,14 @@ bool DmaBlitManager::copyBuffer(device::Memory& srcMemory, device::Memory& dstMe
|
||||
bool DmaBlitManager::copyBufferRect(device::Memory& srcMemory, device::Memory& dstMemory,
|
||||
const amd::BufferRect& srcRect, const amd::BufferRect& dstRect,
|
||||
const amd::Coord3D& size, bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
gpu().releaseGpuMemoryFence();
|
||||
|
||||
if (setup_.disableCopyBufferRect_ ||
|
||||
(srcMemory.isHostMemDirectAccess() && !srcMemory.isCpuUncached() &&
|
||||
dstMemory.isHostMemDirectAccess())) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
return HostBlitManager::copyBufferRect(srcMemory, dstMemory, srcRect, dstRect, size, entire);
|
||||
} else {
|
||||
gpu().releaseGpuMemoryFence(kIgnoreBarrier, kSkipCpuWait);
|
||||
|
||||
void* src = gpuMem(srcMemory).getDeviceMemory();
|
||||
void* dst = gpuMem(dstMemory).getDeviceMemory();
|
||||
@@ -436,25 +441,21 @@ bool DmaBlitManager::copyBufferRect(device::Memory& srcMemory, device::Memory& d
|
||||
}
|
||||
|
||||
if (isSubwindowRectCopy ) {
|
||||
hsa_signal_store_relaxed(completion_signal_, kInitSignalValueOne);
|
||||
hsa_signal_t wait = gpu().Barriers().WaitSignal();
|
||||
hsa_signal_t active = gpu().Barriers().ActiveSignal(kInitSignalValueOne, gpu().timestamp());
|
||||
|
||||
// Copy memory line by line
|
||||
hsa_status_t status =
|
||||
hsa_amd_memory_async_copy_rect(&dstMem, &offset, &srcMem, &offset, &dim, agent,
|
||||
direction, 0, nullptr, completion_signal_);
|
||||
hsa_status_t status = hsa_amd_memory_async_copy_rect(&dstMem, &offset,
|
||||
&srcMem, &offset, &dim, agent, direction, 1, &wait, active);
|
||||
if (status != HSA_STATUS_SUCCESS) {
|
||||
LogPrintfError("DMA buffer failed with code %d", status);
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!WaitForSignal(completion_signal_)) {
|
||||
LogError("Async copy failed");
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
// Fall to line by line copies
|
||||
const hsa_signal_value_t kInitVal = size[2] * size[1];
|
||||
hsa_signal_store_relaxed(completion_signal_, kInitVal);
|
||||
hsa_signal_t wait = gpu().Barriers().WaitSignal();
|
||||
hsa_signal_t active = gpu().Barriers().ActiveSignal(kInitVal, gpu().timestamp());
|
||||
|
||||
for (size_t z = 0; z < size[2]; ++z) {
|
||||
for (size_t y = 0; y < size[1]; ++y) {
|
||||
@@ -462,10 +463,10 @@ bool DmaBlitManager::copyBufferRect(device::Memory& srcMemory, device::Memory& d
|
||||
size_t dstOffset = dstRect.offset(0, y, z);
|
||||
|
||||
// Copy memory line by line
|
||||
hsa_status_t status =
|
||||
hsa_amd_memory_async_copy((reinterpret_cast<address>(dst) + dstOffset), dstAgent,
|
||||
(reinterpret_cast<const_address>(src) + srcOffset), srcAgent,
|
||||
size[0], 0, nullptr, completion_signal_);
|
||||
hsa_status_t status = hsa_amd_memory_async_copy(
|
||||
(reinterpret_cast<address>(dst) + dstOffset), dstAgent,
|
||||
(reinterpret_cast<const_address>(src) + srcOffset), srcAgent,
|
||||
size[0], 1, &wait, active);
|
||||
gpu().setLastCommandSDMA(true) ;
|
||||
if (status != HSA_STATUS_SUCCESS) {
|
||||
LogPrintfError("DMA buffer failed with code %d", status);
|
||||
@@ -473,14 +474,10 @@ bool DmaBlitManager::copyBufferRect(device::Memory& srcMemory, device::Memory& d
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!WaitForSignal(completion_signal_)) {
|
||||
LogError("Async copy failed");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
// Explicit wait for now, until runtime could distinguish compute and sdma operations
|
||||
gpu().Barriers().WaitCurrent();
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -489,12 +486,9 @@ bool DmaBlitManager::copyImageToBuffer(device::Memory& srcMemory, device::Memory
|
||||
const amd::Coord3D& srcOrigin, const amd::Coord3D& dstOrigin,
|
||||
const amd::Coord3D& size, bool entire, size_t rowPitch,
|
||||
size_t slicePitch) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
if (!dev().settings().barrier_sync_ && !gpu().isLastCommandSDMA()) {
|
||||
gpu().releaseGpuMemoryFence(true);
|
||||
} else {
|
||||
gpu().releaseGpuMemoryFence();
|
||||
}
|
||||
// HSA copy functionality with a possible async operation, hence make sure GPU is done
|
||||
bool force_barrier = !dev().settings().barrier_sync_ && !gpu().isLastCommandSDMA();
|
||||
gpu().releaseGpuMemoryFence(force_barrier);
|
||||
|
||||
bool result = false;
|
||||
|
||||
@@ -504,9 +498,6 @@ bool DmaBlitManager::copyImageToBuffer(device::Memory& srcMemory, device::Memory
|
||||
} else {
|
||||
Image& srcImage = static_cast<roc::Image&>(srcMemory);
|
||||
Buffer& dstBuffer = static_cast<roc::Buffer&>(dstMemory);
|
||||
|
||||
// Use ROC path for a transfer
|
||||
// Note: it doesn't support SDMA
|
||||
address dstHost = reinterpret_cast<address>(dstBuffer.getDeviceMemory()) + dstOrigin[0];
|
||||
|
||||
// Use ROCm path for a transfer.
|
||||
@@ -540,12 +531,9 @@ bool DmaBlitManager::copyBufferToImage(device::Memory& srcMemory, device::Memory
|
||||
const amd::Coord3D& srcOrigin, const amd::Coord3D& dstOrigin,
|
||||
const amd::Coord3D& size, bool entire, size_t rowPitch,
|
||||
size_t slicePitch) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
if (!dev().settings().barrier_sync_ && !gpu().isLastCommandSDMA()) {
|
||||
gpu().releaseGpuMemoryFence(true);
|
||||
} else {
|
||||
gpu().releaseGpuMemoryFence();
|
||||
}
|
||||
// HSA copy functionality with a possible async operation, hence make sure GPU is done
|
||||
bool force_barrier = !dev().settings().barrier_sync_ && !gpu().isLastCommandSDMA();
|
||||
gpu().releaseGpuMemoryFence(force_barrier);
|
||||
|
||||
bool result = false;
|
||||
|
||||
@@ -588,7 +576,7 @@ bool DmaBlitManager::copyBufferToImage(device::Memory& srcMemory, device::Memory
|
||||
bool DmaBlitManager::copyImage(device::Memory& srcMemory, device::Memory& dstMemory,
|
||||
const amd::Coord3D& srcOrigin, const amd::Coord3D& dstOrigin,
|
||||
const amd::Coord3D& size, bool entire) const {
|
||||
// HSA copy functionality with a possible async operaiton, hence make sure GPU is done
|
||||
// HSA copy functionality with a possible async operation, hence make sure GPU is done
|
||||
gpu().releaseGpuMemoryFence();
|
||||
|
||||
bool result = false;
|
||||
@@ -610,9 +598,8 @@ bool DmaBlitManager::hsaCopy(const Memory& srcMemory, const Memory& dstMemory,
|
||||
address src = reinterpret_cast<address>(srcMemory.getDeviceMemory());
|
||||
address dst = reinterpret_cast<address>(dstMemory.getDeviceMemory());
|
||||
|
||||
if (!dev().settings().barrier_sync_ && !gpu().isLastCommandSDMA()) {
|
||||
gpu().releaseGpuMemoryFence(true);
|
||||
}
|
||||
bool force_barrier = !dev().settings().barrier_sync_ && !gpu().isLastCommandSDMA();
|
||||
gpu().releaseGpuMemoryFence(force_barrier, kSkipCpuWait);
|
||||
|
||||
src += srcOrigin[0];
|
||||
dst += dstOrigin[0];
|
||||
@@ -620,6 +607,8 @@ bool DmaBlitManager::hsaCopy(const Memory& srcMemory, const Memory& dstMemory,
|
||||
// Just call copy function for full profile
|
||||
hsa_status_t status;
|
||||
if (dev().agent_profile() == HSA_PROFILE_FULL) {
|
||||
// Stall GPU, sicne CPU copy is possible
|
||||
gpu().Barriers().WaitCurrent();
|
||||
status = hsa_memory_copy(dst, src, size[0]);
|
||||
if (status != HSA_STATUS_SUCCESS) {
|
||||
LogPrintfError("Hsa copy of data failed with code %d", status);
|
||||
@@ -649,21 +638,15 @@ bool DmaBlitManager::hsaCopy(const Memory& srcMemory, const Memory& dstMemory,
|
||||
srcAgent = dstAgent = dev().getBackendDevice();
|
||||
}
|
||||
|
||||
hsa_signal_store_relaxed(completion_signal_, kInitSignalValueOne);
|
||||
|
||||
hsa_signal_t wait = gpu().Barriers().WaitSignal();
|
||||
hsa_signal_t active = gpu().Barriers().ActiveSignal(kInitSignalValueOne, gpu().timestamp());
|
||||
// Use SDMA to transfer the data
|
||||
status = hsa_amd_memory_async_copy(dst, dstAgent, src, srcAgent, size[0], 0, nullptr,
|
||||
completion_signal_);
|
||||
status = hsa_amd_memory_async_copy(dst, dstAgent, src, srcAgent, size[0], 1, &wait, active);
|
||||
gpu().setLastCommandSDMA(true);
|
||||
// Explicit wait for now, until runtime could distinguish compute and sdma operations
|
||||
gpu().Barriers().WaitCurrent();
|
||||
if (status == HSA_STATUS_SUCCESS) {
|
||||
hsa_signal_value_t val;
|
||||
|
||||
if (!WaitForSignal(completion_signal_)) {
|
||||
LogError("Async copy failed");
|
||||
status = HSA_STATUS_ERROR;
|
||||
} else {
|
||||
gpu().addSystemScope();
|
||||
}
|
||||
gpu().addSystemScope();
|
||||
} else {
|
||||
LogPrintfError("Hsa copy from host to device failed with code %d", status);
|
||||
}
|
||||
@@ -674,6 +657,10 @@ bool DmaBlitManager::hsaCopy(const Memory& srcMemory, const Memory& dstMemory,
|
||||
// ================================================================================================
|
||||
bool DmaBlitManager::hsaCopyStaged(const_address hostSrc, address hostDst, size_t size,
|
||||
address staging, bool hostToDev) const {
|
||||
// Stall GPU, sicne CPU copy is possible
|
||||
bool force_barrier = !dev().settings().barrier_sync_ && !gpu().isLastCommandSDMA();
|
||||
gpu().releaseGpuMemoryFence(force_barrier);
|
||||
|
||||
// No allocation is necessary for Full Profile
|
||||
hsa_status_t status;
|
||||
if (dev().agent_profile() == HSA_PROFILE_FULL) {
|
||||
@@ -688,14 +675,11 @@ bool DmaBlitManager::hsaCopyStaged(const_address hostSrc, address hostDst, size_
|
||||
size_t offset = 0;
|
||||
|
||||
address hsaBuffer = staging;
|
||||
if (!dev().settings().barrier_sync_ && !gpu().isLastCommandSDMA()) {
|
||||
gpu().releaseGpuMemoryFence(true);
|
||||
}
|
||||
|
||||
// Allocate requested size of memory
|
||||
while (totalSize > 0) {
|
||||
size = std::min(totalSize, dev().settings().stagedXferSize_);
|
||||
hsa_signal_silent_store_relaxed(completion_signal_, kInitSignalValueOne);
|
||||
hsa_signal_t active = gpu().Barriers().ActiveSignal(kInitSignalValueOne, gpu().timestamp());
|
||||
|
||||
// Copy data from Host to Device
|
||||
if (hostToDev) {
|
||||
@@ -707,17 +691,13 @@ bool DmaBlitManager::hsaCopyStaged(const_address hostSrc, address hostDst, size_
|
||||
|
||||
memcpy(hsaBuffer, hostSrc + offset, size);
|
||||
status = hsa_amd_memory_async_copy(hostDst + offset, dev().getBackendDevice(), hsaBuffer,
|
||||
srcAgent, size, 0, nullptr, completion_signal_);
|
||||
srcAgent, size, 0, nullptr, active);
|
||||
gpu().setLastCommandSDMA(true);
|
||||
if (status == HSA_STATUS_SUCCESS) {
|
||||
if (!WaitForSignal(completion_signal_)) {
|
||||
LogError("Async copy failed");
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
if (status != HSA_STATUS_SUCCESS) {
|
||||
LogPrintfError("Hsa copy from host to device failed with code %d", status);
|
||||
return false;
|
||||
}
|
||||
gpu().Barriers().WaitCurrent();
|
||||
totalSize -= size;
|
||||
offset += size;
|
||||
continue;
|
||||
@@ -730,15 +710,11 @@ bool DmaBlitManager::hsaCopyStaged(const_address hostSrc, address hostDst, size_
|
||||
(size <= dev().settings().sdmaCopyThreshold_) ? dev().getBackendDevice() : dev().getCpuAgent();
|
||||
|
||||
// Copy data from Device to Host
|
||||
status =
|
||||
hsa_amd_memory_async_copy(hsaBuffer, dstAgent, hostSrc + offset,
|
||||
dev().getBackendDevice(), size, 0, nullptr, completion_signal_);
|
||||
status = hsa_amd_memory_async_copy(hsaBuffer, dstAgent, hostSrc + offset,
|
||||
dev().getBackendDevice(), size, 0, nullptr, active);
|
||||
gpu().setLastCommandSDMA(true);
|
||||
if (status == HSA_STATUS_SUCCESS) {
|
||||
if (!WaitForSignal(completion_signal_)) {
|
||||
LogError("Async copy failed");
|
||||
return false;
|
||||
}
|
||||
gpu().Barriers().WaitCurrent();
|
||||
memcpy(hostDst + offset, hsaBuffer, size);
|
||||
} else {
|
||||
LogPrintfError("Hsa copy from device to host failed with code %d", status);
|
||||
@@ -1083,11 +1059,7 @@ bool KernelBlitManager::copyBufferToImageKernel(device::Memory& srcMemory,
|
||||
releaseArguments(parameters);
|
||||
if (releaseView) {
|
||||
// todo SRD programming could be changed to avoid a stall
|
||||
if(!dev().settings().barrier_sync_) {
|
||||
gpu().releaseGpuMemoryFence(true);
|
||||
} else {
|
||||
gpu().releaseGpuMemoryFence();
|
||||
}
|
||||
gpu().releaseGpuMemoryFence();
|
||||
dstView->owner()->release();
|
||||
}
|
||||
|
||||
@@ -1285,11 +1257,7 @@ bool KernelBlitManager::copyImageToBufferKernel(device::Memory& srcMemory,
|
||||
releaseArguments(parameters);
|
||||
if (releaseView) {
|
||||
// todo SRD programming could be changed to avoid a stall
|
||||
if(!dev().settings().barrier_sync_) {
|
||||
gpu().releaseGpuMemoryFence(true);
|
||||
} else {
|
||||
gpu().releaseGpuMemoryFence();
|
||||
}
|
||||
gpu().releaseGpuMemoryFence();
|
||||
srcView->owner()->release();
|
||||
}
|
||||
|
||||
@@ -1465,6 +1433,8 @@ bool KernelBlitManager::readImage(device::Memory& srcMemory, void* dstHost,
|
||||
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableReadImage_ || (srcMemory.isHostMemDirectAccess() && !srcMemory.isCpuUncached())) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
result = HostBlitManager::readImage(srcMemory, dstHost, origin, size, rowPitch, slicePitch, entire);
|
||||
synchronize();
|
||||
return result;
|
||||
@@ -1510,6 +1480,8 @@ bool KernelBlitManager::writeImage(const void* srcHost, device::Memory& dstMemor
|
||||
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableWriteImage_ || dstMemory.isHostMemDirectAccess()) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
result = HostBlitManager::writeImage(srcHost, dstMemory, origin, size, rowPitch, slicePitch, entire);
|
||||
synchronize();
|
||||
return result;
|
||||
@@ -1704,6 +1676,8 @@ bool KernelBlitManager::readBuffer(device::Memory& srcMemory, void* dstHost,
|
||||
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableReadBuffer_ || (srcMemory.isHostMemDirectAccess() && !srcMemory.isCpuUncached())) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
result = HostBlitManager::readBuffer(srcMemory, dstHost, origin, size, entire);
|
||||
synchronize();
|
||||
return result;
|
||||
@@ -1753,6 +1727,8 @@ bool KernelBlitManager::readBufferRect(device::Memory& srcMemory, void* dstHost,
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableReadBufferRect_ ||
|
||||
(srcMemory.isHostMemDirectAccess() && !srcMemory.isCpuUncached())) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
result = HostBlitManager::readBufferRect(srcMemory, dstHost, bufRect, hostRect, size, entire);
|
||||
synchronize();
|
||||
return result;
|
||||
@@ -1814,6 +1790,8 @@ bool KernelBlitManager::writeBuffer(const void* srcHost, device::Memory& dstMemo
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableWriteBuffer_ || dstMemory.isHostMemDirectAccess() ||
|
||||
gpuMem(dstMemory).IsPersistentDirectMap()) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
result = HostBlitManager::writeBuffer(srcHost, dstMemory, origin, size, entire);
|
||||
synchronize();
|
||||
return result;
|
||||
@@ -1864,6 +1842,8 @@ bool KernelBlitManager::writeBufferRect(const void* srcHost, device::Memory& dst
|
||||
// Use host copy if memory has direct access
|
||||
if (setup_.disableWriteBufferRect_ || dstMemory.isHostMemDirectAccess() ||
|
||||
gpuMem(dstMemory).IsPersistentDirectMap()) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
result = HostBlitManager::writeBufferRect(srcHost, dstMemory, hostRect, bufRect, size, entire);
|
||||
synchronize();
|
||||
return result;
|
||||
@@ -1913,6 +1893,8 @@ bool KernelBlitManager::fillBuffer(device::Memory& memory, const void* pattern,
|
||||
|
||||
// Use host fill if memory has direct access
|
||||
if (setup_.disableFillBuffer_ || memory.isHostMemDirectAccess()) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
result = HostBlitManager::fillBuffer(memory, pattern, patternSize, origin, size, entire);
|
||||
synchronize();
|
||||
return result;
|
||||
@@ -2074,6 +2056,8 @@ bool KernelBlitManager::fillImage(device::Memory& memory, const void* pattern,
|
||||
|
||||
// Use host fill if memory has direct access
|
||||
if (setup_.disableFillImage_ || memory.isHostMemDirectAccess()) {
|
||||
// Stall GPU before CPU access
|
||||
gpu().releaseGpuMemoryFence();
|
||||
result = HostBlitManager::fillImage(memory, pattern, origin, size, entire);
|
||||
synchronize();
|
||||
return result;
|
||||
|
||||
Reference in New Issue
Block a user