SWDEV-365121 - Use CP DMA for tiny transfers

Sync between compute and SDMA engines can be very expensive under Windows.
Use CP DMA for tiny transfers (< 1KiB) to avoid syncs and improve performance.

Change-Id: I9db39a2199f7b9e337ed08fd36d9cbc150502f1f
This commit is contained in:
German
2022-10-31 14:17:57 -04:00
کامیت شده توسط German Andryeyev
والد 06573ac92f
کامیت 473621c008
4فایلهای تغییر یافته به همراه7 افزوده شده و 1 حذف شده
@@ -1425,7 +1425,9 @@ bool Resource::partialMemCopyTo(VirtualGPU& gpu, const amd::Coord3D& srcOrigin,
}
}
bool cp_dma = dev().settings().disableSdma_;
bool cp_dma = dev().settings().disableSdma_ ||
(!enableCopyRect && desc().buffer_ && dstResource.desc().buffer_ &&
(size[0] < dev().settings().cpDmaCopySizeMax_));
if (cp_dma) {
// Make sure compute is done before CP DMA start
gpu.addBarrier(RgpSqqtBarrierReason::MemDependency);
@@ -171,6 +171,7 @@ Settings::Settings() {
mallPolicy_ = 0;
alwaysResident_ = amd::IS_HIP ? true : false;
prepinnedMinSize_ = 0;
cpDmaCopySizeMax_ = GPU_CP_DMA_COPY_SIZE * Ki;
}
bool Settings::create(const Pal::DeviceProperties& palProp,
@@ -107,6 +107,7 @@ class Settings : public device::Settings {
size_t stagedXferSize_; //!< Staged buffer size
size_t pinnedXferSize_; //!< Pinned buffer size for transfer
size_t pinnedMinXferSize_; //!< Minimal buffer size for pinned transfer
size_t cpDmaCopySizeMax_; //!< Threshold for CP DMA path in copy
size_t resourceCacheSize_; //!< Resource cache size in MB
size_t numMemDependencies_; //!< The array size for memory dependencies tracking
uint64_t maxAllocSize_; //!< Maximum single allocation size
+2
مشاهده پرونده
@@ -62,6 +62,8 @@ release(cstring, GPU_DEVICE_ORDINAL, "", \
"Select the device ordinal (comma seperated list of available devices)") \
release(bool, REMOTE_ALLOC, false, \
"Use remote memory for the global heap allocation") \
release(uint, GPU_CP_DMA_COPY_SIZE, 1, \
"Set maximum size of CP DMA copy in KiB") \
release(uint, GPU_MAX_HEAP_SIZE, 100, \
"Set maximum size of the GPU heap to % of board memory") \
release(uint, GPU_STAGING_BUFFER_SIZE, 1024, \