P4 to Git Change 1451293 by gandryey@gera-w8 on 2017/08/24 13:37:00
SWDEV-129129 - [[CQE OCL][Vega vs Fiji] Upto 12% Performance drop observed on VEGA10 compared to FIJI while running BlackMagic Davinci Resolve
The app creates/destroys hundred resources each frame. PAL path was removing the destroyed resources from the resident list, although the resource was kept in the cache. This change does the follwoing:
- Switch TS tracking from a map in VirtualGPU to resource
- Don't remove references until the actual memory destruction
- Add a residency threshold to avoid OS resident/eviction calls
Affected files ...
... //depot/stg/opencl/drivers/opencl/runtime/device/blit.hpp#5 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palblit.cpp#14 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palblit.hpp#6 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/paldefs.hpp#19 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/paldevice.cpp#50 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/paldevice.hpp#17 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.cpp#35 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.hpp#13 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palmemory.cpp#14 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palprogram.cpp#46 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palprogram.hpp#19 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palresource.cpp#30 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palresource.hpp#13 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.cpp#52 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.hpp#28 edit
[ROCm/clr commit: b82be1113f]
This commit is contained in:
@@ -86,6 +86,7 @@ void Segment::copy(size_t offset, const void* src, size_t size) {
|
||||
if (cpuAccess_ != nullptr) {
|
||||
amd::Os::fastMemcpy(cpuAddress(offset), src, size);
|
||||
} else {
|
||||
amd::ScopedLock k(gpuAccess_->dev().xferMgr().lockXfer());
|
||||
VirtualGPU& gpu = *gpuAccess_->dev().xferQueue();
|
||||
Memory& xferBuf = gpuAccess_->dev().xferWrite().acquire();
|
||||
size_t tmpSize = std::min(static_cast<size_t>(xferBuf.vmSize()), size);
|
||||
@@ -98,7 +99,6 @@ void Segment::copy(size_t offset, const void* src, size_t size) {
|
||||
srcOffs += tmpSize;
|
||||
tmpSize = std::min(static_cast<size_t>(xferBuf.vmSize()), size);
|
||||
}
|
||||
gpu.releaseMemObjects();
|
||||
gpu.waitAllEngines();
|
||||
}
|
||||
}
|
||||
@@ -108,8 +108,8 @@ bool Segment::freeze(bool destroySysmem) {
|
||||
bool result = true;
|
||||
if (cpuAccess_ != nullptr) {
|
||||
assert(gpuAccess_->size() == cpuAccess_->size() && "Backing store size mismatch!");
|
||||
amd::ScopedLock k(gpuAccess_->dev().xferMgr().lockXfer());
|
||||
result = cpuAccess_->partialMemCopyTo(gpu, 0, 0, gpuAccess_->size(), *gpuAccess_, false, true);
|
||||
gpu.releaseMemObjects();
|
||||
gpu.waitAllEngines();
|
||||
}
|
||||
assert(!destroySysmem || (cpuAccess_ == nullptr));
|
||||
@@ -813,8 +813,8 @@ bool HSAILProgram::allocKernelTable() {
|
||||
return true;
|
||||
}
|
||||
|
||||
void HSAILProgram::fillResListWithKernels(std::vector<const Memory*>& memList) const {
|
||||
memList.push_back(&codeSegGpu());
|
||||
void HSAILProgram::fillResListWithKernels(VirtualGPU& gpu) const {
|
||||
gpu.addVmMemory(&codeSegGpu());
|
||||
}
|
||||
|
||||
const aclTargetInfo& HSAILProgram::info(const char* str) {
|
||||
|
||||
Reference in New Issue
Block a user