From f697264676643f6ce4f5411ba99dae3005078283 Mon Sep 17 00:00:00 2001
From: foreman
Date: Fri, 20 Nov 2015 18:51:27 -0500
Subject: [PATCH] P4 to Git Change 1213547 by
skudchad@skudchad_test_win_opencl2 on 2015/11/20 18:45:03
SWDEV-77172 - IOMMUv2 changes for Windows 10 (Part 2)
- Fix SVM FGS malloc arguments being passed to kernel
- Move setting allocation attributes to runtime
- Fix some bugs from my previous checkins
ReviewBoardURL = http://ocltc.amd.com/reviews/r/9035/diff/
Affected files ...
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpukernel.cpp#311 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuresource.cpp#232 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuresource.hpp#86 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuvirtual.cpp#393 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gslbe/src/rt/GSLDevice.cpp#150 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gslbe/src/rt/backend.h#11 edit
---
rocclr/runtime/device/gpu/gpukernel.cpp | 2 +-
rocclr/runtime/device/gpu/gpuresource.cpp | 28 +++++++++++++++----
rocclr/runtime/device/gpu/gpuresource.hpp | 3 +-
rocclr/runtime/device/gpu/gpuvirtual.cpp | 2 ++
.../device/gpu/gslbe/src/rt/GSLDevice.cpp | 2 ++
.../runtime/device/gpu/gslbe/src/rt/backend.h | 1 +
6 files changed, 31 insertions(+), 7 deletions(-)
diff --git a/rocclr/runtime/device/gpu/gpukernel.cpp b/rocclr/runtime/device/gpu/gpukernel.cpp
index c0e2f9fc2e..58d859fd84 100644
--- a/rocclr/runtime/device/gpu/gpukernel.cpp
+++ b/rocclr/runtime/device/gpu/gpukernel.cpp
@@ -3803,7 +3803,7 @@ HSAILKernel::loadArguments(
}
// If finegrainsystem is present then the pointer can be malloced by the app and
// passed to kernel directly. If so copy the pointer location to aqlArgBuf
- else if (dev().isFineGrainedSystem(true)) {
+ else if (!dev().isFineGrainedSystem(true)) {
return NULL;
}
break;
diff --git a/rocclr/runtime/device/gpu/gpuresource.cpp b/rocclr/runtime/device/gpu/gpuresource.cpp
index 155ed575ba..aa3d1fdf07 100644
--- a/rocclr/runtime/device/gpu/gpuresource.cpp
+++ b/rocclr/runtime/device/gpu/gpuresource.cpp
@@ -87,8 +87,9 @@ Resource::Resource(
cal_.buffer_ = true;
cal_.imageArray_ = false;
cal_.imageType_ = 0;
- cal_.SVMRes_ = false;
+ cal_.skipRsrcCache_ = false;
cal_.scratch_ = false;
+ cal_.isAllocSVM_ = false;
cal_.isAllocExecute_ = false;
}
@@ -130,8 +131,9 @@ Resource::Resource(
cal_.buffer_ = false;
cal_.imageArray_ = false;
cal_.imageType_ = imageType;
- cal_.SVMRes_ = false;
+ cal_.skipRsrcCache_ = false;
cal_.scratch_ = false;
+ cal_.isAllocSVM_ = false;
cal_.isAllocExecute_ = false;
switch (imageType) {
@@ -348,21 +350,31 @@ Resource::create(MemoryType memType, CreateParams* params)
desc.vaBase = 0;
desc.minAlignment = 0;
desc.isAllocExecute = false;
+ desc.isAllocSVM = false;
desc.section = GSL_SECTION_REGULAR;
if (NULL != params && NULL != params->owner_) { //make sure params not NULL
mcaddr svmPtr = reinterpret_cast(params->owner_->getSvmPtr());
desc.vaBase = (svmPtr == 1)? 0:svmPtr;
- cal_.SVMRes_ = (svmPtr != 0);
+ // Dont cache coarse\fine grain svm resource as these may not be released
+ // and allocations may fail since there is limited space for coarse\fine grainbuffers
+ cal_.skipRsrcCache_ = (svmPtr != 0);
desc.section = (svmPtr != 0) ? GSL_SECTION_SVM : GSL_SECTION_REGULAR;
if (params->owner_->getMemFlags() & CL_MEM_SVM_ATOMICS) {
desc.section = GSL_SECTION_SVM_ATOMICS;
}
+
+ if (dev().settings().svmFineGrainSystem_ &&
+ (desc.section == GSL_SECTION_SVM ||
+ desc.section == GSL_SECTION_SVM_ATOMICS)) {
+ cal_.isAllocSVM_ = desc.isAllocSVM = true;
+ }
}
if (memType == Shader){
if(dev().settings().svmFineGrainSystem_) {
cal_.isAllocExecute_ = desc.isAllocExecute = true;
+ cal_.isAllocSVM_ = desc.isAllocSVM = true;
}
// force to use remote memory for HW DEBUG or use
// local memory once we determine if FGS is supported
@@ -596,6 +608,11 @@ Resource::create(MemoryType memType, CreateParams* params)
return false;
}
+ if(dev().settings().svmFineGrainSystem_) {
+ cal_.isAllocExecute_ = desc.isAllocExecute = true;
+ cal_.isAllocSVM_ = desc.isAllocSVM = true;
+ }
+
gslResource = dev().resAlloc(&desc);
if (gslResource != 0) {
calRes = true;
@@ -1893,7 +1910,7 @@ ResourceCache::addCalResource(
(desc->type_ == Resource::Remote) ||
(desc->type_ == Resource::RemoteUSWC)) &&
(size < cacheSizeLimit_) &&
- !desc->SVMRes_) {
+ !desc->skipRsrcCache_) {
// Validate the cache size limit. Loop until we have enough space
while ((cacheSize_ + size) > cacheSizeLimit_) {
removeLast();
@@ -1921,7 +1938,7 @@ ResourceCache::findCalResource(Resource::CalResourceDesc* desc)
size_t size = getResourceSize(desc);
// Early exit if resource is too big
- if (size >= cacheSizeLimit_ || desc->SVMRes_) {
+ if (size >= cacheSizeLimit_ || desc->skipRsrcCache_) {
//! \note we may need to free the cache here to reduce memory pressure
return ref;
}
@@ -1939,6 +1956,7 @@ ResourceCache::findCalResource(Resource::CalResourceDesc* desc)
(entry->format_ == desc->format_) &&
(entry->flags_ == desc->flags_) &&
(entry->mipLevels_ == desc->mipLevels_) &&
+ (entry->isAllocSVM_ == desc->isAllocSVM_) &&
(entry->isAllocExecute_ == desc->isAllocExecute_)) {
ref = it.second;
delete it.first;
diff --git a/rocclr/runtime/device/gpu/gpuresource.hpp b/rocclr/runtime/device/gpu/gpuresource.hpp
index 88d67dda99..338fa9050a 100644
--- a/rocclr/runtime/device/gpu/gpuresource.hpp
+++ b/rocclr/runtime/device/gpu/gpuresource.hpp
@@ -173,8 +173,9 @@ public:
uint imageArray_ : 1; //!< GSL resource is an array of images
uint buffer_ : 1; //!< GSL resource is a buffer
uint tiled_ : 1; //!< GSL resource is tiled
- uint SVMRes_ : 1; //!< SVM flag to the cal resource
uint scratch_ : 1; //!< Scratch buffer
+ uint skipRsrcCache_ : 1; //!< Skip caching of a cal resource
+ uint isAllocSVM_ : 1; //!< SVM resource attribute
uint isAllocExecute_ : 1; //!< SVM resource allocation attribute for shader\cmdbuf
};
uint state_;
diff --git a/rocclr/runtime/device/gpu/gpuvirtual.cpp b/rocclr/runtime/device/gpu/gpuvirtual.cpp
index 24efc9a5e4..f79c80a576 100644
--- a/rocclr/runtime/device/gpu/gpuvirtual.cpp
+++ b/rocclr/runtime/device/gpu/gpuvirtual.cpp
@@ -3232,6 +3232,7 @@ VirtualGPU::processMemObjectsHSA(
// Clear memory dependency state
const static bool All = true;
memoryDependency().clear(!All);
+ continue;
}
}
else {
@@ -3270,6 +3271,7 @@ VirtualGPU::processMemObjectsHSA(
// Clear memory dependency state
const static bool All = true;
memoryDependency().clear(!All);
+ continue;
}
}
diff --git a/rocclr/runtime/device/gpu/gslbe/src/rt/GSLDevice.cpp b/rocclr/runtime/device/gpu/gslbe/src/rt/GSLDevice.cpp
index fed1771593..b2908d3c57 100644
--- a/rocclr/runtime/device/gpu/gslbe/src/rt/GSLDevice.cpp
+++ b/rocclr/runtime/device/gpu/gslbe/src/rt/GSLDevice.cpp
@@ -635,6 +635,7 @@ CALGSLDevice::resAlloc(const CALresourceDesc* desc) const
attribs.location = desc->type;
attribs.vaBase = desc->vaBase;
attribs.section = desc->section;
+ attribs.isAllocSVM = desc->isAllocSVM;
attribs.isAllocExecute = desc->isAllocExecute;
attribs.minAlignment = desc->minAlignment;
@@ -744,6 +745,7 @@ CALGSLDevice::resAllocView(gslMemObject res, gslResource3D size, size_t offset,
);
attribs.bytePitch = bytePitch;
attribs.section = res->getAttribs().section;
+ attribs.isAllocSVM = res->getAttribs().isAllocSVM;
attribs.isAllocExecute = res->getAttribs().isAllocExecute;
// Need to get the alignment info from hwl.
diff --git a/rocclr/runtime/device/gpu/gslbe/src/rt/backend.h b/rocclr/runtime/device/gpu/gslbe/src/rt/backend.h
index a8d7ab96bb..0308691384 100644
--- a/rocclr/runtime/device/gpu/gslbe/src/rt/backend.h
+++ b/rocclr/runtime/device/gpu/gslbe/src/rt/backend.h
@@ -149,6 +149,7 @@ typedef struct CALresourceDescRec {
mcaddr vaBase;
gslMemObjectAttribSection section;
CALuint minAlignment;
+ bool isAllocSVM;
bool isAllocExecute;
} CALresourceDesc;