From 9ffcae945087a69ccc06d141259e9ee7e026959b Mon Sep 17 00:00:00 2001
From: foreman
Date: Thu, 22 Mar 2018 17:58:21 -0400
Subject: [PATCH] P4 to Git Change 1530988 by gandryey@gera-w8 on 2018/03/22
17:50:10
SWDEV-79445 - OCL generic changes and code clean-up
- Remove renames support from the Resource object.
Affected files ...
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palblit.cpp#18 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palresource.cpp#54 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palresource.hpp#19 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palsettings.cpp#47 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palsettings.hpp#17 edit
[ROCm/clr commit: ace31f6a1149959ec8142838b9c4b1dda089041a]
---
.../clr/rocclr/runtime/device/pal/palblit.cpp | 5 +-
.../rocclr/runtime/device/pal/palresource.cpp | 166 ++----------------
.../rocclr/runtime/device/pal/palresource.hpp | 26 +--
.../rocclr/runtime/device/pal/palsettings.cpp | 3 -
.../rocclr/runtime/device/pal/palsettings.hpp | 2 -
5 files changed, 18 insertions(+), 184 deletions(-)
diff --git a/projects/clr/rocclr/runtime/device/pal/palblit.cpp b/projects/clr/rocclr/runtime/device/pal/palblit.cpp
index 30a7143b9e..40030668df 100644
--- a/projects/clr/rocclr/runtime/device/pal/palblit.cpp
+++ b/projects/clr/rocclr/runtime/device/pal/palblit.cpp
@@ -278,8 +278,7 @@ bool DmaBlitManager::writeMemoryStaged(const void* srcHost, Memory& dstMemory, M
amd::Coord3D copySize(tmpSize, 0, 0);
// Copy data into the temporary buffer, using CPU
- if (!xferBuf.hostWrite(&gpu(), reinterpret_cast(srcHost) + offset, src, copySize,
- Resource::Discard)) {
+ if (!xferBuf.hostWrite(&gpu(), reinterpret_cast(srcHost) + offset, src, copySize)) {
return false;
}
@@ -412,7 +411,7 @@ bool DmaBlitManager::writeBufferRect(const void* srcHost, device::Memory& dstMem
// Copy data into the temporary buffer, using CPU
if (!xferBuf.hostWrite(&gpu(), reinterpret_cast(srcHost) + hostOffset, src,
- copySize, Resource::Discard)) {
+ copySize)) {
LogError("DmaBlitManager::writeBufferRect failed!");
return false;
}
diff --git a/projects/clr/rocclr/runtime/device/pal/palresource.cpp b/projects/clr/rocclr/runtime/device/pal/palresource.cpp
index 6f3f581ffa..39af9b9283 100644
--- a/projects/clr/rocclr/runtime/device/pal/palresource.cpp
+++ b/projects/clr/rocclr/runtime/device/pal/palresource.cpp
@@ -188,7 +188,6 @@ Resource::Resource(const Device& gpuDev, size_t size)
mapCount_(0),
address_(nullptr),
offset_(0),
- curRename_(0),
memRef_(nullptr),
subOffset_(0),
viewOwner_(nullptr),
@@ -228,7 +227,6 @@ Resource::Resource(const Device& gpuDev, size_t width, size_t height, size_t dep
mapCount_(0),
address_(nullptr),
offset_(0),
- curRename_(0),
memRef_(nullptr),
subOffset_(0),
viewOwner_(nullptr),
@@ -747,7 +745,7 @@ bool Resource::CreateInterop(CreateParams* params)
imgCreateInfo.depthPitch = desc().height_ * imgCreateInfo.rowPitch;
switch (misc) {
- case 1: // NV12 format
+ case 1: // NV12 or P010 formats
switch (layer) {
case -1:
case 0:
@@ -1191,34 +1189,22 @@ void Resource::free()
memRef_->gpu_ = nullptr;
}
- if (renames_.size() == 0) {
- // Destroy GSL resource
- if (iMem() != 0) {
- if (mapCount_ != 0) {
- if ((memoryType() != Remote) && (memoryType() != RemoteUSWC)) {
- //! @note: This is a workaround for bad applications that
- //! don't unmap memory
- unmap(nullptr);
- } else {
- // Delay CPU address unmap until memRef_ destruction
- assert(memRef_->cpuAddress_ == nullptr && "Memref shouldn't have a valid CPU address");
- memRef_->cpuAddress_ = address_;
- }
- }
-
- // Add resource to the cache
- if (!dev().resourceCache().addGpuMemory(&desc_, memRef_, subOffset_)) {
- palFree();
+ // Destroy PAL resource
+ if (iMem() != 0) {
+ if (mapCount_ != 0) {
+ if ((memoryType() != Remote) && (memoryType() != RemoteUSWC)) {
+ //! @note: This is a workaround for bad applications that don't unmap memory
+ unmap(nullptr);
+ } else {
+ // Delay CPU address unmap until memRef_ destruction
+ assert(memRef_->cpuAddress_ == nullptr && "Memref shouldn't have a valid CPU address");
+ memRef_->cpuAddress_ = address_;
}
}
- } else {
- renames_[curRename_]->cpuAddress_ = 0;
- for (size_t i = 0; i < renames_.size(); ++i) {
- memRef_ = renames_[i];
- // Destroy PAL resource
- if (iMem() != 0) {
- palFree();
- }
+
+ // Add resource to the cache
+ if (!dev().resourceCache().addGpuMemory(&desc_, memRef_, subOffset_)) {
+ palFree();
}
}
@@ -1280,9 +1266,7 @@ bool Resource::partialMemCopyTo(VirtualGPU& gpu, const amd::Coord3D& srcOrigin,
GpuEvent event;
EngineType activeEngineID = gpu.engineID_;
static const bool waitOnBusyEngine = true;
-
assert(!(desc().cardMemory_ && dstResource.desc().cardMemory_) && "Unsupported configuraiton!");
-
uint64_t gpuMemoryOffset = 0;
uint64_t gpuMemoryRowPitch = 0;
uint64_t imageOffsetx = 0;
@@ -1714,16 +1698,6 @@ void* Resource::map(VirtualGPU* gpu, uint flags, uint startLayer, uint numLayers
if (flags & WriteOnly) {
}
- // Check if use map discard
- if (flags & Discard) {
- if (gpu != nullptr) {
- // If we use a new renamed allocation, then skip the wait
- if (rename(*gpu)) {
- flags |= NoWait;
- }
- }
- }
-
// Check if we have to wait
if (!(flags & NoWait)) {
if (gpu != nullptr) {
@@ -1803,116 +1777,6 @@ void Resource::unmapLayers(VirtualGPU* gpu) {
Unimplemented();
}
-// ================================================================================================
-void Resource::setActiveRename(VirtualGPU& gpu, GpuMemoryReference* rename) {
- // Copy the unique GSL data
- memRef_ = rename;
- address_ = rename->cpuAddress_;
-}
-
-// ================================================================================================
-bool Resource::getActiveRename(VirtualGPU& gpu, GpuMemoryReference** rename) {
- // Copy the old data to the rename descriptor
- *rename = memRef_;
- return true;
-}
-
-// ================================================================================================
-bool Resource::rename(VirtualGPU& gpu, bool force) {
- GpuEvent* gpuEvent = getGpuEvent(gpu);
- if (!gpuEvent->isValid() && !force) {
- return true;
- }
-
- bool useNext = false;
- uint resSize = desc().width_ * ((desc().height_) ? desc().height_ : 1) * elementSize_;
-
- // Rename will work with real GSL resources
- if (((memoryType() != Local) && (memoryType() != Persistent) && (memoryType() != Remote) &&
- (memoryType() != RemoteUSWC)) ||
- (dev().settings().maxRenames_ == 0)) {
- return false;
- }
-
- // If the resource for renaming is too big, then lets check the current status first
- // at the cost of an extra flush
- if (resSize >= (dev().settings().maxRenameSize_ / dev().settings().maxRenames_)) {
- if (gpu.isDone(gpuEvent)) {
- return true;
- }
- }
-
- // Save the first
- if (renames_.size() == 0) {
- GpuMemoryReference* rename;
- if (mapCount_ > 0) {
- memRef_->cpuAddress_ = address_;
- }
- if (!getActiveRename(gpu, &rename)) {
- return false;
- }
-
- curRename_ = renames_.size();
- renames_.push_back(rename);
- }
-
- // Can we use a new rename?
- if ((renames_.size() <= dev().settings().maxRenames_) &&
- ((renames_.size() * resSize) <= dev().settings().maxRenameSize_)) {
- GpuMemoryReference* rename;
-
- // Create a new GSL allocation
- if (create(memoryType())) {
- if (mapCount_ > 0) {
- assert(!desc().cardMemory_ && "Unsupported memory type!");
- memRef_->cpuAddress_ = gpuMemoryMap(&desc_.pitch_, 0, iMem());
- if (memRef_->cpuAddress_ == nullptr) {
- LogError("gslMap fails on rename!");
- }
- address_ = memRef_->cpuAddress_;
- }
- if (getActiveRename(gpu, &rename)) {
- curRename_ = renames_.size();
- renames_.push_back(rename);
- } else {
- memRef_->release();
- useNext = true;
- }
- } else {
- useNext = true;
- }
- } else {
- useNext = true;
- }
-
- if (useNext) {
- // Get the last submitted
- curRename_++;
- if (curRename_ >= renames_.size()) {
- curRename_ = 0;
- }
- setActiveRename(gpu, renames_[curRename_]);
- return false;
- }
-
- return true;
-}
-
-// ================================================================================================
-void Resource::warmUpRenames(VirtualGPU& gpu) {
- // Make sure OCL touches every command buffer in the queue to avoid delays on the first submit
- uint flush = dev().settings().maxRenames_ / VirtualGPU::Queue::MaxCmdBuffers;
- flush = (flush == 0) ? 1 : flush;
- for (uint i = 1; i <= dev().settings().maxRenames_; ++i) {
- uint dummy = 0;
- const bool Wait = (i % flush == 0) ? true : false;
- // Write 0 for the buffer paging by VidMM
- writeRawData(gpu, 0, sizeof(dummy), &dummy, Wait);
- const bool Force = true;
- rename(gpu, Force);
- }
-}
-
// ================================================================================================
MemorySubAllocator::~MemorySubAllocator()
{
diff --git a/projects/clr/rocclr/runtime/device/pal/palresource.hpp b/projects/clr/rocclr/runtime/device/pal/palresource.hpp
index b7fd9ac341..510c802823 100644
--- a/projects/clr/rocclr/runtime/device/pal/palresource.hpp
+++ b/projects/clr/rocclr/runtime/device/pal/palresource.hpp
@@ -146,7 +146,6 @@ class Resource : public amd::HeapObject {
//! Resource map flags
enum MapFlags {
- Discard = 0x00000001, //!< discard lock
NoOverwrite = 0x00000002, //!< lock with no overwrite
ReadOnly = 0x00000004, //!< lock for read only operation
WriteOnly = 0x00000008, //!< lock for write only operation
@@ -313,17 +312,13 @@ class Resource : public amd::HeapObject {
size_t slicePitch = 0 //!< Raw data slice pitch
);
- //! Warms up the rename list for this resource
- void warmUpRenames(VirtualGPU& gpu);
-
//! Gets the resource element size
uint elementSize() const { return elementSize_; }
//! Get the mapped address of this resource
address data() const { return reinterpret_cast(address_); }
- //! Frees all allocated PAL memories and resources,
- //! associated with this objects. And also destroys all rename structures
+ //! Frees all allocated PAL memories and resources, associated with this objects.
//! Note: doesn't destroy the object itself
void free();
@@ -401,23 +396,6 @@ class Resource : public amd::HeapObject {
//! Disable operator=
Resource& operator=(const Resource&);
- typedef std::vector RenameList;
-
- //! Rename current resource
- bool rename(VirtualGPU& gpu, //!< Virtual GPU device object
- bool force = false //!< Force renaming
- );
-
- //! Sets the rename as active
- void setActiveRename(VirtualGPU& gpu, //!< Virtual GPU device object
- GpuMemoryReference* rename //!< new active rename
- );
-
- //! Gets the active rename
- bool getActiveRename(VirtualGPU& gpu, //!< Virtual GPU device object
- GpuMemoryReference** rename //!< Saved active rename
- );
-
/*! \brief Locks the resource with layers and returns a physical pointer
*
* \return Pointer to the physical memory
@@ -452,8 +430,6 @@ class Resource : public amd::HeapObject {
amd::Atomic mapCount_; //!< Total number of maps
void* address_; //!< Physical address of this resource
size_t offset_; //!< Resource offset
- uint32_t curRename_; //!< Current active rename in the list
- RenameList renames_; //!< Rename resource list
GpuMemoryReference* memRef_; //!< PAL resource reference
Pal::gpusize subOffset_; //!< GPU memory offset in the oririnal resource
const Resource* viewOwner_; //!< GPU resource, which owns this view
diff --git a/projects/clr/rocclr/runtime/device/pal/palsettings.cpp b/projects/clr/rocclr/runtime/device/pal/palsettings.cpp
index 2746439ef8..4a69a13cf1 100644
--- a/projects/clr/rocclr/runtime/device/pal/palsettings.cpp
+++ b/projects/clr/rocclr/runtime/device/pal/palsettings.cpp
@@ -47,9 +47,6 @@ Settings::Settings() {
// By Default persistent writes will be disabled.
stagingWritePersistent_ = GPU_STAGING_WRITE_PERSISTENT;
- maxRenames_ = 4;
- maxRenameSize_ = 4 * Mi;
-
imageSupport_ = false;
hwLDSSize_ = 0;
diff --git a/projects/clr/rocclr/runtime/device/pal/palsettings.hpp b/projects/clr/rocclr/runtime/device/pal/palsettings.hpp
index 1cdad9c095..90dc533bf6 100644
--- a/projects/clr/rocclr/runtime/device/pal/palsettings.hpp
+++ b/projects/clr/rocclr/runtime/device/pal/palsettings.hpp
@@ -73,8 +73,6 @@ class Settings : public device::Settings {
uint oclVersion_; //!< Reported OpenCL version support
uint debugFlags_; //!< Debug GPU flags
- uint maxRenames_; //!< Maximum number of possible renames
- uint maxRenameSize_; //!< Maximum size for all renames
uint hwLDSSize_; //!< HW local data store size
uint maxWorkGroupSize_; //!< Requested workgroup size for this device
uint preferredWorkGroupSize_;//!< Requested preferred workgroup size for this device