diff --git a/projects/clr/rocclr/runtime/device/pal/palblit.cpp b/projects/clr/rocclr/runtime/device/pal/palblit.cpp index 30a7143b9e..40030668df 100644 --- a/projects/clr/rocclr/runtime/device/pal/palblit.cpp +++ b/projects/clr/rocclr/runtime/device/pal/palblit.cpp @@ -278,8 +278,7 @@ bool DmaBlitManager::writeMemoryStaged(const void* srcHost, Memory& dstMemory, M amd::Coord3D copySize(tmpSize, 0, 0); // Copy data into the temporary buffer, using CPU - if (!xferBuf.hostWrite(&gpu(), reinterpret_cast(srcHost) + offset, src, copySize, - Resource::Discard)) { + if (!xferBuf.hostWrite(&gpu(), reinterpret_cast(srcHost) + offset, src, copySize)) { return false; } @@ -412,7 +411,7 @@ bool DmaBlitManager::writeBufferRect(const void* srcHost, device::Memory& dstMem // Copy data into the temporary buffer, using CPU if (!xferBuf.hostWrite(&gpu(), reinterpret_cast(srcHost) + hostOffset, src, - copySize, Resource::Discard)) { + copySize)) { LogError("DmaBlitManager::writeBufferRect failed!"); return false; } diff --git a/projects/clr/rocclr/runtime/device/pal/palresource.cpp b/projects/clr/rocclr/runtime/device/pal/palresource.cpp index 6f3f581ffa..39af9b9283 100644 --- a/projects/clr/rocclr/runtime/device/pal/palresource.cpp +++ b/projects/clr/rocclr/runtime/device/pal/palresource.cpp @@ -188,7 +188,6 @@ Resource::Resource(const Device& gpuDev, size_t size) mapCount_(0), address_(nullptr), offset_(0), - curRename_(0), memRef_(nullptr), subOffset_(0), viewOwner_(nullptr), @@ -228,7 +227,6 @@ Resource::Resource(const Device& gpuDev, size_t width, size_t height, size_t dep mapCount_(0), address_(nullptr), offset_(0), - curRename_(0), memRef_(nullptr), subOffset_(0), viewOwner_(nullptr), @@ -747,7 +745,7 @@ bool Resource::CreateInterop(CreateParams* params) imgCreateInfo.depthPitch = desc().height_ * imgCreateInfo.rowPitch; switch (misc) { - case 1: // NV12 format + case 1: // NV12 or P010 formats switch (layer) { case -1: case 0: @@ -1191,34 +1189,22 @@ void Resource::free() memRef_->gpu_ = nullptr; } - if (renames_.size() == 0) { - // Destroy GSL resource - if (iMem() != 0) { - if (mapCount_ != 0) { - if ((memoryType() != Remote) && (memoryType() != RemoteUSWC)) { - //! @note: This is a workaround for bad applications that - //! don't unmap memory - unmap(nullptr); - } else { - // Delay CPU address unmap until memRef_ destruction - assert(memRef_->cpuAddress_ == nullptr && "Memref shouldn't have a valid CPU address"); - memRef_->cpuAddress_ = address_; - } - } - - // Add resource to the cache - if (!dev().resourceCache().addGpuMemory(&desc_, memRef_, subOffset_)) { - palFree(); + // Destroy PAL resource + if (iMem() != 0) { + if (mapCount_ != 0) { + if ((memoryType() != Remote) && (memoryType() != RemoteUSWC)) { + //! @note: This is a workaround for bad applications that don't unmap memory + unmap(nullptr); + } else { + // Delay CPU address unmap until memRef_ destruction + assert(memRef_->cpuAddress_ == nullptr && "Memref shouldn't have a valid CPU address"); + memRef_->cpuAddress_ = address_; } } - } else { - renames_[curRename_]->cpuAddress_ = 0; - for (size_t i = 0; i < renames_.size(); ++i) { - memRef_ = renames_[i]; - // Destroy PAL resource - if (iMem() != 0) { - palFree(); - } + + // Add resource to the cache + if (!dev().resourceCache().addGpuMemory(&desc_, memRef_, subOffset_)) { + palFree(); } } @@ -1280,9 +1266,7 @@ bool Resource::partialMemCopyTo(VirtualGPU& gpu, const amd::Coord3D& srcOrigin, GpuEvent event; EngineType activeEngineID = gpu.engineID_; static const bool waitOnBusyEngine = true; - assert(!(desc().cardMemory_ && dstResource.desc().cardMemory_) && "Unsupported configuraiton!"); - uint64_t gpuMemoryOffset = 0; uint64_t gpuMemoryRowPitch = 0; uint64_t imageOffsetx = 0; @@ -1714,16 +1698,6 @@ void* Resource::map(VirtualGPU* gpu, uint flags, uint startLayer, uint numLayers if (flags & WriteOnly) { } - // Check if use map discard - if (flags & Discard) { - if (gpu != nullptr) { - // If we use a new renamed allocation, then skip the wait - if (rename(*gpu)) { - flags |= NoWait; - } - } - } - // Check if we have to wait if (!(flags & NoWait)) { if (gpu != nullptr) { @@ -1803,116 +1777,6 @@ void Resource::unmapLayers(VirtualGPU* gpu) { Unimplemented(); } -// ================================================================================================ -void Resource::setActiveRename(VirtualGPU& gpu, GpuMemoryReference* rename) { - // Copy the unique GSL data - memRef_ = rename; - address_ = rename->cpuAddress_; -} - -// ================================================================================================ -bool Resource::getActiveRename(VirtualGPU& gpu, GpuMemoryReference** rename) { - // Copy the old data to the rename descriptor - *rename = memRef_; - return true; -} - -// ================================================================================================ -bool Resource::rename(VirtualGPU& gpu, bool force) { - GpuEvent* gpuEvent = getGpuEvent(gpu); - if (!gpuEvent->isValid() && !force) { - return true; - } - - bool useNext = false; - uint resSize = desc().width_ * ((desc().height_) ? desc().height_ : 1) * elementSize_; - - // Rename will work with real GSL resources - if (((memoryType() != Local) && (memoryType() != Persistent) && (memoryType() != Remote) && - (memoryType() != RemoteUSWC)) || - (dev().settings().maxRenames_ == 0)) { - return false; - } - - // If the resource for renaming is too big, then lets check the current status first - // at the cost of an extra flush - if (resSize >= (dev().settings().maxRenameSize_ / dev().settings().maxRenames_)) { - if (gpu.isDone(gpuEvent)) { - return true; - } - } - - // Save the first - if (renames_.size() == 0) { - GpuMemoryReference* rename; - if (mapCount_ > 0) { - memRef_->cpuAddress_ = address_; - } - if (!getActiveRename(gpu, &rename)) { - return false; - } - - curRename_ = renames_.size(); - renames_.push_back(rename); - } - - // Can we use a new rename? - if ((renames_.size() <= dev().settings().maxRenames_) && - ((renames_.size() * resSize) <= dev().settings().maxRenameSize_)) { - GpuMemoryReference* rename; - - // Create a new GSL allocation - if (create(memoryType())) { - if (mapCount_ > 0) { - assert(!desc().cardMemory_ && "Unsupported memory type!"); - memRef_->cpuAddress_ = gpuMemoryMap(&desc_.pitch_, 0, iMem()); - if (memRef_->cpuAddress_ == nullptr) { - LogError("gslMap fails on rename!"); - } - address_ = memRef_->cpuAddress_; - } - if (getActiveRename(gpu, &rename)) { - curRename_ = renames_.size(); - renames_.push_back(rename); - } else { - memRef_->release(); - useNext = true; - } - } else { - useNext = true; - } - } else { - useNext = true; - } - - if (useNext) { - // Get the last submitted - curRename_++; - if (curRename_ >= renames_.size()) { - curRename_ = 0; - } - setActiveRename(gpu, renames_[curRename_]); - return false; - } - - return true; -} - -// ================================================================================================ -void Resource::warmUpRenames(VirtualGPU& gpu) { - // Make sure OCL touches every command buffer in the queue to avoid delays on the first submit - uint flush = dev().settings().maxRenames_ / VirtualGPU::Queue::MaxCmdBuffers; - flush = (flush == 0) ? 1 : flush; - for (uint i = 1; i <= dev().settings().maxRenames_; ++i) { - uint dummy = 0; - const bool Wait = (i % flush == 0) ? true : false; - // Write 0 for the buffer paging by VidMM - writeRawData(gpu, 0, sizeof(dummy), &dummy, Wait); - const bool Force = true; - rename(gpu, Force); - } -} - // ================================================================================================ MemorySubAllocator::~MemorySubAllocator() { diff --git a/projects/clr/rocclr/runtime/device/pal/palresource.hpp b/projects/clr/rocclr/runtime/device/pal/palresource.hpp index b7fd9ac341..510c802823 100644 --- a/projects/clr/rocclr/runtime/device/pal/palresource.hpp +++ b/projects/clr/rocclr/runtime/device/pal/palresource.hpp @@ -146,7 +146,6 @@ class Resource : public amd::HeapObject { //! Resource map flags enum MapFlags { - Discard = 0x00000001, //!< discard lock NoOverwrite = 0x00000002, //!< lock with no overwrite ReadOnly = 0x00000004, //!< lock for read only operation WriteOnly = 0x00000008, //!< lock for write only operation @@ -313,17 +312,13 @@ class Resource : public amd::HeapObject { size_t slicePitch = 0 //!< Raw data slice pitch ); - //! Warms up the rename list for this resource - void warmUpRenames(VirtualGPU& gpu); - //! Gets the resource element size uint elementSize() const { return elementSize_; } //! Get the mapped address of this resource address data() const { return reinterpret_cast
(address_); } - //! Frees all allocated PAL memories and resources, - //! associated with this objects. And also destroys all rename structures + //! Frees all allocated PAL memories and resources, associated with this objects. //! Note: doesn't destroy the object itself void free(); @@ -401,23 +396,6 @@ class Resource : public amd::HeapObject { //! Disable operator= Resource& operator=(const Resource&); - typedef std::vector RenameList; - - //! Rename current resource - bool rename(VirtualGPU& gpu, //!< Virtual GPU device object - bool force = false //!< Force renaming - ); - - //! Sets the rename as active - void setActiveRename(VirtualGPU& gpu, //!< Virtual GPU device object - GpuMemoryReference* rename //!< new active rename - ); - - //! Gets the active rename - bool getActiveRename(VirtualGPU& gpu, //!< Virtual GPU device object - GpuMemoryReference** rename //!< Saved active rename - ); - /*! \brief Locks the resource with layers and returns a physical pointer * * \return Pointer to the physical memory @@ -452,8 +430,6 @@ class Resource : public amd::HeapObject { amd::Atomic mapCount_; //!< Total number of maps void* address_; //!< Physical address of this resource size_t offset_; //!< Resource offset - uint32_t curRename_; //!< Current active rename in the list - RenameList renames_; //!< Rename resource list GpuMemoryReference* memRef_; //!< PAL resource reference Pal::gpusize subOffset_; //!< GPU memory offset in the oririnal resource const Resource* viewOwner_; //!< GPU resource, which owns this view diff --git a/projects/clr/rocclr/runtime/device/pal/palsettings.cpp b/projects/clr/rocclr/runtime/device/pal/palsettings.cpp index 2746439ef8..4a69a13cf1 100644 --- a/projects/clr/rocclr/runtime/device/pal/palsettings.cpp +++ b/projects/clr/rocclr/runtime/device/pal/palsettings.cpp @@ -47,9 +47,6 @@ Settings::Settings() { // By Default persistent writes will be disabled. stagingWritePersistent_ = GPU_STAGING_WRITE_PERSISTENT; - maxRenames_ = 4; - maxRenameSize_ = 4 * Mi; - imageSupport_ = false; hwLDSSize_ = 0; diff --git a/projects/clr/rocclr/runtime/device/pal/palsettings.hpp b/projects/clr/rocclr/runtime/device/pal/palsettings.hpp index 1cdad9c095..90dc533bf6 100644 --- a/projects/clr/rocclr/runtime/device/pal/palsettings.hpp +++ b/projects/clr/rocclr/runtime/device/pal/palsettings.hpp @@ -73,8 +73,6 @@ class Settings : public device::Settings { uint oclVersion_; //!< Reported OpenCL version support uint debugFlags_; //!< Debug GPU flags - uint maxRenames_; //!< Maximum number of possible renames - uint maxRenameSize_; //!< Maximum size for all renames uint hwLDSSize_; //!< HW local data store size uint maxWorkGroupSize_; //!< Requested workgroup size for this device uint preferredWorkGroupSize_;//!< Requested preferred workgroup size for this device