P4 to Git Change 1599699 by gandryey@gera-w8 on 2018/08/29 18:43:02
SWDEV-79445 - OCL generic changes and code clean-up
- Move WaveLimiter logic to the abstract layer. PAL version was taken as the base, thus performance of GSL path can be affected by this change
Affected files ...
... //depot/stg/opencl/drivers/opencl/runtime/device/device.hpp#315 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/devkernel.cpp#4 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/devkernel.hpp#4 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/devwavelimiter.cpp#1 move/add
... //depot/stg/opencl/drivers/opencl/runtime/device/devwavelimiter.hpp#1 move/add
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpudevice.cpp#598 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpukernel.cpp#331 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpukernel.hpp#133 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuwavelimiter.cpp#15 delete
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuwavelimiter.hpp#11 delete
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/paldevice.cpp#107 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.cpp#64 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.hpp#23 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palwavelimiter.cpp#8 move/delete
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palwavelimiter.hpp#8 move/delete
[ROCm/clr commit: 6cc75de90f]
This commit is contained in:
@@ -7,6 +7,7 @@
|
||||
#include "platform/context.hpp"
|
||||
#include "platform/object.hpp"
|
||||
#include "platform/memory.hpp"
|
||||
#include "devwavelimiter.hpp"
|
||||
|
||||
#if defined(WITH_LIGHTNING_COMPILER)
|
||||
namespace llvm {
|
||||
@@ -37,10 +38,6 @@ class Device;
|
||||
class KernelSignature;
|
||||
class NDRange;
|
||||
|
||||
struct ProfilingCallback : public amd::HeapObject {
|
||||
virtual void callback(ulong duration, uint32_t waves) = 0;
|
||||
};
|
||||
|
||||
struct KernelParameterDescriptor {
|
||||
enum {
|
||||
Value = 0,
|
||||
@@ -124,39 +121,7 @@ class Kernel : public amd::HeapObject {
|
||||
};
|
||||
|
||||
//! Default constructor
|
||||
Kernel(const amd::Device& dev, const std::string& name)
|
||||
: dev_(dev)
|
||||
, name_(name)
|
||||
, signature_(nullptr) {
|
||||
// Instead of memset(&workGroupInfo_, '\0', sizeof(workGroupInfo_));
|
||||
// Due to std::string not being able to be memset to 0
|
||||
workGroupInfo_.size_ = 0;
|
||||
workGroupInfo_.compileSize_[0] = 0;
|
||||
workGroupInfo_.compileSize_[1] = 0;
|
||||
workGroupInfo_.compileSize_[2] = 0;
|
||||
workGroupInfo_.localMemSize_ = 0;
|
||||
workGroupInfo_.preferredSizeMultiple_ = 0;
|
||||
workGroupInfo_.privateMemSize_ = 0;
|
||||
workGroupInfo_.scratchRegs_ = 0;
|
||||
workGroupInfo_.wavefrontPerSIMD_ = 0;
|
||||
workGroupInfo_.wavefrontSize_ = 0;
|
||||
workGroupInfo_.availableGPRs_ = 0;
|
||||
workGroupInfo_.usedGPRs_ = 0;
|
||||
workGroupInfo_.availableSGPRs_ = 0;
|
||||
workGroupInfo_.usedSGPRs_ = 0;
|
||||
workGroupInfo_.availableVGPRs_ = 0;
|
||||
workGroupInfo_.usedVGPRs_ = 0;
|
||||
workGroupInfo_.availableLDSSize_ = 0;
|
||||
workGroupInfo_.usedLDSSize_ = 0;
|
||||
workGroupInfo_.availableStackSize_ = 0;
|
||||
workGroupInfo_.usedStackSize_ = 0;
|
||||
workGroupInfo_.compileSizeHint_[0] = 0;
|
||||
workGroupInfo_.compileSizeHint_[1] = 0;
|
||||
workGroupInfo_.compileSizeHint_[2] = 0;
|
||||
workGroupInfo_.compileVecTypeHint_ = "";
|
||||
workGroupInfo_.uniformWorkGroupSize_ = false;
|
||||
workGroupInfo_.wavesPerSimdHint_ = 0;
|
||||
}
|
||||
Kernel(const amd::Device& dev, const std::string& name);
|
||||
|
||||
//! Default destructor
|
||||
virtual ~Kernel();
|
||||
@@ -196,13 +161,14 @@ class Kernel : public amd::HeapObject {
|
||||
size_t getWorkGroupSizeHint(int dim) const { return workGroupInfo_.compileSizeHint_[dim]; }
|
||||
|
||||
//! Get profiling callback object
|
||||
virtual amd::ProfilingCallback* getProfilingCallback(const device::VirtualDevice* vdv) {
|
||||
return nullptr;
|
||||
}
|
||||
amd::ProfilingCallback* getProfilingCallback(const device::VirtualDevice* vdev) {
|
||||
return waveLimiter_.getProfilingCallback(vdev);
|
||||
};
|
||||
|
||||
virtual uint getWavesPerSH(const device::VirtualDevice* vdv) const {
|
||||
return 0;
|
||||
}
|
||||
//! Get waves per shader array to be used for kernel execution.
|
||||
uint getWavesPerSH(const device::VirtualDevice* vdev) const {
|
||||
return waveLimiter_.getWavesPerSH(vdev);
|
||||
};
|
||||
|
||||
//! Returns GPU device object, associated with this kernel
|
||||
const amd::Device& dev() const { return dev_; }
|
||||
@@ -272,6 +238,7 @@ class Kernel : public amd::HeapObject {
|
||||
amd::KernelSignature* signature_; //!< kernel signature
|
||||
std::string buildLog_; //!< build log
|
||||
std::vector<PrintfInfo> printf_; //!< Format strings for GPU printf support
|
||||
WaveLimiterManager waveLimiter_; //!< adaptively control number of waves
|
||||
|
||||
union Flags {
|
||||
struct {
|
||||
|
||||
Reference in New Issue
Block a user