P4 to Git Change 1250684 by gandryey@gera-w8 on 2016/03/23 17:59:05
SWDEV-86035 - Add PAL backend to OpenCL - Update PAL backend to match the latests PAL interfaces Affected files ... ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/Makefile#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/build/Makefile.pal#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palbe/build/Makefile#1 add ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palbe/build/Makefile.palbe#1 add ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palblit.cpp#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/paldevice.cpp#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/paldevice.hpp#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.cpp#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.hpp#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palprogram.cpp#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palprogram.hpp#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.cpp#2 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.hpp#2 edit
This commit is contained in:
@@ -937,6 +937,8 @@ KernelBlitManager::copyBufferToImage(
|
|||||||
static const bool CopyRect = false;
|
static const bool CopyRect = false;
|
||||||
// Flush DMA for ASYNC copy
|
// Flush DMA for ASYNC copy
|
||||||
static const bool FlushDMA = true;
|
static const bool FlushDMA = true;
|
||||||
|
size_t imgRowPitch = size[0] * gpuMem(dstMemory).elementSize();
|
||||||
|
size_t imgSlicePitch = imgRowPitch * size[1];
|
||||||
|
|
||||||
if (setup_.disableCopyBufferToImage_) {
|
if (setup_.disableCopyBufferToImage_) {
|
||||||
result = DmaBlitManager::copyBufferToImage(
|
result = DmaBlitManager::copyBufferToImage(
|
||||||
@@ -947,7 +949,9 @@ KernelBlitManager::copyBufferToImage(
|
|||||||
}
|
}
|
||||||
// Check if buffer is in system memory with direct access
|
// Check if buffer is in system memory with direct access
|
||||||
else if (gpuMem(srcMemory).isHostMemDirectAccess() &&
|
else if (gpuMem(srcMemory).isHostMemDirectAccess() &&
|
||||||
(rowPitch == 0) && (slicePitch == 0)) {
|
(((rowPitch == 0) && (slicePitch == 0)) ||
|
||||||
|
((rowPitch == imgRowPitch) &&
|
||||||
|
((slicePitch == 0) || (slicePitch == imgSlicePitch))))) {
|
||||||
// First attempt to do this all with DMA,
|
// First attempt to do this all with DMA,
|
||||||
// but there are restriciton with older hardware
|
// but there are restriciton with older hardware
|
||||||
if (dev().settings().imageDMA_) {
|
if (dev().settings().imageDMA_) {
|
||||||
@@ -1327,6 +1331,8 @@ KernelBlitManager::copyImageToBuffer(
|
|||||||
static const bool CopyRect = false;
|
static const bool CopyRect = false;
|
||||||
// Flush DMA for ASYNC copy
|
// Flush DMA for ASYNC copy
|
||||||
static const bool FlushDMA = true;
|
static const bool FlushDMA = true;
|
||||||
|
size_t imgRowPitch = size[0] * gpuMem(srcMemory).elementSize();
|
||||||
|
size_t imgSlicePitch = imgRowPitch * size[1];
|
||||||
|
|
||||||
if (setup_.disableCopyImageToBuffer_) {
|
if (setup_.disableCopyImageToBuffer_) {
|
||||||
result = HostBlitManager::copyImageToBuffer(
|
result = HostBlitManager::copyImageToBuffer(
|
||||||
@@ -1337,7 +1343,9 @@ KernelBlitManager::copyImageToBuffer(
|
|||||||
}
|
}
|
||||||
// Check if buffer is in system memory with direct access
|
// Check if buffer is in system memory with direct access
|
||||||
else if (gpuMem(dstMemory).isHostMemDirectAccess() &&
|
else if (gpuMem(dstMemory).isHostMemDirectAccess() &&
|
||||||
(rowPitch == 0) && (slicePitch == 0)) {
|
(((rowPitch == 0) && (slicePitch == 0)) ||
|
||||||
|
((rowPitch == imgRowPitch) &&
|
||||||
|
((slicePitch == 0) || (slicePitch == imgSlicePitch))))) {
|
||||||
// First attempt to do this all with DMA,
|
// First attempt to do this all with DMA,
|
||||||
// but there are restriciton with older hardware
|
// but there are restriciton with older hardware
|
||||||
if (dev().settings().imageDMA_) {
|
if (dev().settings().imageDMA_) {
|
||||||
|
|||||||
@@ -175,10 +175,10 @@ void NullDevice::fillDeviceInfo(
|
|||||||
|
|
||||||
info_.maxWorkItemDimensions_ = 3;
|
info_.maxWorkItemDimensions_ = 3;
|
||||||
info_.maxComputeUnits_ =
|
info_.maxComputeUnits_ =
|
||||||
palProp.gfxipProperties.engineCore.numOfShaderEngines *
|
palProp.gfxipProperties.shaderCore.numShaderEngines *
|
||||||
palProp.gfxipProperties.engineCore.numOfShaderArrays *
|
palProp.gfxipProperties.shaderCore.numShaderArrays *
|
||||||
palProp.gfxipProperties.engineCore.numOfCUsPerShaderArray;
|
palProp.gfxipProperties.shaderCore.numCusPerShaderArray;
|
||||||
info_.numberOfShaderEngines = palProp.gfxipProperties.engineCore.numOfShaderEngines;
|
info_.numberOfShaderEngines = palProp.gfxipProperties.shaderCore.numShaderEngines;
|
||||||
|
|
||||||
// SI parts are scalar. Also, reads don't need to be 128-bits to get peak rates.
|
// SI parts are scalar. Also, reads don't need to be 128-bits to get peak rates.
|
||||||
// For example, float4 is not faster than float as long as all threads fetch the same
|
// For example, float4 is not faster than float as long as all threads fetch the same
|
||||||
@@ -417,7 +417,7 @@ void NullDevice::fillDeviceInfo(
|
|||||||
info_.simdPerCU_ = hwInfo()->simdPerCU_;
|
info_.simdPerCU_ = hwInfo()->simdPerCU_;
|
||||||
info_.simdWidth_ = hwInfo()->simdWidth_;
|
info_.simdWidth_ = hwInfo()->simdWidth_;
|
||||||
info_.simdInstructionWidth_ = hwInfo()->simdInstructionWidth_;
|
info_.simdInstructionWidth_ = hwInfo()->simdInstructionWidth_;
|
||||||
info_.wavefrontWidth_ = palProp.gfxipProperties.engineCore.wavefrontSize;
|
info_.wavefrontWidth_ = palProp.gfxipProperties.shaderCore.wavefrontSize;
|
||||||
//info_.globalMemChannels_ = calAttr.memBusWidth / 32;
|
//info_.globalMemChannels_ = calAttr.memBusWidth / 32;
|
||||||
//info_.globalMemChannelBanks_ = calAttr.numMemBanks;
|
//info_.globalMemChannelBanks_ = calAttr.numMemBanks;
|
||||||
info_.globalMemChannelBankWidth_ = hwInfo()->memChannelBankWidth_;
|
info_.globalMemChannelBankWidth_ = hwInfo()->memChannelBankWidth_;
|
||||||
@@ -1541,35 +1541,34 @@ Device::createView(amd::Memory& owner, const device::Memory& parent) const
|
|||||||
|
|
||||||
//! Attempt to bind with external graphics API's device/context
|
//! Attempt to bind with external graphics API's device/context
|
||||||
bool
|
bool
|
||||||
Device::bindExternalDevice(intptr_t type, void* pDevice, void* pContext, bool validateOnly)
|
Device::bindExternalDevice(uint flags, void* pDevice, void* pContext, bool validateOnly)
|
||||||
{
|
{
|
||||||
assert(pDevice);
|
assert(pDevice);
|
||||||
|
|
||||||
switch (type) {
|
|
||||||
#ifdef _WIN32
|
#ifdef _WIN32
|
||||||
case CL_CONTEXT_D3D10_DEVICE_KHR:
|
if (flags & amd::Context::Flags::D3D10DeviceKhr) {
|
||||||
if (!associateD3D10Device(pDevice)) {
|
if (!associateD3D10Device(pDevice)) {
|
||||||
LogError("Failed gslD3D10Associate()");
|
LogError("Failed gslD3D10Associate()");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
break;
|
}
|
||||||
case CL_CONTEXT_D3D11_DEVICE_KHR:
|
else if (flags & amd::Context::Flags::D3D11DeviceKhr) {
|
||||||
if (!associateD3D11Device(pDevice)) {
|
if (!associateD3D11Device(pDevice)) {
|
||||||
LogError("Failed gslD3D11Associate()");
|
LogError("Failed gslD3D11Associate()");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
break;
|
}
|
||||||
case CL_CONTEXT_ADAPTER_D3D9_KHR:
|
else if (flags & (amd::Context::Flags::D3D9DeviceKhr |
|
||||||
case CL_CONTEXT_ADAPTER_D3D9EX_KHR:
|
amd::Context::Flags::D3D9DeviceEXKhr)) {
|
||||||
if (!associateD3D9Device(pDevice)) {
|
if (!associateD3D9Device(pDevice)) {
|
||||||
LogWarning("D3D9<->OpenCL adapter mismatch or D3D9Associate() failure");
|
LogWarning("D3D9<->OpenCL adapter mismatch or D3D9Associate() failure");
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
break;
|
}
|
||||||
case CL_CONTEXT_ADAPTER_DXVA_KHR:
|
else if (flags & amd::Context::Flags::D3D9DeviceVAKhr) {
|
||||||
break;
|
}
|
||||||
#endif //_WIN32
|
#endif //_WIN32
|
||||||
case CL_GL_CONTEXT_KHR:
|
if (flags & amd::Context::Flags::GLDeviceKhr) {
|
||||||
// Attempt to associate GSL-OGL
|
// Attempt to associate GSL-OGL
|
||||||
if (!glAssociate(pContext, pDevice)) {
|
if (!glAssociate(pContext, pDevice)) {
|
||||||
if (!validateOnly) {
|
if (!validateOnly) {
|
||||||
@@ -1577,20 +1576,15 @@ Device::bindExternalDevice(intptr_t type, void* pDevice, void* pContext, bool va
|
|||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
break;
|
|
||||||
default:
|
|
||||||
LogError("Unknown external device!");
|
|
||||||
return false;
|
|
||||||
break;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool
|
bool
|
||||||
Device::unbindExternalDevice(intptr_t type, void* pDevice, void* pContext, bool validateOnly)
|
Device::unbindExternalDevice(uint flags, void* pDevice, void* pContext, bool validateOnly)
|
||||||
{
|
{
|
||||||
if (type != CL_GL_CONTEXT_KHR) {
|
if ((flags & amd::Context::Flags::GLDeviceKhr) == 0) {
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1820,8 +1814,8 @@ Device::allocScratch(uint regNum, const VirtualGPU* vgpu)
|
|||||||
// Calculate the size of the scratch buffer for a queue
|
// Calculate the size of the scratch buffer for a queue
|
||||||
uint32_t numTotalCUs = info().maxComputeUnits_;
|
uint32_t numTotalCUs = info().maxComputeUnits_;
|
||||||
uint32_t numMaxWaves =
|
uint32_t numMaxWaves =
|
||||||
properties().gfxipProperties.engineCore.maxScratchWavesPerCU * numTotalCUs;
|
properties().gfxipProperties.shaderCore.maxScratchWavesPerCu * numTotalCUs;
|
||||||
scratchBuf->size_ = properties().gfxipProperties.engineCore.wavefrontSize *
|
scratchBuf->size_ = properties().gfxipProperties.shaderCore.wavefrontSize *
|
||||||
scratchBuf->regNum_ * numMaxWaves * sizeof(uint32_t);
|
scratchBuf->regNum_ * numMaxWaves * sizeof(uint32_t);
|
||||||
scratchBuf->size_ = amd::alignUp(scratchBuf->size_, 0xFFFF);
|
scratchBuf->size_ = amd::alignUp(scratchBuf->size_, 0xFFFF);
|
||||||
scratchBuf->offset_ = offset;
|
scratchBuf->offset_ = offset;
|
||||||
@@ -1920,8 +1914,7 @@ Device::fillHwSampler(
|
|||||||
|
|
||||||
samplerInfo.borderColorType = Pal::BorderColorType::TransparentBlack;
|
samplerInfo.borderColorType = Pal::BorderColorType::TransparentBlack;
|
||||||
|
|
||||||
// Assign defaults
|
samplerInfo.filter.zFilter = Pal::XyFilterPoint;
|
||||||
samplerInfo.filter = Pal::TexFilter::MagPointMinPointMipBase;
|
|
||||||
|
|
||||||
samplerInfo.flags.unnormalizedCoords = !(state & amd::Sampler::StateNormalizedCoordsMask);
|
samplerInfo.flags.unnormalizedCoords = !(state & amd::Sampler::StateNormalizedCoordsMask);
|
||||||
|
|
||||||
@@ -1956,24 +1949,16 @@ Device::fillHwSampler(
|
|||||||
|
|
||||||
// Program texture filter mode
|
// Program texture filter mode
|
||||||
if (state == amd::Sampler::StateFilterLinear) {
|
if (state == amd::Sampler::StateFilterLinear) {
|
||||||
samplerInfo.filter = Pal::TexFilter::MagLinearMinLinearMipBase;
|
samplerInfo.filter.magnification = Pal::XyFilterLinear;
|
||||||
|
samplerInfo.filter.minification = Pal::XyFilterLinear;
|
||||||
|
samplerInfo.filter.zFilter = Pal::ZFilterLinear;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (mipFilter == CL_FILTER_NEAREST) {
|
if (mipFilter == CL_FILTER_NEAREST) {
|
||||||
if (state == amd::Sampler::StateFilterLinear) {
|
samplerInfo.filter.mipFilter = Pal::MipFilterPoint;
|
||||||
samplerInfo.filter = Pal::TexFilter::MagLinearMinLinearMipPoint;
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
samplerInfo.filter = Pal::TexFilter::MagPointMinPointMipPoint;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
else if (mipFilter == CL_FILTER_LINEAR) {
|
else if (mipFilter == CL_FILTER_LINEAR) {
|
||||||
if (state == amd::Sampler::StateFilterLinear) {
|
samplerInfo.filter.mipFilter = Pal::MipFilterLinear;
|
||||||
samplerInfo.filter = Pal::TexFilter::MagLinearMinLinearMipLinear;
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
samplerInfo.filter = Pal::TexFilter::MagPointMinPointMipLinear;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
iDev()->CreateSamplerSrds(1, &samplerInfo, hwState);
|
iDev()->CreateSamplerSrds(1, &samplerInfo, hwState);
|
||||||
|
|||||||
@@ -91,10 +91,10 @@ public:
|
|||||||
//! Needed for OpenGL objects on CPU device
|
//! Needed for OpenGL objects on CPU device
|
||||||
|
|
||||||
virtual bool bindExternalDevice(
|
virtual bool bindExternalDevice(
|
||||||
intptr_t type, void* pDevice, void* pContext, bool validateOnly) { return true; }
|
uint flags, void* pDevice, void* pContext, bool validateOnly) { return true; }
|
||||||
|
|
||||||
virtual bool unbindExternalDevice(
|
virtual bool unbindExternalDevice(
|
||||||
intptr_t type, void* pDevice, void* pContext, bool validateOnly) { return true; }
|
uint flags, void* pDevice, void* pContext, bool validateOnly) { return true; }
|
||||||
|
|
||||||
//! Releases non-blocking map target memory
|
//! Releases non-blocking map target memory
|
||||||
virtual void freeMapTarget(amd::Memory& mem, void* target) {}
|
virtual void freeMapTarget(amd::Memory& mem, void* target) {}
|
||||||
@@ -369,17 +369,11 @@ public:
|
|||||||
|
|
||||||
//! Attempt to bind with external graphics API's device/context
|
//! Attempt to bind with external graphics API's device/context
|
||||||
virtual bool bindExternalDevice(
|
virtual bool bindExternalDevice(
|
||||||
intptr_t type,
|
uint flags, void* pDevice, void* pContext, bool validateOnly);
|
||||||
void* pDevice,
|
|
||||||
void* pContext,
|
|
||||||
bool validateOnly);
|
|
||||||
|
|
||||||
//! Attempt to unbind with external graphics API's device/context
|
//! Attempt to unbind with external graphics API's device/context
|
||||||
virtual bool unbindExternalDevice(
|
virtual bool unbindExternalDevice(
|
||||||
intptr_t type,
|
uint flags, void* pDevice, void* pContext, bool validateOnly);
|
||||||
void* pDevice,
|
|
||||||
void* pContext,
|
|
||||||
bool validateOnly);
|
|
||||||
|
|
||||||
//! Validates kernel before execution
|
//! Validates kernel before execution
|
||||||
virtual bool validateKernel(
|
virtual bool validateKernel(
|
||||||
|
|||||||
@@ -387,40 +387,49 @@ HSAILKernel::aqlCreateHWInfo(amd::hsa::loader::Symbol *sym)
|
|||||||
if (!sym->GetInfo(HSA_EXT_EXECUTABLE_SYMBOL_INFO_KERNEL_OBJECT_ALIGN, reinterpret_cast<void*>(&akc_align))) {
|
if (!sym->GetInfo(HSA_EXT_EXECUTABLE_SYMBOL_INFO_KERNEL_OBJECT_ALIGN, reinterpret_cast<void*>(&akc_align))) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
code_ = new Memory(dev(), amd::alignUp(codeSize_, akc_align));
|
// Allocate HW resources for the real program only
|
||||||
Resource::MemoryType type = Resource::RemoteUSWC;
|
if (!prog().isNull()) {
|
||||||
if (flags_.internalKernel_) {
|
code_ = new Memory(dev(), amd::alignUp(codeSize_, akc_align));
|
||||||
type = Resource::RemoteUSWC;
|
Resource::MemoryType type = Resource::RemoteUSWC;
|
||||||
}
|
if (flags_.internalKernel_) {
|
||||||
// Initialize kernel ISA code
|
type = Resource::RemoteUSWC;
|
||||||
if (code_ && code_->create(type)) {
|
}
|
||||||
address cpuCodePtr = static_cast<address>(code_->map(nullptr, Resource::WriteOnly));
|
// Initialize kernel ISA code
|
||||||
// Copy only amd_kernel_code_t
|
if (code_ && code_->create(type)) {
|
||||||
memcpy(cpuCodePtr, reinterpret_cast<address>(akc), codeSize_);
|
address cpuCodePtr = static_cast<address>(code_->map(nullptr, Resource::WriteOnly));
|
||||||
code_->unmap(nullptr);
|
// Copy only amd_kernel_code_t
|
||||||
}
|
memcpy(cpuCodePtr, reinterpret_cast<address>(akc), codeSize_);
|
||||||
else {
|
code_->unmap(nullptr);
|
||||||
LogError("Failed to allocate ISA code!");
|
}
|
||||||
return false;
|
else {
|
||||||
|
LogError("Failed to allocate ISA code!");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
assert((akc->workitem_private_segment_byte_size & 3) == 0 &&
|
assert((akc->workitem_private_segment_byte_size & 3) == 0 &&
|
||||||
"Scratch must be DWORD aligned");
|
"Scratch must be DWORD aligned");
|
||||||
workGroupInfo_.scratchRegs_ =
|
workGroupInfo_.scratchRegs_ =
|
||||||
amd::alignUp(akc->workitem_private_segment_byte_size, 16) / sizeof(uint);
|
amd::alignUp(akc->workitem_private_segment_byte_size, 16) / sizeof(uint);
|
||||||
/*
|
|
||||||
workGroupInfo_.availableSGPRs_ = dev().gslCtx()->getNumSGPRsAvailable();
|
|
||||||
workGroupInfo_.availableVGPRs_ = dev().gslCtx()->getNumVGPRsAvailable();
|
|
||||||
workGroupInfo_.preferredSizeMultiple_ = dev().getAttribs().wavefrontSize;
|
|
||||||
workGroupInfo_.wavefrontPerSIMD_ = dev().getAttribs().wavefrontSize;
|
|
||||||
*/
|
|
||||||
workGroupInfo_.privateMemSize_ = akc->workitem_private_segment_byte_size;
|
workGroupInfo_.privateMemSize_ = akc->workitem_private_segment_byte_size;
|
||||||
workGroupInfo_.localMemSize_ =
|
workGroupInfo_.localMemSize_ =
|
||||||
workGroupInfo_.usedLDSSize_ = akc->workgroup_group_segment_byte_size;
|
workGroupInfo_.usedLDSSize_ = akc->workgroup_group_segment_byte_size;
|
||||||
workGroupInfo_.usedSGPRs_ = akc->wavefront_sgpr_count;
|
workGroupInfo_.usedSGPRs_ = akc->wavefront_sgpr_count;
|
||||||
workGroupInfo_.usedStackSize_ = 0;
|
workGroupInfo_.usedStackSize_ = 0;
|
||||||
workGroupInfo_.usedVGPRs_ = akc->workitem_vgpr_count;
|
workGroupInfo_.usedVGPRs_ = akc->workitem_vgpr_count;
|
||||||
|
|
||||||
|
if (!prog().isNull()) {
|
||||||
|
workGroupInfo_.availableSGPRs_ = dev().properties().gfxipProperties.shaderCore.numAvailableSgprs;
|
||||||
|
workGroupInfo_.availableVGPRs_ = dev().properties().gfxipProperties.shaderCore.numAvailableVgprs;
|
||||||
|
workGroupInfo_.preferredSizeMultiple_ =
|
||||||
|
workGroupInfo_.wavefrontPerSIMD_ = dev().properties().gfxipProperties.shaderCore.wavefrontSize;
|
||||||
|
}
|
||||||
|
else {
|
||||||
|
workGroupInfo_.availableSGPRs_ = 104;
|
||||||
|
workGroupInfo_.availableVGPRs_ = 256;
|
||||||
|
workGroupInfo_.preferredSizeMultiple_ =
|
||||||
|
workGroupInfo_.wavefrontPerSIMD_ = 64;
|
||||||
|
}
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -633,10 +642,7 @@ HSAILKernel::init(amd::hsa::loader::Symbol *sym, bool finalize)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Allocate HW resources for the real program only
|
aqlCreateHWInfo(sym);
|
||||||
if (!prog().isNull()) {
|
|
||||||
aqlCreateHWInfo(sym);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Pull out metadata from the ELF
|
// Pull out metadata from the ELF
|
||||||
size_t sizeOfArgList;
|
size_t sizeOfArgList;
|
||||||
|
|||||||
@@ -153,7 +153,7 @@ public:
|
|||||||
{ return cpuAqlCode_->workgroup_group_segment_byte_size; }
|
{ return cpuAqlCode_->workgroup_group_segment_byte_size; }
|
||||||
|
|
||||||
//! Returns pointer on CPU to AQL code info
|
//! Returns pointer on CPU to AQL code info
|
||||||
const void* cpuAqlCode() const { return cpuAqlCode_; }
|
const amd_kernel_code_t* cpuAqlCode() const { return cpuAqlCode_; }
|
||||||
|
|
||||||
//! Returns memory object with AQL code
|
//! Returns memory object with AQL code
|
||||||
pal::Memory* gpuAqlCode() const { return code_; }
|
pal::Memory* gpuAqlCode() const { return code_; }
|
||||||
|
|||||||
@@ -505,7 +505,7 @@ HSAILProgram::linkImpl(amd::option::Options* options)
|
|||||||
hsa_agent_t agent;
|
hsa_agent_t agent;
|
||||||
agent.handle = 1;
|
agent.handle = 1;
|
||||||
if (!isNull() && hsaLoad) {
|
if (!isNull() && hsaLoad) {
|
||||||
executable_ = loader_->CreateExecutable(HSA_PROFILE_BASE, nullptr);
|
executable_ = loader_->CreateExecutable(HSA_PROFILE_FULL, NULL);
|
||||||
if (executable_ == nullptr) {
|
if (executable_ == nullptr) {
|
||||||
buildLog_ += "Error: Executable for AMD HSA Code Object isn't created.\n";
|
buildLog_ += "Error: Executable for AMD HSA Code Object isn't created.\n";
|
||||||
return false;
|
return false;
|
||||||
|
|||||||
@@ -55,6 +55,11 @@ public:
|
|||||||
void* SegmentAddress(amdgpu_hsa_elf_segment_t segment,
|
void* SegmentAddress(amdgpu_hsa_elf_segment_t segment,
|
||||||
hsa_agent_t agent, void* seg, size_t offset) override;
|
hsa_agent_t agent, void* seg, size_t offset) override;
|
||||||
|
|
||||||
|
void* SegmentHostAddress(amdgpu_hsa_elf_segment_t segment,
|
||||||
|
hsa_agent_t agent, void* seg, size_t offset) override {
|
||||||
|
return nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
bool SegmentFreeze(amdgpu_hsa_elf_segment_t segment,
|
bool SegmentFreeze(amdgpu_hsa_elf_segment_t segment,
|
||||||
hsa_agent_t agent, void* seg, size_t size) override { return false; }
|
hsa_agent_t agent, void* seg, size_t size) override { return false; }
|
||||||
|
|
||||||
|
|||||||
@@ -43,6 +43,7 @@ VirtualGPU::Queue::Create(
|
|||||||
Pal::QueueCreateInfo qCreateInfo = {};
|
Pal::QueueCreateInfo qCreateInfo = {};
|
||||||
qCreateInfo.engineType = queueType;
|
qCreateInfo.engineType = queueType;
|
||||||
qCreateInfo.engineIndex = engineIdx;
|
qCreateInfo.engineIndex = engineIdx;
|
||||||
|
qCreateInfo.aqlQueue = true;
|
||||||
|
|
||||||
// Find queue object size
|
// Find queue object size
|
||||||
size_t qSize = palDev->GetQueueSize(qCreateInfo, &result);
|
size_t qSize = palDev->GetQueueSize(qCreateInfo, &result);
|
||||||
@@ -181,8 +182,10 @@ VirtualGPU::Queue::flush()
|
|||||||
memRef.push_back(it->first);
|
memRef.push_back(it->first);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (memRef.size() != 0) {
|
if (memRef.size() != 0) {
|
||||||
iDev_->AddGpuMemoryReferences(memRef.size(), &memRef[0], iQueue_);
|
iDev_->AddGpuMemoryReferences(memRef.size(), &memRef[0], iQueue_,
|
||||||
|
Pal::GpuMemoryRefCantTrim);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Submit command buffer to OS
|
// Submit command buffer to OS
|
||||||
@@ -1982,12 +1985,12 @@ VirtualGPU::submitKernelInternal(
|
|||||||
eventBegin(MainEngine);
|
eventBegin(MainEngine);
|
||||||
if (nullptr == scratch) {
|
if (nullptr == scratch) {
|
||||||
iCmd()->CmdDispatchAql(aqlPkt, 0, 0, 0,
|
iCmd()->CmdDispatchAql(aqlPkt, 0, 0, 0,
|
||||||
hsaKernel.cpuAqlCode(), hsaQueueMem_->vmAddress(), pKernelInfo, 0x3ff);
|
hsaKernel.cpuAqlCode(), hsaQueueMem_->vmAddress(), 0x3ff);
|
||||||
}
|
}
|
||||||
else {
|
else {
|
||||||
iCmd()->CmdDispatchAql(aqlPkt, scratch->memObj_->vmAddress(),
|
iCmd()->CmdDispatchAql(aqlPkt, scratch->memObj_->vmAddress(),
|
||||||
scratch->size_, scratch->offset_,
|
scratch->size_, scratch->offset_,
|
||||||
hsaKernel.cpuAqlCode(), hsaQueueMem_->vmAddress(), pKernelInfo, 0x3ff);
|
hsaKernel.cpuAqlCode(), hsaQueueMem_->vmAddress(), 0x3ff);
|
||||||
}
|
}
|
||||||
eventEnd(MainEngine, gpuEvent);
|
eventEnd(MainEngine, gpuEvent);
|
||||||
|
|
||||||
|
|||||||
@@ -69,7 +69,8 @@ public:
|
|||||||
|
|
||||||
void addMemRef(Pal::IGpuMemory* iMem) const
|
void addMemRef(Pal::IGpuMemory* iMem) const
|
||||||
{
|
{
|
||||||
iDev_->AddGpuMemoryReferences(1, &iMem, NULL);
|
iDev_->AddGpuMemoryReferences(1, &iMem, NULL,
|
||||||
|
Pal::GpuMemoryRefCantTrim);
|
||||||
}
|
}
|
||||||
void removeMemRef(Pal::IGpuMemory* iMem) const
|
void removeMemRef(Pal::IGpuMemory* iMem) const
|
||||||
{
|
{
|
||||||
|
|||||||
Verwijs in nieuw issue
Block a user