P4 to Git Change 1568521 by gandryey@gera-w8 on 2018/06/14 17:43:52
SWDEV-79445 - OCL generic changes and code clean-up - Change LDS setup to account the size, since LC forces 4 bytes for LDS offsets always http://ocltc.amd.com/reviews/r/15197/ Affected files ... ... //depot/stg/opencl/drivers/opencl/api/opencl/amdocl/cl_execute.cpp#28 edit ... //depot/stg/opencl/drivers/opencl/api/opencl/amdocl/cl_program.cpp#49 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpublit.cpp#130 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpukernel.cpp#327 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palblit.cpp#25 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.cpp#56 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.cpp#109 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/rocm/rocblit.hpp#11 edit ... //depot/stg/opencl/drivers/opencl/runtime/device/rocm/rockernel.cpp#38 edit ... //depot/stg/opencl/drivers/opencl/runtime/platform/command.cpp#88 edit ... //depot/stg/opencl/drivers/opencl/runtime/platform/kernel.cpp#34 edit ... //depot/stg/opencl/drivers/opencl/runtime/platform/kernel.hpp#25 edit
このコミットが含まれているのは:
@@ -955,13 +955,14 @@ static void setArgument(amd::Kernel* kernel, size_t index, size_t size, const vo
|
||||
const amd::KernelParameterDescriptor& desc = kernel->signature().at(index);
|
||||
|
||||
void* param = kernel->parameters().values() + desc.offset_;
|
||||
assert((desc.type_ == T_POINTER || value != NULL || desc.size_ == 0) &&
|
||||
"not a valid local mem arg");
|
||||
assert((desc.type_ == T_POINTER || value != NULL ||
|
||||
(desc.addressQualifier_ == CL_KERNEL_ARG_ADDRESS_LOCAL)) &&
|
||||
"not a valid local mem arg");
|
||||
|
||||
uint32_t uint32_value = 0;
|
||||
uint64_t uint64_value = 0;
|
||||
|
||||
if (desc.type_ == T_POINTER && desc.size_ != 0) {
|
||||
if (desc.type_ == T_POINTER && (desc.addressQualifier_ != CL_KERNEL_ARG_ADDRESS_LOCAL)) {
|
||||
if ((value == NULL) || (static_cast<const cl_mem*>(value) == NULL)) {
|
||||
LP64_SWITCH(uint32_value, uint64_value) = 0;
|
||||
reinterpret_cast<Memory**>(kernel->parameters().values() +
|
||||
@@ -978,26 +979,24 @@ static void setArgument(amd::Kernel* kernel, size_t index, size_t size, const vo
|
||||
assert(false && "No sampler support in blit manager! Use internal samplers!");
|
||||
} else
|
||||
switch (desc.size_) {
|
||||
case 1:
|
||||
uint32_value = *static_cast<const uint8_t*>(value);
|
||||
break;
|
||||
case 2:
|
||||
uint32_value = *static_cast<const uint16_t*>(value);
|
||||
break;
|
||||
case 4:
|
||||
uint32_value = *static_cast<const uint32_t*>(value);
|
||||
if (desc.addressQualifier_ == CL_KERNEL_ARG_ADDRESS_LOCAL) {
|
||||
uint32_value = size;
|
||||
} else {
|
||||
uint32_value = *static_cast<const uint32_t*>(value);
|
||||
}
|
||||
break;
|
||||
case 8:
|
||||
uint64_value = *static_cast<const uint64_t*>(value);
|
||||
if (desc.addressQualifier_ == CL_KERNEL_ARG_ADDRESS_LOCAL) {
|
||||
uint64_value = size;
|
||||
} else {
|
||||
uint64_value = *static_cast<const uint64_t*>(value);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
switch (desc.size_) {
|
||||
case 0 /*local mem*/:
|
||||
*static_cast<size_t*>(param) = size;
|
||||
break;
|
||||
case sizeof(uint32_t):
|
||||
*static_cast<uint32_t*>(param) = uint32_value;
|
||||
break;
|
||||
|
||||
@@ -378,7 +378,7 @@ size_t KernelArg::size(bool gpuLayer) const {
|
||||
return (gpuLayer) ? 0 : sizeof(cl_mem);
|
||||
case PointerLocal:
|
||||
case PointerHwLocal:
|
||||
return (gpuLayer) ? sizeof(uint32_t) * size_ : 0;
|
||||
return (gpuLayer) ? sizeof(uint32_t) * size_ : sizeof(cl_mem);
|
||||
case PointerPrivate:
|
||||
case PointerHwPrivate:
|
||||
return (gpuLayer) ? sizeof(uint32_t) * size_ : 0;
|
||||
@@ -2991,7 +2991,7 @@ void HSAILKernel::initArgList(const aclArgData* aclArg) {
|
||||
|
||||
// Make a check if it is local or global
|
||||
if (desc.addressQualifier_ == CL_KERNEL_ARG_ADDRESS_LOCAL) {
|
||||
desc.size_ = 0;
|
||||
desc.size_ = sizeof(cl_mem);
|
||||
} else {
|
||||
desc.size_ = GetOclSize(aclArg);
|
||||
}
|
||||
@@ -3000,10 +3000,7 @@ void HSAILKernel::initArgList(const aclArgData* aclArg) {
|
||||
// in multidevice config abstraction layer has a single signature
|
||||
// and CPU sends the paramaters as they are allocated in memory
|
||||
size_t size = desc.size_;
|
||||
if (size == 0) {
|
||||
// Local memory for CPU
|
||||
size = sizeof(cl_mem);
|
||||
}
|
||||
|
||||
offset = amd::alignUp(offset, std::min(size, size_t(16)));
|
||||
desc.offset_ = offset;
|
||||
offset += amd::alignUp(size, sizeof(uint32_t));
|
||||
@@ -3527,8 +3524,12 @@ hsa_kernel_dispatch_packet_t* HSAILKernel::loadArguments(
|
||||
else {
|
||||
assert((arg->addrQual_ == HSAIL_ADDRESS_LOCAL) && "Unsupported address type");
|
||||
ldsAddress = amd::alignUp(ldsAddress, arg->alignment_);
|
||||
WriteAqlArg(&aqlArgBuf, &ldsAddress, sizeof(size_t));
|
||||
ldsAddress += *reinterpret_cast<const size_t*>(paramaddr);
|
||||
WriteAqlArg(&aqlArgBuf, &ldsAddress, desc.size_);
|
||||
if (desc.size_ == 8) {
|
||||
ldsAddress += *reinterpret_cast<const uint64_t*>(paramaddr);
|
||||
} else {
|
||||
ldsAddress += *reinterpret_cast<const uint32_t*>(paramaddr);
|
||||
}
|
||||
}
|
||||
break;
|
||||
case HSAIL_ARGTYPE_VALUE:
|
||||
|
||||
新しいイシューから参照
ユーザーをブロックする