P4 to Git Change 1558704 by gandryey@gera-w8 on 2018/05/23 17:20:01
SWDEV-79445 - OCL generic changes and code clean-up
- ABI clean-up. Stage 1: Separate kernel arguments and OCL objects. OCL objects will be passed in the new arrays of mem objects, samplers and device queue objects. The kernel arguments will contain GPU virtual addresses.
http://ocltc.amd.com/reviews/r/14881/
Affected files ...
... //depot/stg/opencl/drivers/opencl/api/opencl/amdocl/cl_program.cpp#48 edit
... //depot/stg/opencl/drivers/opencl/api/opencl/amdocl/cl_svm.cpp#25 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/device.hpp#302 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpublit.cpp#129 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpukernel.cpp#323 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpukernel.hpp#128 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpumemory.hpp#51 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuvirtual.cpp#417 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palblit.cpp#23 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.cpp#50 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palmemory.hpp#7 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.cpp#97 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/rocm/rocblit.cpp#22 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/rocm/rocblit.hpp#9 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/rocm/rocdevice.hpp#28 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/rocm/rocmemory.hpp#12 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/rocm/rocvirtual.cpp#51 edit
... //depot/stg/opencl/drivers/opencl/runtime/platform/command.cpp#86 edit
... //depot/stg/opencl/drivers/opencl/runtime/platform/kernel.cpp#26 edit
... //depot/stg/opencl/drivers/opencl/runtime/platform/kernel.hpp#20 edit
[ROCm/clr commit: 2176dc3b19]
This commit is contained in:
@@ -934,6 +934,8 @@ hsa_kernel_dispatch_packet_t* HSAILKernel::loadArguments(
|
||||
|
||||
const amd::KernelSignature& signature = kernel.signature();
|
||||
const amd::KernelParameters& kernelParams = kernel.parameters();
|
||||
amd::Memory* const* memories =
|
||||
reinterpret_cast<amd::Memory* const*>(parameters + kernelParams.memoryObjOffset());
|
||||
|
||||
// Find all parameters for the current kernel
|
||||
for (auto arg : arguments_) {
|
||||
@@ -1005,45 +1007,24 @@ hsa_kernel_dispatch_packet_t* HSAILKernel::loadArguments(
|
||||
// If it is a global pointer
|
||||
Memory* gpuMem = nullptr;
|
||||
amd::Memory* mem = nullptr;
|
||||
|
||||
if (kernelParams.boundToSvmPointer(dev(), parameters, arg->index_)) {
|
||||
WriteAqlArg(&aqlArgBuf, paramaddr, sizeof(paramaddr), sizeof(paramaddr));
|
||||
mem = amd::SvmManager::FindSvmBuffer(*reinterpret_cast<void* const*>(paramaddr));
|
||||
if (mem != nullptr) {
|
||||
gpuMem = dev().getGpuMemory(mem);
|
||||
gpuMem->wait(gpu, WaitOnBusyEngine);
|
||||
if ((mem->getMemFlags() & CL_MEM_READ_ONLY) == 0) {
|
||||
mem->signalWrite(&dev());
|
||||
}
|
||||
gpu.addVmMemory(gpuMem);
|
||||
}
|
||||
// If finegrainsystem is present then the pointer can be malloced by the app and
|
||||
// passed to kernel directly. If so copy the pointer location to aqlArgBuf
|
||||
else if (!dev().isFineGrainedSystem(true)) {
|
||||
return nullptr;
|
||||
}
|
||||
break;
|
||||
}
|
||||
uint32_t index = signature.at(arg->index_).info_.arrayIndex_;
|
||||
if (nativeMem) {
|
||||
gpuMem = *reinterpret_cast<Memory* const*>(paramaddr);
|
||||
gpuMem = reinterpret_cast<Memory* const*>(memories)[index];
|
||||
if (nullptr != gpuMem) {
|
||||
mem = gpuMem->owner();
|
||||
}
|
||||
} else {
|
||||
mem = *reinterpret_cast<amd::Memory* const*>(paramaddr);
|
||||
mem = memories[index];
|
||||
if (mem != nullptr) {
|
||||
gpuMem = dev().getGpuMemory(mem);
|
||||
}
|
||||
}
|
||||
|
||||
WriteAqlArg(&aqlArgBuf, paramaddr, sizeof(paramaddr), sizeof(paramaddr));
|
||||
if (gpuMem == nullptr) {
|
||||
WriteAqlArg(&aqlArgBuf, &gpuMem, arg->size_, arg->alignment_);
|
||||
break;
|
||||
}
|
||||
|
||||
//! 64 bit isn't supported with 32 bit binary
|
||||
uint64_t globalAddress = gpuMem->vmAddress();
|
||||
WriteAqlArg(&aqlArgBuf, &globalAddress, arg->size_, arg->alignment_);
|
||||
|
||||
// Wait for resource if it was used on an inactive engine
|
||||
//! \note syncCache may call DRM transfer
|
||||
gpuMem->wait(gpu, WaitOnBusyEngine);
|
||||
@@ -1083,15 +1064,17 @@ hsa_kernel_dispatch_packet_t* HSAILKernel::loadArguments(
|
||||
case HSAIL_ARGTYPE_IMAGE: {
|
||||
Image* image = nullptr;
|
||||
amd::Memory* mem = nullptr;
|
||||
uint32_t index = signature.at(arg->index_).info_.arrayIndex_;
|
||||
if (nativeMem) {
|
||||
image = static_cast<Image*>(*reinterpret_cast<Memory* const*>(paramaddr));
|
||||
} else {
|
||||
mem = *reinterpret_cast<amd::Memory* const*>(paramaddr);
|
||||
if (mem == nullptr) {
|
||||
LogError("The kernel image argument isn't an image object!");
|
||||
return nullptr;
|
||||
image = reinterpret_cast<Image* const*>(memories)[index];
|
||||
if (nullptr != image) {
|
||||
mem = image->owner();
|
||||
}
|
||||
} else {
|
||||
mem = memories[index];
|
||||
if (mem != nullptr) {
|
||||
image = static_cast<Image*>(dev().getGpuMemory(mem));
|
||||
}
|
||||
image = static_cast<Image*>(dev().getGpuMemory(mem));
|
||||
}
|
||||
|
||||
// Wait for resource if it was used on an inactive engine
|
||||
@@ -1127,7 +1110,9 @@ hsa_kernel_dispatch_packet_t* HSAILKernel::loadArguments(
|
||||
break;
|
||||
}
|
||||
case HSAIL_ARGTYPE_SAMPLER: {
|
||||
const amd::Sampler* sampler = *reinterpret_cast<amd::Sampler* const*>(paramaddr);
|
||||
uint32_t index = signature.at(arg->index_).info_.arrayIndex_;
|
||||
const amd::Sampler* sampler = reinterpret_cast<amd::Sampler* const*>(parameters +
|
||||
kernelParams.samplerObjOffset())[index];
|
||||
const Sampler* gpuSampler = static_cast<Sampler*>(sampler->getDeviceSampler(dev()));
|
||||
uint64_t srd = gpuSampler->hwSrd();
|
||||
WriteAqlArg(&aqlArgBuf, &srd, sizeof(srd), sizeof(srd));
|
||||
@@ -1135,7 +1120,9 @@ hsa_kernel_dispatch_packet_t* HSAILKernel::loadArguments(
|
||||
break;
|
||||
}
|
||||
case HSAIL_ARGTYPE_QUEUE: {
|
||||
const amd::DeviceQueue* queue = *reinterpret_cast<amd::DeviceQueue* const*>(paramaddr);
|
||||
uint32_t index = signature.at(arg->index_).info_.arrayIndex_;
|
||||
const amd::DeviceQueue* queue = reinterpret_cast<amd::DeviceQueue* const*>(
|
||||
parameters + kernelParams.queueObjOffset())[index];
|
||||
VirtualGPU* gpuQueue = static_cast<VirtualGPU*>(queue->vDev());
|
||||
uint64_t vmQueue;
|
||||
if (dev().settings().useDeviceQueue_) {
|
||||
|
||||
Reference in New Issue
Block a user