SWDEV-257789 - Initial change to skip kernel arg copy

The optimization is controlled with ROCR_SKIP_KERNEL_ARG_COPY.
This is initial check-in for experiments. Extra changes are
necessary for full support:
- handle graph capture with the original sysmem alloc
- avoid memobject references, otherwise there is a race condition with
reusage of the arg buffer
- Remove arg setup from hip

Change-Id: Ib0af710f93e79834711fa4049a7c66093711e68b
Этот коммит содержится в:
German Andryeyev
2021-10-27 16:22:40 -04:00
родитель 530283e12a
Коммит 7e12cf6318
7 изменённых файлов: 57 добавлений и 29 удалений
+21 -17
Просмотреть файл
@@ -733,10 +733,6 @@ bool VirtualGPU::processMemObjects(const amd::Kernel& kernel, const_address para
const_address srcArgPtr = params + desc.offset_;
if (desc.info_.oclObject_ == amd::KernelParameterDescriptor::ReferenceObject) {
void* mem = allocKernArg(desc.size_, 128);
if (mem == nullptr) {
LogError("Out of memory");
return false;
}
memcpy(mem, srcArgPtr, desc.size_);
const auto it = hsaKernel.patch().find(desc.offset_);
WriteAqlArgAt(const_cast<address>(params), &mem, sizeof(void*), it->second);
@@ -1240,6 +1236,17 @@ void* VirtualGPU::allocKernArg(size_t size, size_t alignment) {
return result;
}
// ================================================================================================
address VirtualGPU::allocKernelArguments(size_t size, size_t alignment) {
if (ROCR_SKIP_KERNEL_ARG_COPY) {
// Make sure VirtualGPU has an exclusive access to the resources
amd::ScopedLock lock(execution());
return reinterpret_cast<address>(allocKernArg(size, alignment));
} else {
return nullptr;
}
}
// ================================================================================================
/* profilingBegin, when profiling is enabled, creates a timestamp to save in
* virtualgpu's timestamp_, and calls start() to get the current host
@@ -2703,17 +2710,6 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
}
}
// Find all parameters for the current kernel
// Allocate buffer to hold kernel arguments
address argBuffer = (address)allocKernArg(gpuKernel.KernargSegmentByteSize(),
gpuKernel.KernargSegmentAlignment());
if (argBuffer == nullptr) {
LogError("Out of memory");
return false;
}
ClPrint(amd::LOG_INFO, amd::LOG_KERN, "ShaderName : %s", gpuKernel.name().c_str());
// Check if runtime has to setup hidden arguments
@@ -2817,8 +2813,16 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
}
}
// Load all kernel arguments
WriteAqlArgAt(argBuffer, parameters, gpuKernel.KernargSegmentByteSize(), 0);
address argBuffer = const_cast<address>(parameters);
// Find all parameters for the current kernel
if (!kernel.parameters().deviceKernelArgs() || gpuKernel.isInternalKernel()) {
// Allocate buffer to hold kernel arguments
argBuffer = reinterpret_cast<address>(allocKernArg(gpuKernel.KernargSegmentByteSize(),
gpuKernel.KernargSegmentAlignment()));
// Load all kernel arguments
WriteAqlArgAt(argBuffer, parameters, gpuKernel.KernargSegmentByteSize(), 0);
}
// Note: In a case of structs the size won't match,
// since HSAIL compiler expects a reference...
assert(gpuKernel.KernargSegmentByteSize() <= signature.paramsSize() &&