SWDEV-257789 - Initial change to skip kernel arg copy
The optimization is controlled with ROCR_SKIP_KERNEL_ARG_COPY. This is initial check-in for experiments. Extra changes are necessary for full support: - handle graph capture with the original sysmem alloc - avoid memobject references, otherwise there is a race condition with reusage of the arg buffer - Remove arg setup from hip Change-Id: Ib0af710f93e79834711fa4049a7c66093711e68b
Этот коммит содержится в:
@@ -733,10 +733,6 @@ bool VirtualGPU::processMemObjects(const amd::Kernel& kernel, const_address para
|
||||
const_address srcArgPtr = params + desc.offset_;
|
||||
if (desc.info_.oclObject_ == amd::KernelParameterDescriptor::ReferenceObject) {
|
||||
void* mem = allocKernArg(desc.size_, 128);
|
||||
if (mem == nullptr) {
|
||||
LogError("Out of memory");
|
||||
return false;
|
||||
}
|
||||
memcpy(mem, srcArgPtr, desc.size_);
|
||||
const auto it = hsaKernel.patch().find(desc.offset_);
|
||||
WriteAqlArgAt(const_cast<address>(params), &mem, sizeof(void*), it->second);
|
||||
@@ -1240,6 +1236,17 @@ void* VirtualGPU::allocKernArg(size_t size, size_t alignment) {
|
||||
return result;
|
||||
}
|
||||
|
||||
// ================================================================================================
|
||||
address VirtualGPU::allocKernelArguments(size_t size, size_t alignment) {
|
||||
if (ROCR_SKIP_KERNEL_ARG_COPY) {
|
||||
// Make sure VirtualGPU has an exclusive access to the resources
|
||||
amd::ScopedLock lock(execution());
|
||||
return reinterpret_cast<address>(allocKernArg(size, alignment));
|
||||
} else {
|
||||
return nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
// ================================================================================================
|
||||
/* profilingBegin, when profiling is enabled, creates a timestamp to save in
|
||||
* virtualgpu's timestamp_, and calls start() to get the current host
|
||||
@@ -2703,17 +2710,6 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
|
||||
}
|
||||
}
|
||||
|
||||
// Find all parameters for the current kernel
|
||||
|
||||
// Allocate buffer to hold kernel arguments
|
||||
address argBuffer = (address)allocKernArg(gpuKernel.KernargSegmentByteSize(),
|
||||
gpuKernel.KernargSegmentAlignment());
|
||||
|
||||
if (argBuffer == nullptr) {
|
||||
LogError("Out of memory");
|
||||
return false;
|
||||
}
|
||||
|
||||
ClPrint(amd::LOG_INFO, amd::LOG_KERN, "ShaderName : %s", gpuKernel.name().c_str());
|
||||
|
||||
// Check if runtime has to setup hidden arguments
|
||||
@@ -2817,8 +2813,16 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
|
||||
}
|
||||
}
|
||||
|
||||
// Load all kernel arguments
|
||||
WriteAqlArgAt(argBuffer, parameters, gpuKernel.KernargSegmentByteSize(), 0);
|
||||
address argBuffer = const_cast<address>(parameters);
|
||||
// Find all parameters for the current kernel
|
||||
if (!kernel.parameters().deviceKernelArgs() || gpuKernel.isInternalKernel()) {
|
||||
// Allocate buffer to hold kernel arguments
|
||||
argBuffer = reinterpret_cast<address>(allocKernArg(gpuKernel.KernargSegmentByteSize(),
|
||||
gpuKernel.KernargSegmentAlignment()));
|
||||
// Load all kernel arguments
|
||||
WriteAqlArgAt(argBuffer, parameters, gpuKernel.KernargSegmentByteSize(), 0);
|
||||
}
|
||||
|
||||
// Note: In a case of structs the size won't match,
|
||||
// since HSAIL compiler expects a reference...
|
||||
assert(gpuKernel.KernargSegmentByteSize() <= signature.paramsSize() &&
|
||||
|
||||
Ссылка в новой задаче
Block a user