P4 to Git Change 1981122 by gandryey@gera-win10 on 2019/08/09 17:59:31

SWDEV-79445 - OCL generic changes and code clean-up
	- Allow async execution with scratch on the same queue. COMPUTE_TMPRING_SIZE.WAVESIZE should be constant across all dispatches.

Affected files ...

... //depot/stg/opencl/drivers/opencl/runtime/device/pal/paldevice.cpp#151 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/paldevice.hpp#40 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.cpp#145 edit
Αυτή η υποβολή περιλαμβάνεται σε:
foreman
2019-08-09 18:04:15 -04:00
γονέας 2fbff434ba
υποβολή 0a2d4e2dc3
3 αρχεία άλλαξαν με 29 προσθήκες και 26 διαγραφές
@@ -2370,12 +2370,15 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
dispatchParam.scratchAddr = scratch->memObj_->vmAddress();
dispatchParam.scratchSize = scratch->size_;
dispatchParam.scratchOffset = scratch->offset_;
// Use maximum available slots for all dispatches to allow async on the same queue
// HW value loaded into SGPR is an offset value calculated as
// wave_slot * COMPUTE_TMPRING_SIZE.WAVESIZE
dispatchParam.workitemPrivateSegmentSize = scratch->privateMemSize_;
}
dispatchParam.pCpuAqlCode = hsaKernel.cpuAqlCode();
dispatchParam.hsaQueueVa = hsaQueueMem_->vmAddress();
dispatchParam.wavesPerSh = (enqueueEvent != nullptr) ? enqueueEvent->profilingInfo().waves_ : 0;
dispatchParam.useAtc = dev().settings().svmFineGrainSystem_ ? true : false;
dispatchParam.workitemPrivateSegmentSize = hsaKernel.spillSegSize();
dispatchParam.kernargSegmentSize = hsaKernel.argsBufferSize();
// Run AQL dispatch in HW
eventBegin(MainEngine);