P4 to Git Change 1109709 by gandryey@gera-dev-w7 on 2015/01/07 18:33:55

ECR #304775 - Enable offline compilation for HSAIL path

Affected files ...

... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpudevice.cpp#487 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpukernel.cpp#272 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuprogram.cpp#187 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuprogram.hpp#57 edit
This commit is contained in:
foreman
2015-01-07 18:40:12 -05:00
parent 187a905974
commit 2ff3f9199a
4 changed files with 79 additions and 15 deletions
+36 -2
View File
@@ -142,6 +142,8 @@ NullDevice::create(CALtarget target)
case CAL_TARGET_HAINAN:
case CAL_TARGET_DEVASTATOR:
case CAL_TARGET_SCRAPPER:
calAttr.doublePrecision = CAL_TRUE;
break;
case CAL_TARGET_BONAIRE:
case CAL_TARGET_SPECTRE:
case CAL_TARGET_SPOOKY:
@@ -155,6 +157,7 @@ NullDevice::create(CALtarget target)
case CAL_TARGET_CARRIZO:
case CAL_TARGET_ELLESMERE:
calAttr.doublePrecision = CAL_TRUE;
calAttr.isOpenCL200Device = CAL_TRUE;
break;
default:
break;
@@ -181,9 +184,34 @@ NullDevice::create(CALtarget target)
::strcpy(info_.vendor_, "Advanced Micro Devices, Inc.");
::snprintf(info_.driverVersion_, sizeof(info_.driverVersion_) - 1,
AMD_BUILD_STRING);
if (gpuSettings->ciPlus_) {
if (settings().hsail_ || (settings().oclVersion_ == OpenCL20)) {
info_.version_ = "OpenCL 2.0 " AMD_PLATFORM_INFO;
info_.oclcVersion_ = "OpenCL C 2.0 ";
// Runtime doesn't know what local size could be on the real board
info_.maxGlobalVariableSize_ = static_cast<size_t>(512 * Mi);
if (NULL == hsaCompiler_) {
const char* library = getenv("HSA_COMPILER_LIBRARY");
aclCompilerOptions opts = {
sizeof(aclCompilerOptions_0_8),
library,
NULL,
NULL,
NULL,
NULL,
NULL,
AMD_OCL_SC_LIB,
&::malloc,
&::free
};
// Initialize the compiler handle
acl_error error;
hsaCompiler_ = aclCompilerInit(&opts, &error);
if (error != ACL_SUCCESS) {
LogError("Error initializing the compiler");
return false;
}
}
}
else {
info_.version_ = "OpenCL 1.2 " AMD_PLATFORM_INFO;
@@ -196,7 +224,13 @@ NullDevice::create(CALtarget target)
device::Program*
NullDevice::createProgram(int oclVer)
{
NullProgram* nullProgram = new NullProgram(*this);
device::Program* nullProgram;
if (settings().hsail_ || (oclVer == 200)) {
nullProgram = new HSAILProgram(*this);
}
else {
nullProgram = new NullProgram(*this);
}
if (nullProgram == NULL) {
LogError("Memory allocation has failed!");
}
+4 -1
View File
@@ -3558,7 +3558,10 @@ HSAILKernel::init(bool finalize)
return false;
}
aqlCreateHWInfo(shader_isa, size_isa);
// Allocate HW resources for the real program only
if (!prog().isNull()) {
aqlCreateHWInfo(shader_isa, size_isa);
}
// Pull out metadata from the ELF
size_t sizeOfArgList;
+34 -12
View File
@@ -1772,6 +1772,25 @@ HSAILProgram::HSAILProgram(Device& device)
, globalStore_(NULL)
, kernels_(NULL)
, maxScratchRegs_(0)
, isNull_(false)
{
memset(&binOpts_, 0, sizeof(binOpts_));
binOpts_.struct_size = sizeof(binOpts_);
binOpts_.elfclass = LP64_SWITCH(ELFCLASS32, ELFCLASS64);
binOpts_.bitness = ELFDATA2LSB;
binOpts_.alloc = &::malloc;
binOpts_.dealloc = &::free;
}
HSAILProgram::HSAILProgram(NullDevice& device)
: Program(device)
, llvmBinary_()
, binaryElf_(NULL)
, rawBinary_(NULL)
, globalStore_(NULL)
, kernels_(NULL)
, maxScratchRegs_(0)
, isNull_(true)
{
memset(&binOpts_, 0, sizeof(binOpts_));
binOpts_.struct_size = sizeof(binOpts_);
@@ -1784,11 +1803,11 @@ HSAILProgram::HSAILProgram(Device& device)
HSAILProgram::~HSAILProgram()
{
// Destroy internal static samplers
for (auto it = staticSamplers_.begin(); it != staticSamplers_.end(); ++it) {
delete *it;
for (auto& it : staticSamplers_) {
delete it;
}
if (rawBinary_ != NULL) {
free( rawBinary_ );
free(rawBinary_);
}
acl_error error;
// Free the elf binary
@@ -2137,11 +2156,14 @@ HSAILProgram::linkImpl(amd::option::Options* options)
return false;
}
// ACL_TYPE_CG stage is always being performed
if (!_aclHsaLoader(dev().hsaCompiler(), binaryElf_, this, &AllocateGPUMemory,
&DmaMemoryCopy, &GetSamplerObjectParams, &InitializeSamplerObject)) {
buildLog_ += "Error while BRIG Codegen phase: loading BRIG globals in the ELF \n";
return false;
if (!isNull()) {
if (!_aclHsaLoader(dev().hsaCompiler(), binaryElf_, this, &AllocateGPUMemory,
&DmaMemoryCopy, &GetSamplerObjectParams, &InitializeSamplerObject)) {
buildLog_ += "Error while BRIG Codegen phase: loading BRIG globals in the ELF \n";
return false;
}
}
size_t kernelNamesSize = 0;
errorCode = aclQueryInfo(dev().hsaCompiler(), binaryElf_, RT_KERNEL_NAMES, NULL, NULL, &kernelNamesSize);
if (errorCode != ACL_SUCCESS) {
@@ -2175,7 +2197,7 @@ HSAILProgram::linkImpl(amd::option::Options* options)
maxScratchRegs_ = std::max(static_cast<uint>(aKernel->workGroupInfo()->scratchRegs_), maxScratchRegs_);
}
// Allocate kernel table for device enqueuing
if (dynamicParallelism && !allocKernelTable()) {
if (!isNull() && dynamicParallelism && !allocKernelTable()) {
return false;
}
}
@@ -2259,8 +2281,8 @@ HSAILProgram::allocKernelTable()
else {
size_t* table = reinterpret_cast<size_t*>(
kernels_->map(NULL, gpu::Resource::WriteOnly));
for (auto it = kernels().begin(); it != kernels().end(); ++it) {
HSAILKernel* kernel = static_cast<HSAILKernel*>(it->second);
for (auto& it : kernels()) {
HSAILKernel* kernel = static_cast<HSAILKernel*>(it.second);
table[kernel->index()] = static_cast<size_t>(
kernel->gpuAqlCode()->vmAddress());
}
@@ -2273,9 +2295,9 @@ void
HSAILProgram::fillResListWithKernels(
std::vector<const Resource*>& memList) const
{
for (auto it = kernels().begin(); it != kernels().end(); ++it) {
for (auto& it : kernels()) {
memList.push_back(
static_cast<HSAILKernel*>(it->second)->gpuAqlCode());
static_cast<HSAILKernel*>(it.second)->gpuAqlCode());
}
}
+5
View File
@@ -377,6 +377,7 @@ class HSAILProgram : public device::Program
public:
//! Default constructor
HSAILProgram(Device& device);
HSAILProgram(NullDevice& device);
//! Default destructor
~HSAILProgram();
@@ -405,6 +406,9 @@ public:
//! Add internal static sampler
void addSampler(Sampler* sampler) { staticSamplers_.push_back(sampler); }
//! Returns TRUE if the program just compiled
bool isNull() const { return isNull_; }
protected:
//! pre-compile setup for GPU
virtual bool initBuild(amd::option::Options* options);
@@ -499,6 +503,7 @@ private:
Memory* kernels_; //!< Table with kernel object pointers
uint maxScratchRegs_; //!< Maximum number of scratch regs used in the program by individual kernel
std::list<Sampler*> staticSamplers_; //!< List od internal static samplers
bool isNull_; //!< Null program no memory allocations
};
/*@}*/} // namespace gpu