Apply .clangformat to all repo source files

Change-Id: I7e79c6058f0303f9a98911e3b7dd2e8596079344


[ROCm/clr commit: 9e47fccc89]
This commit is contained in:
Maneesh Gupta
2018-03-12 11:29:03 +05:30
parent ecbb701440
commit 46ddefedee
293 changed files with 43980 additions and 45830 deletions
+5 -5
View File
@@ -26,20 +26,20 @@ const char SectionName[] = ".note";
const char NoteName[] = "AMD";
// TODO: Move this enum to include/llvm/Support so it can be used in tools?
enum NoteType{
enum NoteType {
NT_AMDGPU_HSA_CODE_OBJECT_VERSION = 1,
NT_AMDGPU_HSA_HSAIL = 2,
NT_AMDGPU_HSA_ISA = 3,
NT_AMDGPU_HSA_PRODUCER = 4,
NT_AMDGPU_HSA_PRODUCER_OPTIONS = 5,
NT_AMDGPU_HSA_EXTENSION = 6,
NT_AMDGPU_HSA_RUNTIME_METADATA_V_1 = 7, // deprecated since 12/14/16.
NT_AMDGPU_HSA_RUNTIME_METADATA_V_1 = 7, // deprecated since 12/14/16.
NT_AMDGPU_HSA_RUNTIME_METADATA_V_2 = 8,
NT_AMDGPU_HSA_RUNTIME_METADATA = NT_AMDGPU_HSA_RUNTIME_METADATA_V_2,
NT_AMDGPU_HSA_HLDEBUG_DEBUG = 101,
NT_AMDGPU_HSA_HLDEBUG_TARGET = 102
};
}
}
} // namespace ElfNote
} // namespace AMDGPU
#endif // LLVM_LIB_TARGET_AMDGPU_AMDGPUNOTETYPE_H
#endif // LLVM_LIB_TARGET_AMDGPU_AMDGPUNOTETYPE_H
+215 -215
View File
@@ -39,252 +39,252 @@
namespace AMDGPU {
namespace RuntimeMD {
// Version and revision of runtime metadata
const unsigned char MDVersion = 2;
const unsigned char MDRevision = 1;
// Version and revision of runtime metadata
const unsigned char MDVersion = 2;
const unsigned char MDRevision = 1;
// Name of keys for runtime metadata.
namespace KeyName {
// Name of keys for runtime metadata.
namespace KeyName {
// Runtime metadata version
const char MDVersion[] = "amd.MDVersion";
// Runtime metadata version
const char MDVersion[] = "amd.MDVersion";
// Instruction set architecture information
const char IsaInfo[] = "amd.IsaInfo";
// Wavefront size
const char IsaInfoWavefrontSize[] = "amd.IsaInfoWavefrontSize";
// Local memory size in bytes
const char IsaInfoLocalMemorySize[] = "amd.IsaInfoLocalMemorySize";
// Number of execution units per compute unit
const char IsaInfoEUsPerCU[] = "amd.IsaInfoEUsPerCU";
// Maximum number of waves per execution unit
const char IsaInfoMaxWavesPerEU[] = "amd.IsaInfoMaxWavesPerEU";
// Maximum flat work group size
const char IsaInfoMaxFlatWorkGroupSize[] = "amd.IsaInfoMaxFlatWorkGroupSize";
// SGPR allocation granularity
const char IsaInfoSGPRAllocGranule[] = "amd.IsaInfoSGPRAllocGranule";
// Total number of SGPRs
const char IsaInfoTotalNumSGPRs[] = "amd.IsaInfoTotalNumSGPRs";
// Addressable number of SGPRs
const char IsaInfoAddressableNumSGPRs[] = "amd.IsaInfoAddressableNumSGPRs";
// VGPR allocation granularity
const char IsaInfoVGPRAllocGranule[] = "amd.IsaInfoVGPRAllocGranule";
// Total number of VGPRs
const char IsaInfoTotalNumVGPRs[] = "amd.IsaInfoTotalNumVGPRs";
// Addressable number of VGPRs
const char IsaInfoAddressableNumVGPRs[] = "amd.IsaInfoAddressableNumVGPRs";
// Instruction set architecture information
const char IsaInfo[] = "amd.IsaInfo";
// Wavefront size
const char IsaInfoWavefrontSize[] = "amd.IsaInfoWavefrontSize";
// Local memory size in bytes
const char IsaInfoLocalMemorySize[] = "amd.IsaInfoLocalMemorySize";
// Number of execution units per compute unit
const char IsaInfoEUsPerCU[] = "amd.IsaInfoEUsPerCU";
// Maximum number of waves per execution unit
const char IsaInfoMaxWavesPerEU[] = "amd.IsaInfoMaxWavesPerEU";
// Maximum flat work group size
const char IsaInfoMaxFlatWorkGroupSize[] = "amd.IsaInfoMaxFlatWorkGroupSize";
// SGPR allocation granularity
const char IsaInfoSGPRAllocGranule[] = "amd.IsaInfoSGPRAllocGranule";
// Total number of SGPRs
const char IsaInfoTotalNumSGPRs[] = "amd.IsaInfoTotalNumSGPRs";
// Addressable number of SGPRs
const char IsaInfoAddressableNumSGPRs[] = "amd.IsaInfoAddressableNumSGPRs";
// VGPR allocation granularity
const char IsaInfoVGPRAllocGranule[] = "amd.IsaInfoVGPRAllocGranule";
// Total number of VGPRs
const char IsaInfoTotalNumVGPRs[] = "amd.IsaInfoTotalNumVGPRs";
// Addressable number of VGPRs
const char IsaInfoAddressableNumVGPRs[] = "amd.IsaInfoAddressableNumVGPRs";
// Language
const char Language[] = "amd.Language";
// Language version
const char LanguageVersion[] = "amd.LanguageVersion";
// Language
const char Language[] = "amd.Language";
// Language version
const char LanguageVersion[] = "amd.LanguageVersion";
// Kernels
const char Kernels[] = "amd.Kernels";
// Kernel name
const char KernelName[] = "amd.KernelName";
// Kernel arguments
const char Args[] = "amd.Args";
// Kernel argument size in bytes
const char ArgSize[] = "amd.ArgSize";
// Kernel argument alignment
const char ArgAlign[] = "amd.ArgAlign";
// Kernel argument type name
const char ArgTypeName[] = "amd.ArgTypeName";
// Kernel argument name
const char ArgName[] = "amd.ArgName";
// Kernel argument kind
const char ArgKind[] = "amd.ArgKind";
// Kernel argument value type
const char ArgValueType[] = "amd.ArgValueType";
// Kernel argument address qualifier
const char ArgAddrQual[] = "amd.ArgAddrQual";
// Kernel argument access qualifier
const char ArgAccQual[] = "amd.ArgAccQual";
// Kernel argument is const qualified
const char ArgIsConst[] = "amd.ArgIsConst";
// Kernel argument is restrict qualified
const char ArgIsRestrict[] = "amd.ArgIsRestrict";
// Kernel argument is volatile qualified
const char ArgIsVolatile[] = "amd.ArgIsVolatile";
// Kernel argument is pipe qualified
const char ArgIsPipe[] = "amd.ArgIsPipe";
// Required work group size
const char ReqdWorkGroupSize[] = "amd.ReqdWorkGroupSize";
// Work group size hint
const char WorkGroupSizeHint[] = "amd.WorkGroupSizeHint";
// Vector type hint
const char VecTypeHint[] = "amd.VecTypeHint";
// Kernel index for device enqueue
const char KernelIndex[] = "amd.KernelIndex";
// No partial work groups
const char NoPartialWorkGroups[] = "amd.NoPartialWorkGroups";
// Prinf function call information
const char PrintfInfo[] = "amd.PrintfInfo";
// The actual kernel argument access qualifier
const char ArgActualAcc[] = "amd.ArgActualAcc";
// Alignment of pointee type
const char ArgPointeeAlign[] = "amd.ArgPointeeAlign";
// Kernels
const char Kernels[] = "amd.Kernels";
// Kernel name
const char KernelName[] = "amd.KernelName";
// Kernel arguments
const char Args[] = "amd.Args";
// Kernel argument size in bytes
const char ArgSize[] = "amd.ArgSize";
// Kernel argument alignment
const char ArgAlign[] = "amd.ArgAlign";
// Kernel argument type name
const char ArgTypeName[] = "amd.ArgTypeName";
// Kernel argument name
const char ArgName[] = "amd.ArgName";
// Kernel argument kind
const char ArgKind[] = "amd.ArgKind";
// Kernel argument value type
const char ArgValueType[] = "amd.ArgValueType";
// Kernel argument address qualifier
const char ArgAddrQual[] = "amd.ArgAddrQual";
// Kernel argument access qualifier
const char ArgAccQual[] = "amd.ArgAccQual";
// Kernel argument is const qualified
const char ArgIsConst[] = "amd.ArgIsConst";
// Kernel argument is restrict qualified
const char ArgIsRestrict[] = "amd.ArgIsRestrict";
// Kernel argument is volatile qualified
const char ArgIsVolatile[] = "amd.ArgIsVolatile";
// Kernel argument is pipe qualified
const char ArgIsPipe[] = "amd.ArgIsPipe";
// Required work group size
const char ReqdWorkGroupSize[] = "amd.ReqdWorkGroupSize";
// Work group size hint
const char WorkGroupSizeHint[] = "amd.WorkGroupSizeHint";
// Vector type hint
const char VecTypeHint[] = "amd.VecTypeHint";
// Kernel index for device enqueue
const char KernelIndex[] = "amd.KernelIndex";
// No partial work groups
const char NoPartialWorkGroups[] = "amd.NoPartialWorkGroups";
// Prinf function call information
const char PrintfInfo[] = "amd.PrintfInfo";
// The actual kernel argument access qualifier
const char ArgActualAcc[] = "amd.ArgActualAcc";
// Alignment of pointee type
const char ArgPointeeAlign[] = "amd.ArgPointeeAlign";
} // end namespace KeyName
} // end namespace KeyName
namespace KernelArg {
namespace KernelArg {
enum Kind : uint8_t {
ByValue = 0,
GlobalBuffer = 1,
DynamicSharedPointer = 2,
Sampler = 3,
Image = 4,
Pipe = 5,
Queue = 6,
HiddenGlobalOffsetX = 7,
HiddenGlobalOffsetY = 8,
HiddenGlobalOffsetZ = 9,
HiddenNone = 10,
HiddenPrintfBuffer = 11,
HiddenDefaultQueue = 12,
HiddenCompletionAction = 13,
};
enum Kind : uint8_t {
ByValue = 0,
GlobalBuffer = 1,
DynamicSharedPointer = 2,
Sampler = 3,
Image = 4,
Pipe = 5,
Queue = 6,
HiddenGlobalOffsetX = 7,
HiddenGlobalOffsetY = 8,
HiddenGlobalOffsetZ = 9,
HiddenNone = 10,
HiddenPrintfBuffer = 11,
HiddenDefaultQueue = 12,
HiddenCompletionAction = 13,
};
enum ValueType : uint16_t {
Struct = 0,
I8 = 1,
U8 = 2,
I16 = 3,
U16 = 4,
F16 = 5,
I32 = 6,
U32 = 7,
F32 = 8,
I64 = 9,
U64 = 10,
F64 = 11,
};
enum ValueType : uint16_t {
Struct = 0,
I8 = 1,
U8 = 2,
I16 = 3,
U16 = 4,
F16 = 5,
I32 = 6,
U32 = 7,
F32 = 8,
I64 = 9,
U64 = 10,
F64 = 11,
};
// Avoid using 'None' since it conflicts with a macro in X11 header file.
enum AccessQualifer : uint8_t {
AccNone = 0,
ReadOnly = 1,
WriteOnly = 2,
ReadWrite = 3,
};
// Avoid using 'None' since it conflicts with a macro in X11 header file.
enum AccessQualifer : uint8_t {
AccNone = 0,
ReadOnly = 1,
WriteOnly = 2,
ReadWrite = 3,
};
enum AddressSpaceQualifer : uint8_t {
Private = 0,
Global = 1,
Constant = 2,
Local = 3,
Generic = 4,
Region = 5,
};
enum AddressSpaceQualifer : uint8_t {
Private = 0,
Global = 1,
Constant = 2,
Local = 3,
Generic = 4,
Region = 5,
};
} // end namespace KernelArg
} // end namespace KernelArg
// Invalid values are used to indicate an optional key should not be emitted.
const uint8_t INVALID_ADDR_QUAL = 0xff;
const uint8_t INVALID_ACC_QUAL = 0xff;
const uint32_t INVALID_KERNEL_INDEX = ~0U;
// Invalid values are used to indicate an optional key should not be emitted.
const uint8_t INVALID_ADDR_QUAL = 0xff;
const uint8_t INVALID_ACC_QUAL = 0xff;
const uint32_t INVALID_KERNEL_INDEX = ~0U;
namespace KernelArg {
namespace KernelArg {
// In-memory representation of kernel argument information.
struct Metadata {
uint32_t Size = 0;
uint32_t Align = 0;
uint32_t PointeeAlign = 0;
uint8_t Kind = 0;
uint16_t ValueType = 0;
std::string TypeName;
std::string Name;
uint8_t AddrQual = INVALID_ADDR_QUAL;
uint8_t AccQual = INVALID_ACC_QUAL;
uint8_t IsVolatile = 0;
uint8_t IsConst = 0;
uint8_t IsRestrict = 0;
uint8_t IsPipe = 0;
// In-memory representation of kernel argument information.
struct Metadata {
uint32_t Size = 0;
uint32_t Align = 0;
uint32_t PointeeAlign = 0;
uint8_t Kind = 0;
uint16_t ValueType = 0;
std::string TypeName;
std::string Name;
uint8_t AddrQual = INVALID_ADDR_QUAL;
uint8_t AccQual = INVALID_ACC_QUAL;
uint8_t IsVolatile = 0;
uint8_t IsConst = 0;
uint8_t IsRestrict = 0;
uint8_t IsPipe = 0;
Metadata() = default;
};
Metadata() = default;
};
} // end namespace KernelArg
} // end namespace KernelArg
namespace Kernel {
namespace Kernel {
// In-memory representation of kernel information.
struct Metadata {
std::string Name;
std::string Language;
std::vector<uint8_t> LanguageVersion;
std::vector<uint32_t> ReqdWorkGroupSize;
std::vector<uint32_t> WorkGroupSizeHint;
std::string VecTypeHint;
uint32_t KernelIndex = INVALID_KERNEL_INDEX;
uint8_t NoPartialWorkGroups = 0;
std::vector<KernelArg::Metadata> Args;
// In-memory representation of kernel information.
struct Metadata {
std::string Name;
std::string Language;
std::vector<uint8_t> LanguageVersion;
std::vector<uint32_t> ReqdWorkGroupSize;
std::vector<uint32_t> WorkGroupSizeHint;
std::string VecTypeHint;
uint32_t KernelIndex = INVALID_KERNEL_INDEX;
uint8_t NoPartialWorkGroups = 0;
std::vector<KernelArg::Metadata> Args;
Metadata() = default;
};
Metadata() = default;
};
} // end namespace Kernel
} // end namespace Kernel
namespace IsaInfo {
namespace IsaInfo {
/// \brief In-memory representation of instruction set architecture
/// information.
struct Metadata {
/// \brief Wavefront size.
unsigned WavefrontSize = 0;
/// \brief Local memory size in bytes.
unsigned LocalMemorySize = 0;
/// \brief Number of execution units per compute unit.
unsigned EUsPerCU = 0;
/// \brief Maximum number of waves per execution unit.
unsigned MaxWavesPerEU = 0;
/// \brief Maximum flat work group size.
unsigned MaxFlatWorkGroupSize = 0;
/// \brief SGPR allocation granularity.
unsigned SGPRAllocGranule = 0;
/// \brief Total number of SGPRs.
unsigned TotalNumSGPRs = 0;
/// \brief Addressable number of SGPRs.
unsigned AddressableNumSGPRs = 0;
/// \brief VGPR allocation granularity.
unsigned VGPRAllocGranule = 0;
/// \brief Total number of VGPRs.
unsigned TotalNumVGPRs = 0;
/// \brief Addressable number of VGPRs.
unsigned AddressableNumVGPRs = 0;
/// \brief In-memory representation of instruction set architecture
/// information.
struct Metadata {
/// \brief Wavefront size.
unsigned WavefrontSize = 0;
/// \brief Local memory size in bytes.
unsigned LocalMemorySize = 0;
/// \brief Number of execution units per compute unit.
unsigned EUsPerCU = 0;
/// \brief Maximum number of waves per execution unit.
unsigned MaxWavesPerEU = 0;
/// \brief Maximum flat work group size.
unsigned MaxFlatWorkGroupSize = 0;
/// \brief SGPR allocation granularity.
unsigned SGPRAllocGranule = 0;
/// \brief Total number of SGPRs.
unsigned TotalNumSGPRs = 0;
/// \brief Addressable number of SGPRs.
unsigned AddressableNumSGPRs = 0;
/// \brief VGPR allocation granularity.
unsigned VGPRAllocGranule = 0;
/// \brief Total number of VGPRs.
unsigned TotalNumVGPRs = 0;
/// \brief Addressable number of VGPRs.
unsigned AddressableNumVGPRs = 0;
Metadata() = default;
};
Metadata() = default;
};
} // end namespace IsaInfo
} // end namespace IsaInfo
namespace Program {
namespace Program {
// In-memory representation of program information.
struct Metadata {
std::vector<uint8_t> MDVersionSeq;
IsaInfo::Metadata IsaInfo;
std::vector<std::string> PrintfInfo;
std::vector<Kernel::Metadata> Kernels;
// In-memory representation of program information.
struct Metadata {
std::vector<uint8_t> MDVersionSeq;
IsaInfo::Metadata IsaInfo;
std::vector<std::string> PrintfInfo;
std::vector<Kernel::Metadata> Kernels;
explicit Metadata() = default;
explicit Metadata() = default;
// Construct from an YAML string.
explicit Metadata(const std::string &YAML);
// Construct from an YAML string.
explicit Metadata(const std::string& YAML);
// Convert to YAML string.
std::string toYAML();
// Convert to YAML string.
std::string toYAML();
// Convert from YAML string.
static Metadata fromYAML(const std::string &S);
};
// Convert from YAML string.
static Metadata fromYAML(const std::string& S);
};
} //end namespace Program
} // end namespace Program
} // end namespace RuntimeMD
} // end namespace AMDGPU
} // end namespace RuntimeMD
} // end namespace AMDGPU
#endif // LLVM_LIB_TARGET_AMDGPU_AMDGPURUNTIMEMETADATA_H
#endif // LLVM_LIB_TARGET_AMDGPU_AMDGPURUNTIMEMETADATA_H
+5 -10
View File
@@ -10,8 +10,7 @@
using namespace std;
hsa_isa_t hip_impl::triple_to_hsa_isa(const std::string& triple)
{
hsa_isa_t hip_impl::triple_to_hsa_isa(const std::string& triple) {
static constexpr const char prefix[] = "hcc-amdgcn--amdhsa-gfx";
static constexpr size_t prefix_sz = sizeof(prefix) - 1;
@@ -38,20 +37,16 @@ constexpr const char hip_impl::Bundled_code_header::magic_string_[];
// CREATORS
hip_impl::Bundled_code_header::Bundled_code_header(const vector<char>& x)
: Bundled_code_header{x.cbegin(), x.cend()}
{}
: Bundled_code_header{x.cbegin(), x.cend()} {}
hip_impl::Bundled_code_header::Bundled_code_header(const void* p)
{ // This is a pretty terrible interface, useful only because
hip_impl::Bundled_code_header::Bundled_code_header(
const void* p) { // This is a pretty terrible interface, useful only because
// hipLoadModuleData is so poorly specified (for no fault of its own).
if (!p) return;
auto ph = static_cast<const Header_*>(p);
if (!equal(
magic_string_,
magic_string_ + magic_string_sz_,
ph->bundler_magic_string_)) {
if (!equal(magic_string_, magic_string_ + magic_string_sz_, ph->bundler_magic_string_)) {
return;
}
+214 -429
View File
@@ -28,521 +28,306 @@ extern "C" float __ocml_rint_f32(float);
extern "C" float __ocml_ceil_f32(float);
extern "C" float __ocml_trunc_f32(float);
__device__ float __double2float_rd(double x)
{
return (double)x;
__device__ float __double2float_rd(double x) { return (double)x; }
__device__ float __double2float_rn(double x) { return (double)x; }
__device__ float __double2float_ru(double x) { return (double)x; }
__device__ float __double2float_rz(double x) { return (double)x; }
__device__ int __double2hiint(double x) {
static_assert(sizeof(double) == 2 * sizeof(int), "");
int tmp[2];
__builtin_memcpy(tmp, &x, sizeof(tmp));
return tmp[1];
}
__device__ float __double2float_rn(double x)
{
return (double)x;
}
__device__ float __double2float_ru(double x)
{
return (double)x;
}
__device__ float __double2float_rz(double x)
{
return (double)x;
__device__ int __double2loint(double x) {
static_assert(sizeof(double) == 2 * sizeof(int), "");
int tmp[2];
__builtin_memcpy(tmp, &x, sizeof(tmp));
return tmp[0];
}
__device__ int __double2hiint(double x)
{
static_assert(sizeof(double) == 2 * sizeof(int), "");
__device__ int __double2int_rd(double x) { return (int)x; }
__device__ int __double2int_rn(double x) { return (int)x; }
__device__ int __double2int_ru(double x) { return (int)x; }
__device__ int __double2int_rz(double x) { return (int)x; }
int tmp[2];
__builtin_memcpy(tmp, &x, sizeof(tmp));
return tmp[1];
}
__device__ int __double2loint(double x)
{
static_assert(sizeof(double) == 2 * sizeof(int), "");
int tmp[2];
__builtin_memcpy(tmp, &x, sizeof(tmp));
return tmp[0];
}
__device__ long long int __double2ll_rd(double x) { return (long long int)x; }
__device__ long long int __double2ll_rn(double x) { return (long long int)x; }
__device__ long long int __double2ll_ru(double x) { return (long long int)x; }
__device__ long long int __double2ll_rz(double x) { return (long long int)x; }
__device__ int __double2int_rd(double x)
{
return (int)x;
}
__device__ int __double2int_rn(double x)
{
return (int)x;
}
__device__ int __double2int_ru(double x)
{
return (int)x;
}
__device__ int __double2int_rz(double x)
{
return (int)x;
__device__ unsigned int __double2uint_rd(double x) { return (unsigned int)x; }
__device__ unsigned int __double2uint_rn(double x) { return (unsigned int)x; }
__device__ unsigned int __double2uint_ru(double x) { return (unsigned int)x; }
__device__ unsigned int __double2uint_rz(double x) { return (unsigned int)x; }
__device__ unsigned long long int __double2ull_rd(double x) { return (unsigned long long int)x; }
__device__ unsigned long long int __double2ull_rn(double x) { return (unsigned long long int)x; }
__device__ unsigned long long int __double2ull_ru(double x) { return (unsigned long long int)x; }
__device__ unsigned long long int __double2ull_rz(double x) { return (unsigned long long int)x; }
__device__ long long int __double_as_longlong(double x) {
static_assert(sizeof(long long) == sizeof(double), "");
long long tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ long long int __double2ll_rd(double x)
{
return (long long int)x;
__device__ int __float2int_rd(float x) { return (int)__ocml_floor_f32(x); }
__device__ int __float2int_rn(float x) { return (int)__ocml_rint_f32(x); }
__device__ int __float2int_ru(float x) { return (int)__ocml_ceil_f32(x); }
__device__ int __float2int_rz(float x) { return (int)__ocml_trunc_f32(x); }
__device__ long long int __float2ll_rd(float x) { return (long long int)x; }
__device__ long long int __float2ll_rn(float x) { return (long long int)x; }
__device__ long long int __float2ll_ru(float x) { return (long long int)x; }
__device__ long long int __float2ll_rz(float x) { return (long long int)x; }
__device__ unsigned int __float2uint_rd(float x) { return (unsigned int)x; }
__device__ unsigned int __float2uint_rn(float x) { return (unsigned int)x; }
__device__ unsigned int __float2uint_ru(float x) { return (unsigned int)x; }
__device__ unsigned int __float2uint_rz(float x) { return (unsigned int)x; }
__device__ unsigned long long int __float2ull_rd(float x) { return (unsigned long long int)x; }
__device__ unsigned long long int __float2ull_rn(float x) { return (unsigned long long int)x; }
__device__ unsigned long long int __float2ull_ru(float x) { return (unsigned long long int)x; }
__device__ unsigned long long int __float2ull_rz(float x) { return (unsigned long long int)x; }
__device__ int __float_as_int(float x) {
static_assert(sizeof(int) == sizeof(float), "");
int tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ long long int __double2ll_rn(double x)
{
return (long long int)x;
__device__ unsigned int __float_as_uint(float x) {
static_assert(sizeof(unsigned int) == sizeof(float), "");
unsigned int tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ long long int __double2ll_ru(double x)
{
return (long long int)x;
__device__ double __hiloint2double(int32_t hi, int32_t lo) {
static_assert(sizeof(double) == sizeof(uint64_t), "");
uint64_t tmp0 = (static_cast<uint64_t>(hi) << 32ull) | static_cast<uint32_t>(lo);
double tmp1;
__builtin_memcpy(&tmp1, &tmp0, sizeof(tmp0));
return tmp1;
}
__device__ long long int __double2ll_rz(double x)
{
return (long long int)x;
__device__ double __int2double_rn(int x) { return (double)x; }
__device__ float __int2float_rd(int x) { return (float)x; }
__device__ float __int2float_rn(int x) { return (float)x; }
__device__ float __int2float_ru(int x) { return (float)x; }
__device__ float __int2float_rz(int x) { return (float)x; }
__device__ float __int_as_float(int x) {
static_assert(sizeof(float) == sizeof(int), "");
float tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ double __ll2double_rd(long long int x) { return (double)x; }
__device__ double __ll2double_rn(long long int x) { return (double)x; }
__device__ double __ll2double_ru(long long int x) { return (double)x; }
__device__ double __ll2double_rz(long long int x) { return (double)x; }
__device__ unsigned int __double2uint_rd(double x)
{
return (unsigned int)x;
}
__device__ unsigned int __double2uint_rn(double x)
{
return (unsigned int)x;
}
__device__ unsigned int __double2uint_ru(double x)
{
return (unsigned int)x;
}
__device__ unsigned int __double2uint_rz(double x)
{
return (unsigned int)x;
__device__ float __ll2float_rd(long long int x) { return (float)x; }
__device__ float __ll2float_rn(long long int x) { return (float)x; }
__device__ float __ll2float_ru(long long int x) { return (float)x; }
__device__ float __ll2float_rz(long long int x) { return (float)x; }
__device__ double __longlong_as_double(long long int x) {
static_assert(sizeof(double) == sizeof(long long), "");
double tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return x;
}
__device__ unsigned long long int __double2ull_rd(double x)
{
return (unsigned long long int)x;
}
__device__ unsigned long long int __double2ull_rn(double x)
{
return (unsigned long long int)x;
}
__device__ unsigned long long int __double2ull_ru(double x)
{
return (unsigned long long int)x;
}
__device__ unsigned long long int __double2ull_rz(double x)
{
return (unsigned long long int)x;
__device__ double __uint2double_rn(int x) { return (double)x; }
__device__ float __uint2float_rd(unsigned int x) { return (float)x; }
__device__ float __uint2float_rn(unsigned int x) { return (float)x; }
__device__ float __uint2float_ru(unsigned int x) { return (float)x; }
__device__ float __uint2float_rz(unsigned int x) { return (float)x; }
__device__ float __uint_as_float(unsigned int x) {
static_assert(sizeof(float) == sizeof(unsigned int), "");
float tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ long long int __double_as_longlong(double x)
{
static_assert(sizeof(long long) == sizeof(double), "");
__device__ double __ull2double_rd(unsigned long long int x) { return (double)x; }
__device__ double __ull2double_rn(unsigned long long int x) { return (double)x; }
__device__ double __ull2double_ru(unsigned long long int x) { return (double)x; }
__device__ double __ull2double_rz(unsigned long long int x) { return (double)x; }
long long tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ int __float2int_rd(float x)
{
return (int)__ocml_floor_f32(x);
}
__device__ int __float2int_rn(float x)
{
return (int)__ocml_rint_f32(x);
}
__device__ int __float2int_ru(float x)
{
return (int)__ocml_ceil_f32(x);
}
__device__ int __float2int_rz(float x)
{
return (int)__ocml_trunc_f32(x);
}
__device__ long long int __float2ll_rd(float x)
{
return (long long int)x;
}
__device__ long long int __float2ll_rn(float x)
{
return (long long int)x;
}
__device__ long long int __float2ll_ru(float x)
{
return (long long int)x;
}
__device__ long long int __float2ll_rz(float x)
{
return (long long int)x;
}
__device__ unsigned int __float2uint_rd(float x)
{
return (unsigned int)x;
}
__device__ unsigned int __float2uint_rn(float x)
{
return (unsigned int)x;
}
__device__ unsigned int __float2uint_ru(float x)
{
return (unsigned int)x;
}
__device__ unsigned int __float2uint_rz(float x)
{
return (unsigned int)x;
}
__device__ unsigned long long int __float2ull_rd(float x)
{
return (unsigned long long int)x;
}
__device__ unsigned long long int __float2ull_rn(float x)
{
return (unsigned long long int)x;
}
__device__ unsigned long long int __float2ull_ru(float x)
{
return (unsigned long long int)x;
}
__device__ unsigned long long int __float2ull_rz(float x)
{
return (unsigned long long int)x;
}
__device__ int __float_as_int(float x)
{
static_assert(sizeof(int) == sizeof(float), "");
int tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ unsigned int __float_as_uint(float x)
{
static_assert(sizeof(unsigned int) == sizeof(float), "");
unsigned int tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ double __hiloint2double(int32_t hi, int32_t lo)
{
static_assert(sizeof(double) == sizeof(uint64_t), "");
uint64_t tmp0 =
(static_cast<uint64_t>(hi) << 32ull) | static_cast<uint32_t>(lo);
double tmp1;
__builtin_memcpy(&tmp1, &tmp0, sizeof(tmp0));
return tmp1;
}
__device__ double __int2double_rn(int x)
{
return (double)x;
}
__device__ float __int2float_rd(int x)
{
return (float)x;
}
__device__ float __int2float_rn(int x)
{
return (float)x;
}
__device__ float __int2float_ru(int x)
{
return (float)x;
}
__device__ float __int2float_rz(int x)
{
return (float)x;
}
__device__ float __int_as_float(int x)
{
static_assert(sizeof(float) == sizeof(int), "");
float tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ double __ll2double_rd(long long int x)
{
return (double)x;
}
__device__ double __ll2double_rn(long long int x)
{
return (double)x;
}
__device__ double __ll2double_ru(long long int x)
{
return (double)x;
}
__device__ double __ll2double_rz(long long int x)
{
return (double)x;
}
__device__ float __ll2float_rd(long long int x)
{
return (float)x;
}
__device__ float __ll2float_rn(long long int x)
{
return (float)x;
}
__device__ float __ll2float_ru(long long int x)
{
return (float)x;
}
__device__ float __ll2float_rz(long long int x)
{
return (float)x;
}
__device__ double __longlong_as_double(long long int x)
{
static_assert(sizeof(double) == sizeof(long long), "");
double tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return x;
}
__device__ double __uint2double_rn(int x)
{
return (double)x;
}
__device__ float __uint2float_rd(unsigned int x)
{
return (float)x;
}
__device__ float __uint2float_rn(unsigned int x)
{
return (float)x;
}
__device__ float __uint2float_ru(unsigned int x)
{
return (float)x;
}
__device__ float __uint2float_rz(unsigned int x)
{
return (float)x;
}
__device__ float __uint_as_float(unsigned int x)
{
static_assert(sizeof(float) == sizeof(unsigned int), "");
float tmp;
__builtin_memcpy(&tmp, &x, sizeof(tmp));
return tmp;
}
__device__ double __ull2double_rd(unsigned long long int x)
{
return (double)x;
}
__device__ double __ull2double_rn(unsigned long long int x)
{
return (double)x;
}
__device__ double __ull2double_ru(unsigned long long int x)
{
return (double)x;
}
__device__ double __ull2double_rz(unsigned long long int x)
{
return (double)x;
}
__device__ float __ull2float_rd(unsigned long long int x)
{
return (float)x;
}
__device__ float __ull2float_rn(unsigned long long int x)
{
return (float)x;
}
__device__ float __ull2float_ru(unsigned long long int x)
{
return (float)x;
}
__device__ float __ull2float_rz(unsigned long long int x)
{
return (float)x;
}
__device__ float __ull2float_rd(unsigned long long int x) { return (float)x; }
__device__ float __ull2float_rn(unsigned long long int x) { return (float)x; }
__device__ float __ull2float_ru(unsigned long long int x) { return (float)x; }
__device__ float __ull2float_rz(unsigned long long int x) { return (float)x; }
/*
Integer Intrinsics
*/
// integer intrinsic function __poc __clz __ffs __brev
__device__ unsigned int __popc( unsigned int input)
{
return hc::__popcount_u32_b32(input);
}
__device__ unsigned int __popc(unsigned int input) { return hc::__popcount_u32_b32(input); }
__device__ unsigned int __popcll( unsigned long long int input)
{
__device__ unsigned int __popcll(unsigned long long int input) {
return hc::__popcount_u32_b64(input);
}
__device__ unsigned int __clz(unsigned int input)
{
__device__ unsigned int __clz(unsigned int input) {
#ifdef NVCC_COMPAT
return input == 0 ? 32 : hc::__firstbit_u32_u32( input);
return input == 0 ? 32 : hc::__firstbit_u32_u32(input);
#else
return hc::__firstbit_u32_u32( input);
return hc::__firstbit_u32_u32(input);
#endif
}
__device__ unsigned int __clzll(unsigned long long int input)
{
__device__ unsigned int __clzll(unsigned long long int input) {
#ifdef NVCC_COMPAT
return input == 0 ? 64 : hc::__firstbit_u32_u64( input);
return input == 0 ? 64 : hc::__firstbit_u32_u64(input);
#else
return hc::__firstbit_u32_u64( input);
return hc::__firstbit_u32_u64(input);
#endif
}
__device__ unsigned int __clz( int input)
{
__device__ unsigned int __clz(int input) {
#ifdef NVCC_COMPAT
return input == 0 ? 32 : hc::__firstbit_u32_s32( input);
return input == 0 ? 32 : hc::__firstbit_u32_s32(input);
#else
return hc::__firstbit_u32_s32( input);
return hc::__firstbit_u32_s32(input);
#endif
}
__device__ unsigned int __clzll( long long int input)
{
__device__ unsigned int __clzll(long long int input) {
#ifdef NVCC_COMPAT
return input == 0 ? 64 : hc::__firstbit_u32_s64( input);
return input == 0 ? 64 : hc::__firstbit_u32_s64(input);
#else
return hc::__firstbit_u32_s64( input);
return hc::__firstbit_u32_s64(input);
#endif
}
__device__ unsigned int __ffs(unsigned int input)
{
__device__ unsigned int __ffs(unsigned int input) {
#ifdef NVCC_COMPAT
return hc::__lastbit_u32_u32( input)+1;
return hc::__lastbit_u32_u32(input) + 1;
#else
return hc::__lastbit_u32_u32( input);
return hc::__lastbit_u32_u32(input);
#endif
}
__device__ unsigned int __ffsll(unsigned long long int input)
{
__device__ unsigned int __ffsll(unsigned long long int input) {
#ifdef NVCC_COMPAT
return hc::__lastbit_u32_u64( input)+1;
return hc::__lastbit_u32_u64(input) + 1;
#else
return hc::__lastbit_u32_u64( input);
return hc::__lastbit_u32_u64(input);
#endif
}
__device__ unsigned int __ffs( int input)
{
__device__ unsigned int __ffs(int input) {
#ifdef NVCC_COMPAT
return hc::__lastbit_u32_s32( input)+1;
return hc::__lastbit_u32_s32(input) + 1;
#else
return hc::__lastbit_u32_s32( input);
return hc::__lastbit_u32_s32(input);
#endif
}
__device__ unsigned int __ffsll( long long int input)
{
__device__ unsigned int __ffsll(long long int input) {
#ifdef NVCC_COMPAT
return hc::__lastbit_u32_s64( input)+1;
return hc::__lastbit_u32_s64(input) + 1;
#else
return hc::__lastbit_u32_s64( input);
return hc::__lastbit_u32_s64(input);
#endif
}
__device__ unsigned int __brev( unsigned int input)
{
return hc::__bitrev_b32( input);
}
__device__ unsigned int __brev(unsigned int input) { return hc::__bitrev_b32(input); }
__device__ unsigned long long int __brevll( unsigned long long int input)
{
return hc::__bitrev_b64( input);
__device__ unsigned long long int __brevll(unsigned long long int input) {
return hc::__bitrev_b64(input);
}
struct ucharHolder {
union {
unsigned char c[4];
unsigned int ui;
};
}__attribute__((aligned(4)));
union {
unsigned char c[4];
unsigned int ui;
};
} __attribute__((aligned(4)));
struct uchar2Holder {
union {
unsigned int ui[2];
unsigned char c[8];
};
}__attribute__((aligned(8)));
union {
unsigned int ui[2];
unsigned char c[8];
};
} __attribute__((aligned(8)));
struct intHolder {
union {
signed int si[2];
signed int long sl;
};
}__attribute__((aligned(8)));
union {
signed int si[2];
signed int long sl;
};
} __attribute__((aligned(8)));
struct uintHolder {
union {
signed int ui[2];
signed int long ul;
};
}__attribute__((aligned(8)));
union {
signed int ui[2];
signed int long ul;
};
} __attribute__((aligned(8)));
__device__ unsigned int __byte_perm(unsigned int x, unsigned int y, unsigned int s)
{
struct uchar2Holder cHoldVal;
struct ucharHolder cHoldKey;
struct ucharHolder cHoldOut;
cHoldKey.ui = s;
cHoldVal.ui[0] = x;
cHoldVal.ui[1] = y;
cHoldOut.c[0] = cHoldVal.c[cHoldKey.c[0]];
cHoldOut.c[1] = cHoldVal.c[cHoldKey.c[1]];
cHoldOut.c[2] = cHoldVal.c[cHoldKey.c[2]];
cHoldOut.c[3] = cHoldVal.c[cHoldKey.c[3]];
return cHoldOut.ui;
__device__ unsigned int __byte_perm(unsigned int x, unsigned int y, unsigned int s) {
struct uchar2Holder cHoldVal;
struct ucharHolder cHoldKey;
struct ucharHolder cHoldOut;
cHoldKey.ui = s;
cHoldVal.ui[0] = x;
cHoldVal.ui[1] = y;
cHoldOut.c[0] = cHoldVal.c[cHoldKey.c[0]];
cHoldOut.c[1] = cHoldVal.c[cHoldKey.c[1]];
cHoldOut.c[2] = cHoldVal.c[cHoldKey.c[2]];
cHoldOut.c[3] = cHoldVal.c[cHoldKey.c[3]];
return cHoldOut.ui;
}
__device__ long long __mul64hi(long long int x, long long int y)
{
struct intHolder iHold1;
struct intHolder iHold2;
iHold1.sl = x;
iHold2.sl = y;
iHold1.sl = iHold1.si[1] * iHold2.si[1];
return iHold1.sl;
__device__ long long __mul64hi(long long int x, long long int y) {
struct intHolder iHold1;
struct intHolder iHold2;
iHold1.sl = x;
iHold2.sl = y;
iHold1.sl = iHold1.si[1] * iHold2.si[1];
return iHold1.sl;
}
__device__ unsigned long long __umul64hi(unsigned long long int x, unsigned long long int y)
{
struct uintHolder uHold1;
struct uintHolder uHold2;
uHold1.ul = x;
uHold2.ul = y;
uHold1.ul = uHold1.ui[1] * uHold2.ui[1];
return uHold1.ul;
__device__ unsigned long long __umul64hi(unsigned long long int x, unsigned long long int y) {
struct uintHolder uHold1;
struct uintHolder uHold2;
uHold1.ul = x;
uHold2.ul = y;
uHold1.ul = uHold1.ui[1] * uHold2.ui[1];
return uHold1.ul;
}
/*
File diff suppressed because it is too large Load Diff
+10 -11
View File
@@ -23,18 +23,18 @@ THE SOFTWARE.
#ifndef DEVICE_UTIL_H
#define DEVICE_UTIL_H
#include<hip/hcc_detail/hip_runtime.h>
#include <hip/hcc_detail/hip_runtime.h>
/*
Heap size computation for malloc and free device functions.
*/
#define NUM_PAGES_PER_THREAD 16
#define SIZE_OF_PAGE 64
#define NUM_THREADS_PER_CU 64
#define NUM_CUS_PER_GPU 64 // Specific for r9 Nano
#define NUM_PAGES NUM_PAGES_PER_THREAD * NUM_THREADS_PER_CU * NUM_CUS_PER_GPU
#define SIZE_MALLOC NUM_PAGES * SIZE_OF_PAGE
#define NUM_PAGES_PER_THREAD 16
#define SIZE_OF_PAGE 64
#define NUM_THREADS_PER_CU 64
#define NUM_CUS_PER_GPU 64 // Specific for r9 Nano
#define NUM_PAGES NUM_PAGES_PER_THREAD* NUM_THREADS_PER_CU* NUM_CUS_PER_GPU
#define SIZE_MALLOC NUM_PAGES* SIZE_OF_PAGE
#define SIZE_OF_HEAP SIZE_MALLOC
#define HIP_SQRT_2 1.41421356237
@@ -98,7 +98,7 @@ __device__ float __hip_precise_log10f(float x);
__device__ float __hip_precise_log2f(float x);
__device__ float __hip_precise_logf(float x);
__device__ float __hip_precise_powf(float base, float exponent);
__device__ void __hip_precise_sincosf(float x, float *s, float *c);
__device__ void __hip_precise_sincosf(float x, float* s, float* c);
__device__ float __hip_precise_sinf(float x);
__device__ float __hip_precise_tanf(float x);
// Double Precision Math
@@ -108,7 +108,6 @@ __device__ double __hip_precise_dsqrt_ru(double x);
__device__ double __hip_precise_dsqrt_rz(double x);
// Float Fast Math
__device__ float __hip_fast_exp10f(float x);
__device__ float __hip_fast_expf(float x);
@@ -119,14 +118,14 @@ __device__ float __hip_fast_fsqrt_rz(float x);
__device__ float __hip_fast_log10f(float x);
__device__ float __hip_fast_logf(float x);
__device__ float __hip_fast_powf(float base, float exponent);
__device__ void __hip_fast_sincosf(float x, float *s, float *c);
__device__ void __hip_fast_sincosf(float x, float* s, float* c);
__device__ float __hip_fast_tanf(float x);
// Double Precision Math
__device__ double __hip_fast_dsqrt_rd(double x);
__device__ double __hip_fast_dsqrt_rn(double x);
__device__ double __hip_fast_dsqrt_ru(double x);
__device__ double __hip_fast_dsqrt_rz(double x);
__device__ void __threadfence_system(void);
__device__ void __threadfence_system(void);
float __hip_host_j0f(float x);
double __hip_host_j0(double x);
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+135 -171
View File
@@ -26,228 +26,192 @@ THE SOFTWARE.
namespace ELFIO {
//------------------------------------------------------------------------------
class dynamic_section_accessor
{
public:
//------------------------------------------------------------------------------
dynamic_section_accessor( const elfio& elf_file_, section* section_ ) :
elf_file( elf_file_ ),
dynamic_section( section_ )
{
}
class dynamic_section_accessor {
public:
//------------------------------------------------------------------------------
dynamic_section_accessor(const elfio& elf_file_, section* section_)
: elf_file(elf_file_), dynamic_section(section_) {}
//------------------------------------------------------------------------------
Elf_Xword
get_entries_num() const
{
//------------------------------------------------------------------------------
Elf_Xword get_entries_num() const {
Elf_Xword nRet = 0;
if ( 0 != dynamic_section->get_entry_size() ) {
if (0 != dynamic_section->get_entry_size()) {
nRet = dynamic_section->get_size() / dynamic_section->get_entry_size();
}
return nRet;
}
//------------------------------------------------------------------------------
bool
get_entry( Elf_Xword index,
Elf_Xword& tag,
Elf_Xword& value,
std::string& str ) const
{
if ( index >= get_entries_num() ) { // Is index valid
//------------------------------------------------------------------------------
bool get_entry(Elf_Xword index, Elf_Xword& tag, Elf_Xword& value, std::string& str) const {
if (index >= get_entries_num()) { // Is index valid
return false;
}
if ( elf_file.get_class() == ELFCLASS32 ) {
generic_get_entry_dyn< Elf32_Dyn >( index, tag, value );
}
else {
generic_get_entry_dyn< Elf64_Dyn >( index, tag, value );
if (elf_file.get_class() == ELFCLASS32) {
generic_get_entry_dyn<Elf32_Dyn>(index, tag, value);
} else {
generic_get_entry_dyn<Elf64_Dyn>(index, tag, value);
}
// If the tag may have a string table reference, prepare the string
if ( tag == DT_NEEDED ||
tag == DT_SONAME ||
tag == DT_RPATH ||
tag == DT_RUNPATH ) {
string_section_accessor strsec =
elf_file.sections[ get_string_table_index() ];
const char* result = strsec.get_string( value );
if ( 0 == result ) {
if (tag == DT_NEEDED || tag == DT_SONAME || tag == DT_RPATH || tag == DT_RUNPATH) {
string_section_accessor strsec = elf_file.sections[get_string_table_index()];
const char* result = strsec.get_string(value);
if (0 == result) {
str.clear();
return false;
}
str = result;
}
else {
} else {
str.clear();
}
return true;
}
//------------------------------------------------------------------------------
void
add_entry( Elf_Xword& tag,
Elf_Xword& value )
{
if ( elf_file.get_class() == ELFCLASS32 ) {
generic_add_entry< Elf32_Dyn >( tag, value );
}
else {
generic_add_entry< Elf64_Dyn >( tag, value );
//------------------------------------------------------------------------------
void add_entry(Elf_Xword& tag, Elf_Xword& value) {
if (elf_file.get_class() == ELFCLASS32) {
generic_add_entry<Elf32_Dyn>(tag, value);
} else {
generic_add_entry<Elf64_Dyn>(tag, value);
}
}
//------------------------------------------------------------------------------
void
add_entry( Elf_Xword& tag,
std::string& str )
{
string_section_accessor strsec =
elf_file.sections[ get_string_table_index() ];
Elf_Xword value = strsec.add_string( str );
add_entry( tag, value );
//------------------------------------------------------------------------------
void add_entry(Elf_Xword& tag, std::string& str) {
string_section_accessor strsec = elf_file.sections[get_string_table_index()];
Elf_Xword value = strsec.add_string(str);
add_entry(tag, value);
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
Elf_Half
get_string_table_index() const
{
return (Elf_Half)dynamic_section->get_link();
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
Elf_Half get_string_table_index() const { return (Elf_Half)dynamic_section->get_link(); }
//------------------------------------------------------------------------------
template< class T >
void
generic_get_entry_dyn( Elf_Xword index,
Elf_Xword& tag,
Elf_Xword& value ) const
{
//------------------------------------------------------------------------------
template <class T>
void generic_get_entry_dyn(Elf_Xword index, Elf_Xword& tag, Elf_Xword& value) const {
const endianess_convertor& convertor = elf_file.get_convertor();
// Check unusual case when dynamic section has no data
if( dynamic_section->get_data() == 0 ||
( index + 1 ) * dynamic_section->get_entry_size() > dynamic_section->get_size() ) {
tag = DT_NULL;
if (dynamic_section->get_data() == 0 ||
(index + 1) * dynamic_section->get_entry_size() > dynamic_section->get_size()) {
tag = DT_NULL;
value = 0;
return;
}
const T* pEntry = reinterpret_cast<const T*>(
dynamic_section->get_data() +
index * dynamic_section->get_entry_size() );
tag = convertor( pEntry->d_tag );
switch ( tag ) {
case DT_NULL:
case DT_SYMBOLIC:
case DT_TEXTREL:
case DT_BIND_NOW:
value = 0;
break;
case DT_NEEDED:
case DT_PLTRELSZ:
case DT_RELASZ:
case DT_RELAENT:
case DT_STRSZ:
case DT_SYMENT:
case DT_SONAME:
case DT_RPATH:
case DT_RELSZ:
case DT_RELENT:
case DT_PLTREL:
case DT_INIT_ARRAYSZ:
case DT_FINI_ARRAYSZ:
case DT_RUNPATH:
case DT_FLAGS:
case DT_PREINIT_ARRAYSZ:
value = convertor( pEntry->d_un.d_val );
break;
case DT_PLTGOT:
case DT_HASH:
case DT_STRTAB:
case DT_SYMTAB:
case DT_RELA:
case DT_INIT:
case DT_FINI:
case DT_REL:
case DT_DEBUG:
case DT_JMPREL:
case DT_INIT_ARRAY:
case DT_FINI_ARRAY:
case DT_PREINIT_ARRAY:
default:
value = convertor( pEntry->d_un.d_ptr );
break;
const T* pEntry = reinterpret_cast<const T*>(dynamic_section->get_data() +
index * dynamic_section->get_entry_size());
tag = convertor(pEntry->d_tag);
switch (tag) {
case DT_NULL:
case DT_SYMBOLIC:
case DT_TEXTREL:
case DT_BIND_NOW:
value = 0;
break;
case DT_NEEDED:
case DT_PLTRELSZ:
case DT_RELASZ:
case DT_RELAENT:
case DT_STRSZ:
case DT_SYMENT:
case DT_SONAME:
case DT_RPATH:
case DT_RELSZ:
case DT_RELENT:
case DT_PLTREL:
case DT_INIT_ARRAYSZ:
case DT_FINI_ARRAYSZ:
case DT_RUNPATH:
case DT_FLAGS:
case DT_PREINIT_ARRAYSZ:
value = convertor(pEntry->d_un.d_val);
break;
case DT_PLTGOT:
case DT_HASH:
case DT_STRTAB:
case DT_SYMTAB:
case DT_RELA:
case DT_INIT:
case DT_FINI:
case DT_REL:
case DT_DEBUG:
case DT_JMPREL:
case DT_INIT_ARRAY:
case DT_FINI_ARRAY:
case DT_PREINIT_ARRAY:
default:
value = convertor(pEntry->d_un.d_ptr);
break;
}
}
//------------------------------------------------------------------------------
template< class T >
void
generic_add_entry( Elf_Xword tag, Elf_Xword value )
{
//------------------------------------------------------------------------------
template <class T>
void generic_add_entry(Elf_Xword tag, Elf_Xword value) {
const endianess_convertor& convertor = elf_file.get_convertor();
T entry;
switch ( tag ) {
case DT_NULL:
case DT_SYMBOLIC:
case DT_TEXTREL:
case DT_BIND_NOW:
value = 0;
case DT_NEEDED:
case DT_PLTRELSZ:
case DT_RELASZ:
case DT_RELAENT:
case DT_STRSZ:
case DT_SYMENT:
case DT_SONAME:
case DT_RPATH:
case DT_RELSZ:
case DT_RELENT:
case DT_PLTREL:
case DT_INIT_ARRAYSZ:
case DT_FINI_ARRAYSZ:
case DT_RUNPATH:
case DT_FLAGS:
case DT_PREINIT_ARRAYSZ:
entry.d_un.d_val = convertor( value );
break;
case DT_PLTGOT:
case DT_HASH:
case DT_STRTAB:
case DT_SYMTAB:
case DT_RELA:
case DT_INIT:
case DT_FINI:
case DT_REL:
case DT_DEBUG:
case DT_JMPREL:
case DT_INIT_ARRAY:
case DT_FINI_ARRAY:
case DT_PREINIT_ARRAY:
default:
entry.d_un.d_ptr = convertor( value );
break;
switch (tag) {
case DT_NULL:
case DT_SYMBOLIC:
case DT_TEXTREL:
case DT_BIND_NOW:
value = 0;
case DT_NEEDED:
case DT_PLTRELSZ:
case DT_RELASZ:
case DT_RELAENT:
case DT_STRSZ:
case DT_SYMENT:
case DT_SONAME:
case DT_RPATH:
case DT_RELSZ:
case DT_RELENT:
case DT_PLTREL:
case DT_INIT_ARRAYSZ:
case DT_FINI_ARRAYSZ:
case DT_RUNPATH:
case DT_FLAGS:
case DT_PREINIT_ARRAYSZ:
entry.d_un.d_val = convertor(value);
break;
case DT_PLTGOT:
case DT_HASH:
case DT_STRTAB:
case DT_SYMTAB:
case DT_RELA:
case DT_INIT:
case DT_FINI:
case DT_REL:
case DT_DEBUG:
case DT_JMPREL:
case DT_INIT_ARRAY:
case DT_FINI_ARRAY:
case DT_PREINIT_ARRAY:
default:
entry.d_un.d_ptr = convertor(value);
break;
}
entry.d_tag = convertor( tag );
entry.d_tag = convertor(tag);
dynamic_section->append_data( reinterpret_cast<char*>( &entry ), sizeof( entry ) );
dynamic_section->append_data(reinterpret_cast<char*>(&entry), sizeof(entry));
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
private:
const elfio& elf_file;
section* dynamic_section;
section* dynamic_section;
};
} // namespace ELFIO
} // namespace ELFIO
#endif // ELFIO_DYNAMIC_HPP
#endif // ELFIO_DYNAMIC_HPP
+77 -81
View File
@@ -27,120 +27,116 @@ THE SOFTWARE.
namespace ELFIO {
class elf_header
{
public:
virtual ~elf_header() {};
virtual bool load( std::istream& stream ) = 0;
virtual bool save( std::ostream& stream ) const = 0;
class elf_header {
public:
virtual ~elf_header(){};
virtual bool load(std::istream& stream) = 0;
virtual bool save(std::ostream& stream) const = 0;
// ELF header functions
ELFIO_GET_ACCESS_DECL( unsigned char, class );
ELFIO_GET_ACCESS_DECL( unsigned char, elf_version );
ELFIO_GET_ACCESS_DECL( unsigned char, encoding );
ELFIO_GET_ACCESS_DECL( Elf_Word, version );
ELFIO_GET_ACCESS_DECL( Elf_Half, header_size );
ELFIO_GET_ACCESS_DECL( Elf_Half, section_entry_size );
ELFIO_GET_ACCESS_DECL( Elf_Half, segment_entry_size );
ELFIO_GET_ACCESS_DECL(unsigned char, class);
ELFIO_GET_ACCESS_DECL(unsigned char, elf_version);
ELFIO_GET_ACCESS_DECL(unsigned char, encoding);
ELFIO_GET_ACCESS_DECL(Elf_Word, version);
ELFIO_GET_ACCESS_DECL(Elf_Half, header_size);
ELFIO_GET_ACCESS_DECL(Elf_Half, section_entry_size);
ELFIO_GET_ACCESS_DECL(Elf_Half, segment_entry_size);
ELFIO_GET_SET_ACCESS_DECL( unsigned char, os_abi );
ELFIO_GET_SET_ACCESS_DECL( unsigned char, abi_version );
ELFIO_GET_SET_ACCESS_DECL( Elf_Half, type );
ELFIO_GET_SET_ACCESS_DECL( Elf_Half, machine );
ELFIO_GET_SET_ACCESS_DECL( Elf_Word, flags );
ELFIO_GET_SET_ACCESS_DECL( Elf64_Addr, entry );
ELFIO_GET_SET_ACCESS_DECL( Elf_Half, sections_num );
ELFIO_GET_SET_ACCESS_DECL( Elf64_Off, sections_offset );
ELFIO_GET_SET_ACCESS_DECL( Elf_Half, segments_num );
ELFIO_GET_SET_ACCESS_DECL( Elf64_Off, segments_offset );
ELFIO_GET_SET_ACCESS_DECL( Elf_Half, section_name_str_index );
ELFIO_GET_SET_ACCESS_DECL(unsigned char, os_abi);
ELFIO_GET_SET_ACCESS_DECL(unsigned char, abi_version);
ELFIO_GET_SET_ACCESS_DECL(Elf_Half, type);
ELFIO_GET_SET_ACCESS_DECL(Elf_Half, machine);
ELFIO_GET_SET_ACCESS_DECL(Elf_Word, flags);
ELFIO_GET_SET_ACCESS_DECL(Elf64_Addr, entry);
ELFIO_GET_SET_ACCESS_DECL(Elf_Half, sections_num);
ELFIO_GET_SET_ACCESS_DECL(Elf64_Off, sections_offset);
ELFIO_GET_SET_ACCESS_DECL(Elf_Half, segments_num);
ELFIO_GET_SET_ACCESS_DECL(Elf64_Off, segments_offset);
ELFIO_GET_SET_ACCESS_DECL(Elf_Half, section_name_str_index);
};
template< class T > struct elf_header_impl_types;
template<> struct elf_header_impl_types<Elf32_Ehdr> {
template <class T>
struct elf_header_impl_types;
template <>
struct elf_header_impl_types<Elf32_Ehdr> {
typedef Elf32_Phdr Phdr_type;
typedef Elf32_Shdr Shdr_type;
static const unsigned char file_class = ELFCLASS32;
};
template<> struct elf_header_impl_types<Elf64_Ehdr> {
template <>
struct elf_header_impl_types<Elf64_Ehdr> {
typedef Elf64_Phdr Phdr_type;
typedef Elf64_Shdr Shdr_type;
static const unsigned char file_class = ELFCLASS64;
};
template< class T > class elf_header_impl : public elf_header
{
public:
elf_header_impl( endianess_convertor* convertor_,
unsigned char encoding )
{
template <class T>
class elf_header_impl : public elf_header {
public:
elf_header_impl(endianess_convertor* convertor_, unsigned char encoding) {
convertor = convertor_;
std::fill_n( reinterpret_cast<char*>( &header ), sizeof( header ), '\0' );
std::fill_n(reinterpret_cast<char*>(&header), sizeof(header), '\0');
header.e_ident[EI_MAG0] = ELFMAG0;
header.e_ident[EI_MAG1] = ELFMAG1;
header.e_ident[EI_MAG2] = ELFMAG2;
header.e_ident[EI_MAG3] = ELFMAG3;
header.e_ident[EI_CLASS] = elf_header_impl_types<T>::file_class;
header.e_ident[EI_DATA] = encoding;
header.e_ident[EI_MAG0] = ELFMAG0;
header.e_ident[EI_MAG1] = ELFMAG1;
header.e_ident[EI_MAG2] = ELFMAG2;
header.e_ident[EI_MAG3] = ELFMAG3;
header.e_ident[EI_CLASS] = elf_header_impl_types<T>::file_class;
header.e_ident[EI_DATA] = encoding;
header.e_ident[EI_VERSION] = EV_CURRENT;
header.e_version = EV_CURRENT;
header.e_version = (*convertor)( header.e_version );
header.e_ehsize = ( sizeof( header ) );
header.e_ehsize = (*convertor)( header.e_ehsize );
header.e_shstrndx = (*convertor)( (Elf_Half)1 );
header.e_phentsize = sizeof( typename elf_header_impl_types<T>::Phdr_type );
header.e_shentsize = sizeof( typename elf_header_impl_types<T>::Shdr_type );
header.e_phentsize = (*convertor)( header.e_phentsize );
header.e_shentsize = (*convertor)( header.e_shentsize );
header.e_version = EV_CURRENT;
header.e_version = (*convertor)(header.e_version);
header.e_ehsize = (sizeof(header));
header.e_ehsize = (*convertor)(header.e_ehsize);
header.e_shstrndx = (*convertor)((Elf_Half)1);
header.e_phentsize = sizeof(typename elf_header_impl_types<T>::Phdr_type);
header.e_shentsize = sizeof(typename elf_header_impl_types<T>::Shdr_type);
header.e_phentsize = (*convertor)(header.e_phentsize);
header.e_shentsize = (*convertor)(header.e_shentsize);
}
bool
load( std::istream& stream )
{
stream.seekg( 0 );
stream.read( reinterpret_cast<char*>( &header ), sizeof( header ) );
bool load(std::istream& stream) {
stream.seekg(0);
stream.read(reinterpret_cast<char*>(&header), sizeof(header));
return (stream.gcount() == sizeof( header ) );
return (stream.gcount() == sizeof(header));
}
bool
save( std::ostream& stream ) const
{
stream.seekp( 0 );
stream.write( reinterpret_cast<const char*>( &header ), sizeof( header ) );
bool save(std::ostream& stream) const {
stream.seekp(0);
stream.write(reinterpret_cast<const char*>(&header), sizeof(header));
return stream.good();
}
// ELF header functions
ELFIO_GET_ACCESS( unsigned char, class, header.e_ident[EI_CLASS] );
ELFIO_GET_ACCESS( unsigned char, elf_version, header.e_ident[EI_VERSION] );
ELFIO_GET_ACCESS( unsigned char, encoding, header.e_ident[EI_DATA] );
ELFIO_GET_ACCESS( Elf_Word, version, header.e_version );
ELFIO_GET_ACCESS( Elf_Half, header_size, header.e_ehsize );
ELFIO_GET_ACCESS( Elf_Half, section_entry_size, header.e_shentsize );
ELFIO_GET_ACCESS( Elf_Half, segment_entry_size, header.e_phentsize );
ELFIO_GET_ACCESS(unsigned char, class, header.e_ident[EI_CLASS]);
ELFIO_GET_ACCESS(unsigned char, elf_version, header.e_ident[EI_VERSION]);
ELFIO_GET_ACCESS(unsigned char, encoding, header.e_ident[EI_DATA]);
ELFIO_GET_ACCESS(Elf_Word, version, header.e_version);
ELFIO_GET_ACCESS(Elf_Half, header_size, header.e_ehsize);
ELFIO_GET_ACCESS(Elf_Half, section_entry_size, header.e_shentsize);
ELFIO_GET_ACCESS(Elf_Half, segment_entry_size, header.e_phentsize);
ELFIO_GET_SET_ACCESS( unsigned char, os_abi, header.e_ident[EI_OSABI] );
ELFIO_GET_SET_ACCESS( unsigned char, abi_version, header.e_ident[EI_ABIVERSION] );
ELFIO_GET_SET_ACCESS( Elf_Half, type, header.e_type );
ELFIO_GET_SET_ACCESS( Elf_Half, machine, header.e_machine );
ELFIO_GET_SET_ACCESS( Elf_Word, flags, header.e_flags );
ELFIO_GET_SET_ACCESS( Elf_Half, section_name_str_index, header.e_shstrndx );
ELFIO_GET_SET_ACCESS( Elf64_Addr, entry, header.e_entry );
ELFIO_GET_SET_ACCESS( Elf_Half, sections_num, header.e_shnum );
ELFIO_GET_SET_ACCESS( Elf64_Off, sections_offset, header.e_shoff );
ELFIO_GET_SET_ACCESS( Elf_Half, segments_num, header.e_phnum );
ELFIO_GET_SET_ACCESS( Elf64_Off, segments_offset, header.e_phoff );
ELFIO_GET_SET_ACCESS(unsigned char, os_abi, header.e_ident[EI_OSABI]);
ELFIO_GET_SET_ACCESS(unsigned char, abi_version, header.e_ident[EI_ABIVERSION]);
ELFIO_GET_SET_ACCESS(Elf_Half, type, header.e_type);
ELFIO_GET_SET_ACCESS(Elf_Half, machine, header.e_machine);
ELFIO_GET_SET_ACCESS(Elf_Word, flags, header.e_flags);
ELFIO_GET_SET_ACCESS(Elf_Half, section_name_str_index, header.e_shstrndx);
ELFIO_GET_SET_ACCESS(Elf64_Addr, entry, header.e_entry);
ELFIO_GET_SET_ACCESS(Elf_Half, sections_num, header.e_shnum);
ELFIO_GET_SET_ACCESS(Elf64_Off, sections_offset, header.e_shoff);
ELFIO_GET_SET_ACCESS(Elf_Half, segments_num, header.e_phnum);
ELFIO_GET_SET_ACCESS(Elf64_Off, segments_offset, header.e_phoff);
private:
private:
T header;
endianess_convertor* convertor;
};
} // namespace ELFIO
} // namespace ELFIO
#endif // ELF_HEADER_HPP
#endif // ELF_HEADER_HPP
+61 -83
View File
@@ -38,129 +38,107 @@ namespace ELFIO {
//------------------------------------------------------------------------------
//------------------------------------------------------------------------------
class note_section_accessor
{
public:
//------------------------------------------------------------------------------
note_section_accessor( const elfio& elf_file_, section* section_ ) :
elf_file( elf_file_ ), note_section( section_ )
{
class note_section_accessor {
public:
//------------------------------------------------------------------------------
note_section_accessor(const elfio& elf_file_, section* section_)
: elf_file(elf_file_), note_section(section_) {
process_section();
}
//------------------------------------------------------------------------------
Elf_Word
get_notes_num() const
{
return (Elf_Word)note_start_positions.size();
}
//------------------------------------------------------------------------------
Elf_Word get_notes_num() const { return (Elf_Word)note_start_positions.size(); }
//------------------------------------------------------------------------------
bool
get_note( Elf_Word index,
Elf_Word& type,
std::string& name,
void*& desc,
Elf_Word& descSize ) const
{
if ( index >= note_section->get_size() ) {
//------------------------------------------------------------------------------
bool get_note(Elf_Word index, Elf_Word& type, std::string& name, void*& desc,
Elf_Word& descSize) const {
if (index >= note_section->get_size()) {
return false;
}
const char* pData = note_section->get_data() + note_start_positions[index];
int align = sizeof( Elf_Word );
int align = sizeof(Elf_Word);
const endianess_convertor& convertor = elf_file.get_convertor();
type = convertor( *(Elf_Word*)( pData + 2*align ) );
Elf_Word namesz = convertor( *(Elf_Word*)( pData ) );
descSize = convertor( *(Elf_Word*)( pData + sizeof( namesz ) ) );
type = convertor(*(Elf_Word*)(pData + 2 * align));
Elf_Word namesz = convertor(*(Elf_Word*)(pData));
descSize = convertor(*(Elf_Word*)(pData + sizeof(namesz)));
Elf_Word max_name_size = note_section->get_size() - note_start_positions[index];
if ( namesz > max_name_size ||
namesz + descSize > max_name_size ) {
if (namesz > max_name_size || namesz + descSize > max_name_size) {
return false;
}
name.assign( pData + 3*align, namesz - 1);
if ( 0 == descSize ) {
name.assign(pData + 3 * align, namesz - 1);
if (0 == descSize) {
desc = 0;
}
else {
desc = const_cast<char*> ( pData + 3*align +
( ( namesz + align - 1 )/align )*align );
} else {
desc = const_cast<char*>(pData + 3 * align + ((namesz + align - 1) / align) * align);
}
return true;
}
//------------------------------------------------------------------------------
void add_note( Elf_Word type,
const std::string& name,
const void* desc,
Elf_Word descSize )
{
//------------------------------------------------------------------------------
void add_note(Elf_Word type, const std::string& name, const void* desc, Elf_Word descSize) {
const endianess_convertor& convertor = elf_file.get_convertor();
int align = sizeof( Elf_Word );
Elf_Word nameLen = (Elf_Word)name.size() + 1;
Elf_Word nameLenConv = convertor( nameLen );
std::string buffer( reinterpret_cast<char*>( &nameLenConv ), align );
Elf_Word descSizeConv = convertor( descSize );
buffer.append( reinterpret_cast<char*>( &descSizeConv ), align );
type = convertor( type );
buffer.append( reinterpret_cast<char*>( &type ), align );
buffer.append( name );
buffer.append( 1, '\x00' );
const char pad[] = { '\0', '\0', '\0', '\0' };
if ( nameLen % align != 0 ) {
buffer.append( pad, align - nameLen % align );
int align = sizeof(Elf_Word);
Elf_Word nameLen = (Elf_Word)name.size() + 1;
Elf_Word nameLenConv = convertor(nameLen);
std::string buffer(reinterpret_cast<char*>(&nameLenConv), align);
Elf_Word descSizeConv = convertor(descSize);
buffer.append(reinterpret_cast<char*>(&descSizeConv), align);
type = convertor(type);
buffer.append(reinterpret_cast<char*>(&type), align);
buffer.append(name);
buffer.append(1, '\x00');
const char pad[] = {'\0', '\0', '\0', '\0'};
if (nameLen % align != 0) {
buffer.append(pad, align - nameLen % align);
}
if ( desc != 0 && descSize != 0 ) {
buffer.append( reinterpret_cast<const char*>( desc ), descSize );
if ( descSize % align != 0 ) {
buffer.append( pad, align - descSize % align );
if (desc != 0 && descSize != 0) {
buffer.append(reinterpret_cast<const char*>(desc), descSize);
if (descSize % align != 0) {
buffer.append(pad, align - descSize % align);
}
}
note_start_positions.push_back( note_section->get_size() );
note_section->append_data( buffer );
note_start_positions.push_back(note_section->get_size());
note_section->append_data(buffer);
}
private:
//------------------------------------------------------------------------------
void process_section()
{
private:
//------------------------------------------------------------------------------
void process_section() {
const endianess_convertor& convertor = elf_file.get_convertor();
const char* data = note_section->get_data();
Elf_Xword size = note_section->get_size();
Elf_Xword current = 0;
const char* data = note_section->get_data();
Elf_Xword size = note_section->get_size();
Elf_Xword current = 0;
note_start_positions.clear();
// Is it empty?
if ( 0 == data || 0 == size ) {
if (0 == data || 0 == size) {
return;
}
int align = sizeof( Elf_Word );
while ( current + 3*align <= size ) {
note_start_positions.push_back( current );
Elf_Word namesz = convertor(
*(Elf_Word*)( data + current ) );
Elf_Word descsz = convertor(
*(Elf_Word*)( data + current + sizeof( namesz ) ) );
int align = sizeof(Elf_Word);
while (current + 3 * align <= size) {
note_start_positions.push_back(current);
Elf_Word namesz = convertor(*(Elf_Word*)(data + current));
Elf_Word descsz = convertor(*(Elf_Word*)(data + current + sizeof(namesz)));
current += 3*sizeof( Elf_Word ) +
( ( namesz + align - 1 ) / align ) * align +
( ( descsz + align - 1 ) / align ) * align;
current += 3 * sizeof(Elf_Word) + ((namesz + align - 1) / align) * align +
((descsz + align - 1) / align) * align;
}
}
//------------------------------------------------------------------------------
private:
const elfio& elf_file;
section* note_section;
//------------------------------------------------------------------------------
private:
const elfio& elf_file;
section* note_section;
std::vector<Elf_Xword> note_start_positions;
};
} // namespace ELFIO
} // namespace ELFIO
#endif // ELFIO_NOTE_HPP
#endif // ELFIO_NOTE_HPP
+166 -255
View File
@@ -25,345 +25,256 @@ THE SOFTWARE.
namespace ELFIO {
template<typename T> struct get_sym_and_type;
template<> struct get_sym_and_type< Elf32_Rel >
{
static int get_r_sym( Elf_Xword info )
{
return ELF32_R_SYM( (Elf_Word)info );
}
static int get_r_type( Elf_Xword info )
{
return ELF32_R_TYPE( (Elf_Word)info );
}
template <typename T>
struct get_sym_and_type;
template <>
struct get_sym_and_type<Elf32_Rel> {
static int get_r_sym(Elf_Xword info) { return ELF32_R_SYM((Elf_Word)info); }
static int get_r_type(Elf_Xword info) { return ELF32_R_TYPE((Elf_Word)info); }
};
template<> struct get_sym_and_type< Elf32_Rela >
{
static int get_r_sym( Elf_Xword info )
{
return ELF32_R_SYM( (Elf_Word)info );
}
static int get_r_type( Elf_Xword info )
{
return ELF32_R_TYPE( (Elf_Word)info );
}
template <>
struct get_sym_and_type<Elf32_Rela> {
static int get_r_sym(Elf_Xword info) { return ELF32_R_SYM((Elf_Word)info); }
static int get_r_type(Elf_Xword info) { return ELF32_R_TYPE((Elf_Word)info); }
};
template<> struct get_sym_and_type< Elf64_Rel >
{
static int get_r_sym( Elf_Xword info )
{
return ELF64_R_SYM( info );
}
static int get_r_type( Elf_Xword info )
{
return ELF64_R_TYPE( info );
}
template <>
struct get_sym_and_type<Elf64_Rel> {
static int get_r_sym(Elf_Xword info) { return ELF64_R_SYM(info); }
static int get_r_type(Elf_Xword info) { return ELF64_R_TYPE(info); }
};
template<> struct get_sym_and_type< Elf64_Rela >
{
static int get_r_sym( Elf_Xword info )
{
return ELF64_R_SYM( info );
}
static int get_r_type( Elf_Xword info )
{
return ELF64_R_TYPE( info );
}
template <>
struct get_sym_and_type<Elf64_Rela> {
static int get_r_sym(Elf_Xword info) { return ELF64_R_SYM(info); }
static int get_r_type(Elf_Xword info) { return ELF64_R_TYPE(info); }
};
//------------------------------------------------------------------------------
class relocation_section_accessor
{
public:
//------------------------------------------------------------------------------
relocation_section_accessor( const elfio& elf_file_, section* section_ ) :
elf_file( elf_file_ ),
relocation_section( section_ )
{
}
class relocation_section_accessor {
public:
//------------------------------------------------------------------------------
relocation_section_accessor(const elfio& elf_file_, section* section_)
: elf_file(elf_file_), relocation_section(section_) {}
//------------------------------------------------------------------------------
Elf_Xword
get_entries_num() const
{
//------------------------------------------------------------------------------
Elf_Xword get_entries_num() const {
Elf_Xword nRet = 0;
if ( 0 != relocation_section->get_entry_size() ) {
if (0 != relocation_section->get_entry_size()) {
nRet = relocation_section->get_size() / relocation_section->get_entry_size();
}
return nRet;
}
//------------------------------------------------------------------------------
bool
get_entry( Elf_Xword index,
Elf64_Addr& offset,
Elf_Word& symbol,
Elf_Word& type,
Elf_Sxword& addend ) const
{
if ( index >= get_entries_num() ) { // Is index valid
//------------------------------------------------------------------------------
bool get_entry(Elf_Xword index, Elf64_Addr& offset, Elf_Word& symbol, Elf_Word& type,
Elf_Sxword& addend) const {
if (index >= get_entries_num()) { // Is index valid
return false;
}
if ( elf_file.get_class() == ELFCLASS32 ) {
if ( SHT_REL == relocation_section->get_type() ) {
generic_get_entry_rel< Elf32_Rel >( index, offset, symbol,
type, addend );
if (elf_file.get_class() == ELFCLASS32) {
if (SHT_REL == relocation_section->get_type()) {
generic_get_entry_rel<Elf32_Rel>(index, offset, symbol, type, addend);
} else if (SHT_RELA == relocation_section->get_type()) {
generic_get_entry_rela<Elf32_Rela>(index, offset, symbol, type, addend);
}
else if ( SHT_RELA == relocation_section->get_type() ) {
generic_get_entry_rela< Elf32_Rela >( index, offset, symbol,
type, addend );
}
}
else {
if ( SHT_REL == relocation_section->get_type() ) {
generic_get_entry_rel< Elf64_Rel >( index, offset, symbol,
type, addend );
}
else if ( SHT_RELA == relocation_section->get_type() ) {
generic_get_entry_rela< Elf64_Rela >( index, offset, symbol,
type, addend );
} else {
if (SHT_REL == relocation_section->get_type()) {
generic_get_entry_rel<Elf64_Rel>(index, offset, symbol, type, addend);
} else if (SHT_RELA == relocation_section->get_type()) {
generic_get_entry_rela<Elf64_Rela>(index, offset, symbol, type, addend);
}
}
return true;
}
//------------------------------------------------------------------------------
bool
get_entry( Elf_Xword index,
Elf64_Addr& offset,
Elf64_Addr& symbolValue,
std::string& symbolName,
Elf_Word& type,
Elf_Sxword& addend,
Elf_Sxword& calcValue ) const
{
//------------------------------------------------------------------------------
bool get_entry(Elf_Xword index, Elf64_Addr& offset, Elf64_Addr& symbolValue,
std::string& symbolName, Elf_Word& type, Elf_Sxword& addend,
Elf_Sxword& calcValue) const {
// Do regular job
Elf_Word symbol;
bool ret = get_entry( index, offset, symbol, type, addend );
bool ret = get_entry(index, offset, symbol, type, addend);
// Find the symbol
Elf_Xword size;
Elf_Xword size;
unsigned char bind;
unsigned char symbolType;
Elf_Half section;
Elf_Half section;
unsigned char other;
symbol_section_accessor symbols( elf_file, elf_file.sections[get_symbol_table_index()] );
ret = ret && symbols.get_symbol( symbol, symbolName, symbolValue,
size, bind, symbolType, section, other );
symbol_section_accessor symbols(elf_file, elf_file.sections[get_symbol_table_index()]);
ret = ret && symbols.get_symbol(symbol, symbolName, symbolValue, size, bind, symbolType,
section, other);
if ( ret ) { // Was it successful?
switch ( type ) {
case R_386_NONE: // none
calcValue = 0;
break;
case R_386_32: // S + A
calcValue = symbolValue + addend;
break;
case R_386_PC32: // S + A - P
calcValue = symbolValue + addend - offset;
break;
case R_386_GOT32: // G + A - P
calcValue = 0;
break;
case R_386_PLT32: // L + A - P
calcValue = 0;
break;
case R_386_COPY: // none
calcValue = 0;
break;
case R_386_GLOB_DAT: // S
case R_386_JMP_SLOT: // S
calcValue = symbolValue;
break;
case R_386_RELATIVE: // B + A
calcValue = addend;
break;
case R_386_GOTOFF: // S + A - GOT
calcValue = 0;
break;
case R_386_GOTPC: // GOT + A - P
calcValue = 0;
break;
default: // Not recognized symbol!
calcValue = 0;
break;
if (ret) { // Was it successful?
switch (type) {
case R_386_NONE: // none
calcValue = 0;
break;
case R_386_32: // S + A
calcValue = symbolValue + addend;
break;
case R_386_PC32: // S + A - P
calcValue = symbolValue + addend - offset;
break;
case R_386_GOT32: // G + A - P
calcValue = 0;
break;
case R_386_PLT32: // L + A - P
calcValue = 0;
break;
case R_386_COPY: // none
calcValue = 0;
break;
case R_386_GLOB_DAT: // S
case R_386_JMP_SLOT: // S
calcValue = symbolValue;
break;
case R_386_RELATIVE: // B + A
calcValue = addend;
break;
case R_386_GOTOFF: // S + A - GOT
calcValue = 0;
break;
case R_386_GOTPC: // GOT + A - P
calcValue = 0;
break;
default: // Not recognized symbol!
calcValue = 0;
break;
}
}
return ret;
}
//------------------------------------------------------------------------------
void
add_entry( Elf64_Addr offset, Elf_Xword info )
{
if ( elf_file.get_class() == ELFCLASS32 ) {
generic_add_entry< Elf32_Rel >( offset, info );
}
else {
generic_add_entry< Elf64_Rel >( offset, info );
//------------------------------------------------------------------------------
void add_entry(Elf64_Addr offset, Elf_Xword info) {
if (elf_file.get_class() == ELFCLASS32) {
generic_add_entry<Elf32_Rel>(offset, info);
} else {
generic_add_entry<Elf64_Rel>(offset, info);
}
}
//------------------------------------------------------------------------------
void
add_entry( Elf64_Addr offset, Elf_Word symbol, unsigned char type )
{
//------------------------------------------------------------------------------
void add_entry(Elf64_Addr offset, Elf_Word symbol, unsigned char type) {
Elf_Xword info;
if ( elf_file.get_class() == ELFCLASS32 ) {
info = ELF32_R_INFO( (Elf_Xword)symbol, type );
}
else {
info = ELF64_R_INFO((Elf_Xword)symbol, type );
if (elf_file.get_class() == ELFCLASS32) {
info = ELF32_R_INFO((Elf_Xword)symbol, type);
} else {
info = ELF64_R_INFO((Elf_Xword)symbol, type);
}
add_entry( offset, info );
add_entry(offset, info);
}
//------------------------------------------------------------------------------
void
add_entry( Elf64_Addr offset, Elf_Xword info, Elf_Sxword addend )
{
if ( elf_file.get_class() == ELFCLASS32 ) {
generic_add_entry< Elf32_Rela >( offset, info, addend );
}
else {
generic_add_entry< Elf64_Rela >( offset, info, addend );
//------------------------------------------------------------------------------
void add_entry(Elf64_Addr offset, Elf_Xword info, Elf_Sxword addend) {
if (elf_file.get_class() == ELFCLASS32) {
generic_add_entry<Elf32_Rela>(offset, info, addend);
} else {
generic_add_entry<Elf64_Rela>(offset, info, addend);
}
}
//------------------------------------------------------------------------------
void
add_entry( Elf64_Addr offset, Elf_Word symbol, unsigned char type,
Elf_Sxword addend )
{
//------------------------------------------------------------------------------
void add_entry(Elf64_Addr offset, Elf_Word symbol, unsigned char type, Elf_Sxword addend) {
Elf_Xword info;
if ( elf_file.get_class() == ELFCLASS32 ) {
info = ELF32_R_INFO( (Elf_Xword)symbol, type );
}
else {
info = ELF64_R_INFO( (Elf_Xword)symbol, type );
if (elf_file.get_class() == ELFCLASS32) {
info = ELF32_R_INFO((Elf_Xword)symbol, type);
} else {
info = ELF64_R_INFO((Elf_Xword)symbol, type);
}
add_entry( offset, info, addend );
add_entry(offset, info, addend);
}
//------------------------------------------------------------------------------
void
add_entry( string_section_accessor str_writer,
const char* str,
symbol_section_accessor sym_writer,
Elf64_Addr value,
Elf_Word size,
unsigned char sym_info,
unsigned char other,
Elf_Half shndx,
Elf64_Addr offset,
unsigned char type )
{
Elf_Word str_index = str_writer.add_string( str );
Elf_Word sym_index = sym_writer.add_symbol( str_index, value, size,
sym_info, other, shndx );
add_entry( offset, sym_index, type );
//------------------------------------------------------------------------------
void add_entry(string_section_accessor str_writer, const char* str,
symbol_section_accessor sym_writer, Elf64_Addr value, Elf_Word size,
unsigned char sym_info, unsigned char other, Elf_Half shndx, Elf64_Addr offset,
unsigned char type) {
Elf_Word str_index = str_writer.add_string(str);
Elf_Word sym_index = sym_writer.add_symbol(str_index, value, size, sym_info, other, shndx);
add_entry(offset, sym_index, type);
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
Elf_Half
get_symbol_table_index() const
{
return (Elf_Half)relocation_section->get_link();
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
Elf_Half get_symbol_table_index() const { return (Elf_Half)relocation_section->get_link(); }
//------------------------------------------------------------------------------
template< class T >
void
generic_get_entry_rel( Elf_Xword index,
Elf64_Addr& offset,
Elf_Word& symbol,
Elf_Word& type,
Elf_Sxword& addend ) const
{
//------------------------------------------------------------------------------
template <class T>
void generic_get_entry_rel(Elf_Xword index, Elf64_Addr& offset, Elf_Word& symbol,
Elf_Word& type, Elf_Sxword& addend) const {
const endianess_convertor& convertor = elf_file.get_convertor();
const T* pEntry = reinterpret_cast<const T*>(
relocation_section->get_data() +
index * relocation_section->get_entry_size() );
offset = convertor( pEntry->r_offset );
Elf_Xword tmp = convertor( pEntry->r_info );
symbol = get_sym_and_type<T>::get_r_sym( tmp );
type = get_sym_and_type<T>::get_r_type( tmp );
addend = 0;
const T* pEntry = reinterpret_cast<const T*>(relocation_section->get_data() +
index * relocation_section->get_entry_size());
offset = convertor(pEntry->r_offset);
Elf_Xword tmp = convertor(pEntry->r_info);
symbol = get_sym_and_type<T>::get_r_sym(tmp);
type = get_sym_and_type<T>::get_r_type(tmp);
addend = 0;
}
//------------------------------------------------------------------------------
template< class T >
void
generic_get_entry_rela( Elf_Xword index,
Elf64_Addr& offset,
Elf_Word& symbol,
Elf_Word& type,
Elf_Sxword& addend ) const
{
//------------------------------------------------------------------------------
template <class T>
void generic_get_entry_rela(Elf_Xword index, Elf64_Addr& offset, Elf_Word& symbol,
Elf_Word& type, Elf_Sxword& addend) const {
const endianess_convertor& convertor = elf_file.get_convertor();
const T* pEntry = reinterpret_cast<const T*>(
relocation_section->get_data() +
index * relocation_section->get_entry_size() );
offset = convertor( pEntry->r_offset );
Elf_Xword tmp = convertor( pEntry->r_info );
symbol = get_sym_and_type<T>::get_r_sym( tmp );
type = get_sym_and_type<T>::get_r_type( tmp );
addend = convertor( pEntry->r_addend );
const T* pEntry = reinterpret_cast<const T*>(relocation_section->get_data() +
index * relocation_section->get_entry_size());
offset = convertor(pEntry->r_offset);
Elf_Xword tmp = convertor(pEntry->r_info);
symbol = get_sym_and_type<T>::get_r_sym(tmp);
type = get_sym_and_type<T>::get_r_type(tmp);
addend = convertor(pEntry->r_addend);
}
//------------------------------------------------------------------------------
template< class T >
void
generic_add_entry( Elf64_Addr offset, Elf_Xword info )
{
//------------------------------------------------------------------------------
template <class T>
void generic_add_entry(Elf64_Addr offset, Elf_Xword info) {
const endianess_convertor& convertor = elf_file.get_convertor();
T entry;
entry.r_offset = offset;
entry.r_info = info;
entry.r_offset = convertor( entry.r_offset );
entry.r_info = convertor( entry.r_info );
entry.r_info = info;
entry.r_offset = convertor(entry.r_offset);
entry.r_info = convertor(entry.r_info);
relocation_section->append_data( reinterpret_cast<char*>( &entry ), sizeof( entry ) );
relocation_section->append_data(reinterpret_cast<char*>(&entry), sizeof(entry));
}
//------------------------------------------------------------------------------
template< class T >
void
generic_add_entry( Elf64_Addr offset, Elf_Xword info, Elf_Sxword addend )
{
//------------------------------------------------------------------------------
template <class T>
void generic_add_entry(Elf64_Addr offset, Elf_Xword info, Elf_Sxword addend) {
const endianess_convertor& convertor = elf_file.get_convertor();
T entry;
entry.r_offset = offset;
entry.r_info = info;
entry.r_info = info;
entry.r_addend = addend;
entry.r_offset = convertor( entry.r_offset );
entry.r_info = convertor( entry.r_info );
entry.r_addend = convertor( entry.r_addend );
entry.r_offset = convertor(entry.r_offset);
entry.r_info = convertor(entry.r_info);
entry.r_addend = convertor(entry.r_addend);
relocation_section->append_data( reinterpret_cast<char*>( &entry ), sizeof( entry ) );
relocation_section->append_data(reinterpret_cast<char*>(&entry), sizeof(entry));
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
private:
const elfio& elf_file;
section* relocation_section;
section* relocation_section;
};
} // namespace ELFIO
} // namespace ELFIO
#endif // ELFIO_RELOCATION_HPP
#endif // ELFIO_RELOCATION_HPP
+132 -188
View File
@@ -28,269 +28,213 @@ THE SOFTWARE.
namespace ELFIO {
class section
{
class section {
friend class elfio;
public:
virtual ~section() {};
ELFIO_GET_ACCESS_DECL ( Elf_Half, index );
ELFIO_GET_SET_ACCESS_DECL( std::string, name );
ELFIO_GET_SET_ACCESS_DECL( Elf_Word, type );
ELFIO_GET_SET_ACCESS_DECL( Elf_Xword, flags );
ELFIO_GET_SET_ACCESS_DECL( Elf_Word, info );
ELFIO_GET_SET_ACCESS_DECL( Elf_Word, link );
ELFIO_GET_SET_ACCESS_DECL( Elf_Xword, addr_align );
ELFIO_GET_SET_ACCESS_DECL( Elf_Xword, entry_size );
ELFIO_GET_SET_ACCESS_DECL( Elf64_Addr, address );
ELFIO_GET_SET_ACCESS_DECL( Elf_Xword, size );
ELFIO_GET_SET_ACCESS_DECL( Elf_Word, name_string_offset );
public:
virtual ~section(){};
virtual const char* get_data() const = 0;
virtual void set_data( const char* pData, Elf_Word size ) = 0;
virtual void set_data( const std::string& data ) = 0;
virtual void append_data( const char* pData, Elf_Word size ) = 0;
virtual void append_data( const std::string& data ) = 0;
ELFIO_GET_ACCESS_DECL(Elf_Half, index);
ELFIO_GET_SET_ACCESS_DECL(std::string, name);
ELFIO_GET_SET_ACCESS_DECL(Elf_Word, type);
ELFIO_GET_SET_ACCESS_DECL(Elf_Xword, flags);
ELFIO_GET_SET_ACCESS_DECL(Elf_Word, info);
ELFIO_GET_SET_ACCESS_DECL(Elf_Word, link);
ELFIO_GET_SET_ACCESS_DECL(Elf_Xword, addr_align);
ELFIO_GET_SET_ACCESS_DECL(Elf_Xword, entry_size);
ELFIO_GET_SET_ACCESS_DECL(Elf64_Addr, address);
ELFIO_GET_SET_ACCESS_DECL(Elf_Xword, size);
ELFIO_GET_SET_ACCESS_DECL(Elf_Word, name_string_offset);
protected:
ELFIO_GET_SET_ACCESS_DECL( Elf64_Off, offset );
ELFIO_SET_ACCESS_DECL( Elf_Half, index );
virtual void load( std::istream& f,
std::streampos header_offset ) = 0;
virtual void save( std::ostream& f,
std::streampos header_offset,
std::streampos data_offset ) = 0;
virtual bool is_address_initialized() const = 0;
virtual const char* get_data() const = 0;
virtual void set_data(const char* pData, Elf_Word size) = 0;
virtual void set_data(const std::string& data) = 0;
virtual void append_data(const char* pData, Elf_Word size) = 0;
virtual void append_data(const std::string& data) = 0;
protected:
ELFIO_GET_SET_ACCESS_DECL(Elf64_Off, offset);
ELFIO_SET_ACCESS_DECL(Elf_Half, index);
virtual void load(std::istream& f, std::streampos header_offset) = 0;
virtual void save(std::ostream& f, std::streampos header_offset,
std::streampos data_offset) = 0;
virtual bool is_address_initialized() const = 0;
};
template< class T >
class section_impl : public section
{
public:
//------------------------------------------------------------------------------
section_impl( const endianess_convertor* convertor_ ) : convertor( convertor_ )
{
std::fill_n( reinterpret_cast<char*>( &header ), sizeof( header ), '\0' );
template <class T>
class section_impl : public section {
public:
//------------------------------------------------------------------------------
section_impl(const endianess_convertor* convertor_) : convertor(convertor_) {
std::fill_n(reinterpret_cast<char*>(&header), sizeof(header), '\0');
is_address_set = false;
data = 0;
data_size = 0;
data = 0;
data_size = 0;
}
//------------------------------------------------------------------------------
~section_impl()
{
delete [] data;
}
//------------------------------------------------------------------------------
~section_impl() { delete[] data; }
//------------------------------------------------------------------------------
//------------------------------------------------------------------------------
// Section info functions
ELFIO_GET_SET_ACCESS( Elf_Word, type, header.sh_type );
ELFIO_GET_SET_ACCESS( Elf_Xword, flags, header.sh_flags );
ELFIO_GET_SET_ACCESS( Elf_Xword, size, header.sh_size );
ELFIO_GET_SET_ACCESS( Elf_Word, link, header.sh_link );
ELFIO_GET_SET_ACCESS( Elf_Word, info, header.sh_info );
ELFIO_GET_SET_ACCESS( Elf_Xword, addr_align, header.sh_addralign );
ELFIO_GET_SET_ACCESS( Elf_Xword, entry_size, header.sh_entsize );
ELFIO_GET_SET_ACCESS( Elf_Word, name_string_offset, header.sh_name );
ELFIO_GET_ACCESS ( Elf64_Addr, address, header.sh_addr );
ELFIO_GET_SET_ACCESS(Elf_Word, type, header.sh_type);
ELFIO_GET_SET_ACCESS(Elf_Xword, flags, header.sh_flags);
ELFIO_GET_SET_ACCESS(Elf_Xword, size, header.sh_size);
ELFIO_GET_SET_ACCESS(Elf_Word, link, header.sh_link);
ELFIO_GET_SET_ACCESS(Elf_Word, info, header.sh_info);
ELFIO_GET_SET_ACCESS(Elf_Xword, addr_align, header.sh_addralign);
ELFIO_GET_SET_ACCESS(Elf_Xword, entry_size, header.sh_entsize);
ELFIO_GET_SET_ACCESS(Elf_Word, name_string_offset, header.sh_name);
ELFIO_GET_ACCESS(Elf64_Addr, address, header.sh_addr);
//------------------------------------------------------------------------------
Elf_Half
get_index() const
{
return index;
}
//------------------------------------------------------------------------------
Elf_Half get_index() const { return index; }
//------------------------------------------------------------------------------
std::string
get_name() const
{
return name;
}
//------------------------------------------------------------------------------
std::string get_name() const { return name; }
//------------------------------------------------------------------------------
void
set_name( std::string name_ )
{
name = name_;
}
//------------------------------------------------------------------------------
void set_name(std::string name_) { name = name_; }
//------------------------------------------------------------------------------
void
set_address( Elf64_Addr value )
{
//------------------------------------------------------------------------------
void set_address(Elf64_Addr value) {
header.sh_addr = value;
header.sh_addr = (*convertor)( header.sh_addr );
header.sh_addr = (*convertor)(header.sh_addr);
is_address_set = true;
}
//------------------------------------------------------------------------------
bool
is_address_initialized() const
{
return is_address_set;
}
//------------------------------------------------------------------------------
bool is_address_initialized() const { return is_address_set; }
//------------------------------------------------------------------------------
const char*
get_data() const
{
return data;
}
//------------------------------------------------------------------------------
const char* get_data() const { return data; }
//------------------------------------------------------------------------------
void
set_data( const char* raw_data, Elf_Word size )
{
if ( get_type() != SHT_NOBITS ) {
delete [] data;
//------------------------------------------------------------------------------
void set_data(const char* raw_data, Elf_Word size) {
if (get_type() != SHT_NOBITS) {
delete[] data;
try {
data = new char[size];
} catch (const std::bad_alloc&) {
data = 0;
data = 0;
data_size = 0;
size = 0;
size = 0;
}
if ( 0 != data && 0 != raw_data ) {
if (0 != data && 0 != raw_data) {
data_size = size;
std::copy( raw_data, raw_data + size, data );
std::copy(raw_data, raw_data + size, data);
}
}
set_size( size );
set_size(size);
}
//------------------------------------------------------------------------------
void
set_data( const std::string& str_data )
{
return set_data( str_data.c_str(), (Elf_Word)str_data.size() );
//------------------------------------------------------------------------------
void set_data(const std::string& str_data) {
return set_data(str_data.c_str(), (Elf_Word)str_data.size());
}
//------------------------------------------------------------------------------
void
append_data( const char* raw_data, Elf_Word size )
{
if ( get_type() != SHT_NOBITS ) {
if ( get_size() + size < data_size ) {
std::copy( raw_data, raw_data + size, data + get_size() );
}
else {
data_size = 2*( data_size + size);
//------------------------------------------------------------------------------
void append_data(const char* raw_data, Elf_Word size) {
if (get_type() != SHT_NOBITS) {
if (get_size() + size < data_size) {
std::copy(raw_data, raw_data + size, data + get_size());
} else {
data_size = 2 * (data_size + size);
char* new_data;
try {
new_data = new char[data_size];
} catch (const std::bad_alloc&) {
new_data = 0;
size = 0;
size = 0;
}
if ( 0 != new_data ) {
std::copy( data, data + get_size(), new_data );
std::copy( raw_data, raw_data + size, new_data + get_size() );
delete [] data;
if (0 != new_data) {
std::copy(data, data + get_size(), new_data);
std::copy(raw_data, raw_data + size, new_data + get_size());
delete[] data;
data = new_data;
}
}
set_size( get_size() + size );
set_size(get_size() + size);
}
}
//------------------------------------------------------------------------------
void
append_data( const std::string& str_data )
{
return append_data( str_data.c_str(), (Elf_Word)str_data.size() );
//------------------------------------------------------------------------------
void append_data(const std::string& str_data) {
return append_data(str_data.c_str(), (Elf_Word)str_data.size());
}
//------------------------------------------------------------------------------
protected:
//------------------------------------------------------------------------------
ELFIO_GET_SET_ACCESS( Elf64_Off, offset, header.sh_offset );
//------------------------------------------------------------------------------
protected:
//------------------------------------------------------------------------------
ELFIO_GET_SET_ACCESS(Elf64_Off, offset, header.sh_offset);
//------------------------------------------------------------------------------
void
set_index( Elf_Half value )
{
index = value;
}
//------------------------------------------------------------------------------
void set_index(Elf_Half value) { index = value; }
//------------------------------------------------------------------------------
void
load( std::istream& stream,
std::streampos header_offset )
{
std::fill_n( reinterpret_cast<char*>( &header ), sizeof( header ), '\0' );
stream.seekg( header_offset );
stream.read( reinterpret_cast<char*>( &header ), sizeof( header ) );
//------------------------------------------------------------------------------
void load(std::istream& stream, std::streampos header_offset) {
std::fill_n(reinterpret_cast<char*>(&header), sizeof(header), '\0');
stream.seekg(header_offset);
stream.read(reinterpret_cast<char*>(&header), sizeof(header));
Elf_Xword size = get_size();
if ( 0 == data && SHT_NULL != get_type() && SHT_NOBITS != get_type() ) {
if (0 == data && SHT_NULL != get_type() && SHT_NOBITS != get_type()) {
try {
data = new char[size];
} catch (const std::bad_alloc&) {
data = 0;
data = 0;
data_size = 0;
}
if ( 0 != size ) {
stream.seekg( (*convertor)( header.sh_offset ) );
stream.read( data, size );
if (0 != size) {
stream.seekg((*convertor)(header.sh_offset));
stream.read(data, size);
data_size = size;
}
}
}
//------------------------------------------------------------------------------
void
save( std::ostream& f,
std::streampos header_offset,
std::streampos data_offset )
{
if ( 0 != get_index() ) {
//------------------------------------------------------------------------------
void save(std::ostream& f, std::streampos header_offset, std::streampos data_offset) {
if (0 != get_index()) {
header.sh_offset = data_offset;
header.sh_offset = (*convertor)( header.sh_offset );
header.sh_offset = (*convertor)(header.sh_offset);
}
save_header( f, header_offset );
if ( get_type() != SHT_NOBITS && get_type() != SHT_NULL &&
get_size() != 0 && data != 0 ) {
save_data( f, data_offset );
save_header(f, header_offset);
if (get_type() != SHT_NOBITS && get_type() != SHT_NULL && get_size() != 0 && data != 0) {
save_data(f, data_offset);
}
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
void
save_header( std::ostream& f,
std::streampos header_offset ) const
{
f.seekp( header_offset );
f.write( reinterpret_cast<const char*>( &header ), sizeof( header ) );
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
void save_header(std::ostream& f, std::streampos header_offset) const {
f.seekp(header_offset);
f.write(reinterpret_cast<const char*>(&header), sizeof(header));
}
//------------------------------------------------------------------------------
void
save_data( std::ostream& f,
std::streampos data_offset ) const
{
f.seekp( data_offset );
f.write( get_data(), get_size() );
//------------------------------------------------------------------------------
void save_data(std::ostream& f, std::streampos data_offset) const {
f.seekp(data_offset);
f.write(get_data(), get_size());
}
//------------------------------------------------------------------------------
private:
T header;
Elf_Half index;
std::string name;
char* data;
Elf_Word data_size;
//------------------------------------------------------------------------------
private:
T header;
Elf_Half index;
std::string name;
char* data;
Elf_Word data_size;
const endianess_convertor* convertor;
bool is_address_set;
bool is_address_set;
};
} // namespace ELFIO
} // namespace ELFIO
#endif // ELFIO_SECTION_HPP
#endif // ELFIO_SECTION_HPP
+90 -132
View File
@@ -28,193 +28,151 @@ THE SOFTWARE.
namespace ELFIO {
class segment
{
class segment {
friend class elfio;
public:
virtual ~segment() {};
ELFIO_GET_ACCESS_DECL ( Elf_Half, index );
ELFIO_GET_SET_ACCESS_DECL( Elf_Word, type );
ELFIO_GET_SET_ACCESS_DECL( Elf_Word, flags );
ELFIO_GET_SET_ACCESS_DECL( Elf_Xword, align );
ELFIO_GET_SET_ACCESS_DECL( Elf64_Addr, virtual_address );
ELFIO_GET_SET_ACCESS_DECL( Elf64_Addr, physical_address );
ELFIO_GET_SET_ACCESS_DECL( Elf_Xword, file_size );
ELFIO_GET_SET_ACCESS_DECL( Elf_Xword, memory_size );
ELFIO_GET_ACCESS_DECL( Elf64_Off, offset );
public:
virtual ~segment(){};
ELFIO_GET_ACCESS_DECL(Elf_Half, index);
ELFIO_GET_SET_ACCESS_DECL(Elf_Word, type);
ELFIO_GET_SET_ACCESS_DECL(Elf_Word, flags);
ELFIO_GET_SET_ACCESS_DECL(Elf_Xword, align);
ELFIO_GET_SET_ACCESS_DECL(Elf64_Addr, virtual_address);
ELFIO_GET_SET_ACCESS_DECL(Elf64_Addr, physical_address);
ELFIO_GET_SET_ACCESS_DECL(Elf_Xword, file_size);
ELFIO_GET_SET_ACCESS_DECL(Elf_Xword, memory_size);
ELFIO_GET_ACCESS_DECL(Elf64_Off, offset);
virtual const char* get_data() const = 0;
virtual Elf_Half add_section_index( Elf_Half index, Elf_Xword addr_align ) = 0;
virtual Elf_Half get_sections_num() const = 0;
virtual Elf_Half get_section_index_at( Elf_Half num ) const = 0;
virtual bool is_offset_initialized() const = 0;
virtual Elf_Half add_section_index(Elf_Half index, Elf_Xword addr_align) = 0;
virtual Elf_Half get_sections_num() const = 0;
virtual Elf_Half get_section_index_at(Elf_Half num) const = 0;
virtual bool is_offset_initialized() const = 0;
protected:
ELFIO_SET_ACCESS_DECL( Elf64_Off, offset );
ELFIO_SET_ACCESS_DECL( Elf_Half, index );
virtual const std::vector<Elf_Half>& get_sections() const = 0;
virtual void load( std::istream& stream, std::streampos header_offset ) = 0;
virtual void save( std::ostream& f, std::streampos header_offset,
std::streampos data_offset ) = 0;
protected:
ELFIO_SET_ACCESS_DECL(Elf64_Off, offset);
ELFIO_SET_ACCESS_DECL(Elf_Half, index);
virtual const std::vector<Elf_Half>& get_sections() const = 0;
virtual void load(std::istream& stream, std::streampos header_offset) = 0;
virtual void save(std::ostream& f, std::streampos header_offset,
std::streampos data_offset) = 0;
};
//------------------------------------------------------------------------------
template< class T >
class segment_impl : public segment
{
public:
//------------------------------------------------------------------------------
segment_impl( endianess_convertor* convertor_ ) :
convertor( convertor_ )
{
template <class T>
class segment_impl : public segment {
public:
//------------------------------------------------------------------------------
segment_impl(endianess_convertor* convertor_) : convertor(convertor_) {
is_offset_set = false;
std::fill_n( reinterpret_cast<char*>( &ph ), sizeof( ph ), '\0' );
std::fill_n(reinterpret_cast<char*>(&ph), sizeof(ph), '\0');
data = 0;
}
//------------------------------------------------------------------------------
virtual ~segment_impl()
{
delete [] data;
}
//------------------------------------------------------------------------------
virtual ~segment_impl() { delete[] data; }
//------------------------------------------------------------------------------
//------------------------------------------------------------------------------
// Section info functions
ELFIO_GET_SET_ACCESS( Elf_Word, type, ph.p_type );
ELFIO_GET_SET_ACCESS( Elf_Word, flags, ph.p_flags );
ELFIO_GET_SET_ACCESS( Elf_Xword, align, ph.p_align );
ELFIO_GET_SET_ACCESS( Elf64_Addr, virtual_address, ph.p_vaddr );
ELFIO_GET_SET_ACCESS( Elf64_Addr, physical_address, ph.p_paddr );
ELFIO_GET_SET_ACCESS( Elf_Xword, file_size, ph.p_filesz );
ELFIO_GET_SET_ACCESS( Elf_Xword, memory_size, ph.p_memsz );
ELFIO_GET_ACCESS( Elf64_Off, offset, ph.p_offset );
ELFIO_GET_SET_ACCESS(Elf_Word, type, ph.p_type);
ELFIO_GET_SET_ACCESS(Elf_Word, flags, ph.p_flags);
ELFIO_GET_SET_ACCESS(Elf_Xword, align, ph.p_align);
ELFIO_GET_SET_ACCESS(Elf64_Addr, virtual_address, ph.p_vaddr);
ELFIO_GET_SET_ACCESS(Elf64_Addr, physical_address, ph.p_paddr);
ELFIO_GET_SET_ACCESS(Elf_Xword, file_size, ph.p_filesz);
ELFIO_GET_SET_ACCESS(Elf_Xword, memory_size, ph.p_memsz);
ELFIO_GET_ACCESS(Elf64_Off, offset, ph.p_offset);
//------------------------------------------------------------------------------
Elf_Half
get_index() const
{
return index;
}
//------------------------------------------------------------------------------
Elf_Half get_index() const { return index; }
//------------------------------------------------------------------------------
const char*
get_data() const
{
return data;
}
//------------------------------------------------------------------------------
const char* get_data() const { return data; }
//------------------------------------------------------------------------------
Elf_Half
add_section_index( Elf_Half sec_index, Elf_Xword addr_align )
{
sections.push_back( sec_index );
if ( addr_align > get_align() ) {
set_align( addr_align );
//------------------------------------------------------------------------------
Elf_Half add_section_index(Elf_Half sec_index, Elf_Xword addr_align) {
sections.push_back(sec_index);
if (addr_align > get_align()) {
set_align(addr_align);
}
return (Elf_Half)sections.size();
}
//------------------------------------------------------------------------------
Elf_Half
get_sections_num() const
{
return (Elf_Half)sections.size();
}
//------------------------------------------------------------------------------
Elf_Half get_sections_num() const { return (Elf_Half)sections.size(); }
//------------------------------------------------------------------------------
Elf_Half
get_section_index_at( Elf_Half num ) const
{
if ( num < sections.size() ) {
//------------------------------------------------------------------------------
Elf_Half get_section_index_at(Elf_Half num) const {
if (num < sections.size()) {
return sections[num];
}
return -1;
}
//------------------------------------------------------------------------------
protected:
//------------------------------------------------------------------------------
//------------------------------------------------------------------------------
protected:
//------------------------------------------------------------------------------
//------------------------------------------------------------------------------
void
set_offset( Elf64_Off value )
{
//------------------------------------------------------------------------------
void set_offset(Elf64_Off value) {
ph.p_offset = value;
ph.p_offset = (*convertor)( ph.p_offset );
ph.p_offset = (*convertor)(ph.p_offset);
is_offset_set = true;
}
//------------------------------------------------------------------------------
bool
is_offset_initialized() const
{
return is_offset_set;
}
//------------------------------------------------------------------------------
bool is_offset_initialized() const { return is_offset_set; }
//------------------------------------------------------------------------------
const std::vector<Elf_Half>&
get_sections() const
{
return sections;
}
//------------------------------------------------------------------------------
void
set_index( Elf_Half value )
{
index = value;
}
//------------------------------------------------------------------------------
const std::vector<Elf_Half>& get_sections() const { return sections; }
//------------------------------------------------------------------------------
void
load( std::istream& stream,
std::streampos header_offset )
{
stream.seekg( header_offset );
stream.read( reinterpret_cast<char*>( &ph ), sizeof( ph ) );
//------------------------------------------------------------------------------
void set_index(Elf_Half value) { index = value; }
//------------------------------------------------------------------------------
void load(std::istream& stream, std::streampos header_offset) {
stream.seekg(header_offset);
stream.read(reinterpret_cast<char*>(&ph), sizeof(ph));
is_offset_set = true;
if ( PT_NULL != get_type() && 0 != get_file_size() ) {
stream.seekg( (*convertor)( ph.p_offset ) );
if (PT_NULL != get_type() && 0 != get_file_size()) {
stream.seekg((*convertor)(ph.p_offset));
Elf_Xword size = get_file_size();
try {
data = new char[size];
} catch (const std::bad_alloc&) {
data = 0;
}
if ( 0 != data ) {
stream.read( data, size );
if (0 != data) {
stream.read(data, size);
}
}
}
//------------------------------------------------------------------------------
void save( std::ostream& f,
std::streampos header_offset,
std::streampos data_offset )
{
//------------------------------------------------------------------------------
void save(std::ostream& f, std::streampos header_offset, std::streampos data_offset) {
ph.p_offset = data_offset;
ph.p_offset = (*convertor)(ph.p_offset);
f.seekp( header_offset );
f.write( reinterpret_cast<const char*>( &ph ), sizeof( ph ) );
f.seekp(header_offset);
f.write(reinterpret_cast<const char*>(&ph), sizeof(ph));
}
//------------------------------------------------------------------------------
private:
T ph;
Elf_Half index;
char* data;
//------------------------------------------------------------------------------
private:
T ph;
Elf_Half index;
char* data;
std::vector<Elf_Half> sections;
endianess_convertor* convertor;
bool is_offset_set;
endianess_convertor* convertor;
bool is_offset_set;
};
} // namespace ELFIO
} // namespace ELFIO
#endif // ELFIO_SEGMENT_HPP
#endif // ELFIO_SEGMENT_HPP
+21 -33
View File
@@ -30,24 +30,18 @@ THE SOFTWARE.
namespace ELFIO {
//------------------------------------------------------------------------------
class string_section_accessor
{
public:
//------------------------------------------------------------------------------
string_section_accessor( section* section_ ) :
string_section( section_ )
{
}
class string_section_accessor {
public:
//------------------------------------------------------------------------------
string_section_accessor(section* section_) : string_section(section_) {}
//------------------------------------------------------------------------------
const char*
get_string( Elf_Word index ) const
{
if ( string_section ) {
if ( index < string_section->get_size() ) {
//------------------------------------------------------------------------------
const char* get_string(Elf_Word index) const {
if (string_section) {
if (index < string_section->get_size()) {
const char* data = string_section->get_data();
if ( 0 != data ) {
if (0 != data) {
return data + index;
}
}
@@ -57,40 +51,34 @@ class string_section_accessor
}
//------------------------------------------------------------------------------
Elf_Word
add_string( const char* str )
{
//------------------------------------------------------------------------------
Elf_Word add_string(const char* str) {
Elf_Word current_position = 0;
if (string_section) {
// Strings are addeded to the end of the current section data
current_position = (Elf_Word)string_section->get_size();
if ( current_position == 0 ) {
if (current_position == 0) {
char empty_string = '\0';
string_section->append_data( &empty_string, 1 );
string_section->append_data(&empty_string, 1);
current_position++;
}
string_section->append_data( str, (Elf_Word)std::strlen( str ) + 1 );
string_section->append_data(str, (Elf_Word)std::strlen(str) + 1);
}
return current_position;
}
//------------------------------------------------------------------------------
Elf_Word
add_string( const std::string& str )
{
return add_string( str.c_str() );
}
//------------------------------------------------------------------------------
Elf_Word add_string(const std::string& str) { return add_string(str.c_str()); }
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
private:
section* string_section;
};
} // namespace ELFIO
} // namespace ELFIO
#endif // ELFIO_STRINGS_HPP
#endif // ELFIO_STRINGS_HPP
+108 -166
View File
@@ -26,82 +26,61 @@ THE SOFTWARE.
namespace ELFIO {
//------------------------------------------------------------------------------
class symbol_section_accessor
{
public:
//------------------------------------------------------------------------------
symbol_section_accessor( const elfio& elf_file_, section* symbol_section_ ) :
elf_file( elf_file_ ),
symbol_section( symbol_section_ )
{
class symbol_section_accessor {
public:
//------------------------------------------------------------------------------
symbol_section_accessor(const elfio& elf_file_, section* symbol_section_)
: elf_file(elf_file_), symbol_section(symbol_section_) {
find_hash_section();
}
//------------------------------------------------------------------------------
Elf_Xword
get_symbols_num() const
{
//------------------------------------------------------------------------------
Elf_Xword get_symbols_num() const {
Elf_Xword nRet = 0;
if ( 0 != symbol_section->get_entry_size() ) {
if (0 != symbol_section->get_entry_size()) {
nRet = symbol_section->get_size() / symbol_section->get_entry_size();
}
return nRet;
}
//------------------------------------------------------------------------------
bool
get_symbol( Elf_Xword index,
std::string& name,
Elf64_Addr& value,
Elf_Xword& size,
unsigned char& bind,
unsigned char& type,
Elf_Half& section_index,
unsigned char& other ) const
{
//------------------------------------------------------------------------------
bool get_symbol(Elf_Xword index, std::string& name, Elf64_Addr& value, Elf_Xword& size,
unsigned char& bind, unsigned char& type, Elf_Half& section_index,
unsigned char& other) const {
bool ret = false;
if ( elf_file.get_class() == ELFCLASS32 ) {
ret = generic_get_symbol<Elf32_Sym>( index, name, value, size, bind,
type, section_index, other );
}
else {
ret = generic_get_symbol<Elf64_Sym>( index, name, value, size, bind,
type, section_index, other );
if (elf_file.get_class() == ELFCLASS32) {
ret = generic_get_symbol<Elf32_Sym>(index, name, value, size, bind, type, section_index,
other);
} else {
ret = generic_get_symbol<Elf64_Sym>(index, name, value, size, bind, type, section_index,
other);
}
return ret;
}
//------------------------------------------------------------------------------
bool
get_symbol( const std::string& name,
Elf64_Addr& value,
Elf_Xword& size,
unsigned char& bind,
unsigned char& type,
Elf_Half& section_index,
unsigned char& other ) const
{
//------------------------------------------------------------------------------
bool get_symbol(const std::string& name, Elf64_Addr& value, Elf_Xword& size,
unsigned char& bind, unsigned char& type, Elf_Half& section_index,
unsigned char& other) const {
bool ret = false;
if ( 0 != get_hash_table_index() ) {
if (0 != get_hash_table_index()) {
Elf_Word nbucket = *(Elf_Word*)hash_section->get_data();
Elf_Word nchain = *(Elf_Word*)( hash_section->get_data() +
sizeof( Elf_Word ) );
Elf_Word val = elf_hash( (const unsigned char*)name.c_str() );
Elf_Word nchain = *(Elf_Word*)(hash_section->get_data() + sizeof(Elf_Word));
Elf_Word val = elf_hash((const unsigned char*)name.c_str());
Elf_Word y = *(Elf_Word*)( hash_section->get_data() +
( 2 + val % nbucket ) * sizeof( Elf_Word ) );
std::string str;
get_symbol( y, str, value, size, bind, type, section_index, other );
while ( str != name && STN_UNDEF != y && y < nchain ) {
y = *(Elf_Word*)( hash_section->get_data() +
( 2 + nbucket + y ) * sizeof( Elf_Word ) );
get_symbol( y, str, value, size, bind, type, section_index, other );
Elf_Word y =
*(Elf_Word*)(hash_section->get_data() + (2 + val % nbucket) * sizeof(Elf_Word));
std::string str;
get_symbol(y, str, value, size, bind, type, section_index, other);
while (str != name && STN_UNDEF != y && y < nchain) {
y = *(Elf_Word*)(hash_section->get_data() + (2 + nbucket + y) * sizeof(Elf_Word));
get_symbol(y, str, value, size, bind, type, section_index, other);
}
if ( str == name ) {
if (str == name) {
ret = true;
}
}
@@ -109,128 +88,95 @@ class symbol_section_accessor
return ret;
}
//------------------------------------------------------------------------------
Elf_Word
add_symbol( Elf_Word name, Elf64_Addr value, Elf_Xword size,
unsigned char info, unsigned char other,
Elf_Half shndx )
{
//------------------------------------------------------------------------------
Elf_Word add_symbol(Elf_Word name, Elf64_Addr value, Elf_Xword size, unsigned char info,
unsigned char other, Elf_Half shndx) {
Elf_Word nRet;
if ( symbol_section->get_size() == 0 ) {
if ( elf_file.get_class() == ELFCLASS32 ) {
nRet = generic_add_symbol<Elf32_Sym>( 0, 0, 0, 0, 0, 0 );
}
else {
nRet = generic_add_symbol<Elf64_Sym>( 0, 0, 0, 0, 0, 0 );
if (symbol_section->get_size() == 0) {
if (elf_file.get_class() == ELFCLASS32) {
nRet = generic_add_symbol<Elf32_Sym>(0, 0, 0, 0, 0, 0);
} else {
nRet = generic_add_symbol<Elf64_Sym>(0, 0, 0, 0, 0, 0);
}
}
if ( elf_file.get_class() == ELFCLASS32 ) {
nRet = generic_add_symbol<Elf32_Sym>( name, value, size, info, other,
shndx );
}
else {
nRet = generic_add_symbol<Elf64_Sym>( name, value, size, info, other,
shndx );
if (elf_file.get_class() == ELFCLASS32) {
nRet = generic_add_symbol<Elf32_Sym>(name, value, size, info, other, shndx);
} else {
nRet = generic_add_symbol<Elf64_Sym>(name, value, size, info, other, shndx);
}
return nRet;
}
//------------------------------------------------------------------------------
Elf_Word
add_symbol( Elf_Word name, Elf64_Addr value, Elf_Xword size,
unsigned char bind, unsigned char type, unsigned char other,
Elf_Half shndx )
{
return add_symbol( name, value, size, ELF_ST_INFO( bind, type ), other, shndx );
//------------------------------------------------------------------------------
Elf_Word add_symbol(Elf_Word name, Elf64_Addr value, Elf_Xword size, unsigned char bind,
unsigned char type, unsigned char other, Elf_Half shndx) {
return add_symbol(name, value, size, ELF_ST_INFO(bind, type), other, shndx);
}
//------------------------------------------------------------------------------
Elf_Word
add_symbol( string_section_accessor& pStrWriter, const char* str,
Elf64_Addr value, Elf_Xword size,
unsigned char info, unsigned char other,
Elf_Half shndx )
{
Elf_Word index = pStrWriter.add_string( str );
return add_symbol( index, value, size, info, other, shndx );
//------------------------------------------------------------------------------
Elf_Word add_symbol(string_section_accessor& pStrWriter, const char* str, Elf64_Addr value,
Elf_Xword size, unsigned char info, unsigned char other, Elf_Half shndx) {
Elf_Word index = pStrWriter.add_string(str);
return add_symbol(index, value, size, info, other, shndx);
}
//------------------------------------------------------------------------------
Elf_Word
add_symbol( string_section_accessor& pStrWriter, const char* str,
Elf64_Addr value, Elf_Xword size,
unsigned char bind, unsigned char type, unsigned char other,
Elf_Half shndx )
{
return add_symbol( pStrWriter, str, value, size, ELF_ST_INFO( bind, type ), other, shndx );
//------------------------------------------------------------------------------
Elf_Word add_symbol(string_section_accessor& pStrWriter, const char* str, Elf64_Addr value,
Elf_Xword size, unsigned char bind, unsigned char type, unsigned char other,
Elf_Half shndx) {
return add_symbol(pStrWriter, str, value, size, ELF_ST_INFO(bind, type), other, shndx);
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
void
find_hash_section()
{
hash_section = 0;
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
void find_hash_section() {
hash_section = 0;
hash_section_index = 0;
Elf_Half nSecNo = elf_file.sections.size();
for ( Elf_Half i = 0; i < nSecNo && 0 == hash_section_index; ++i ) {
for (Elf_Half i = 0; i < nSecNo && 0 == hash_section_index; ++i) {
const section* sec = elf_file.sections[i];
if ( sec->get_link() == symbol_section->get_index() ) {
hash_section = sec;
if (sec->get_link() == symbol_section->get_index()) {
hash_section = sec;
hash_section_index = i;
}
}
}
//------------------------------------------------------------------------------
Elf_Half
get_string_table_index() const
{
return (Elf_Half)symbol_section->get_link();
}
//------------------------------------------------------------------------------
Elf_Half get_string_table_index() const { return (Elf_Half)symbol_section->get_link(); }
//------------------------------------------------------------------------------
Elf_Half
get_hash_table_index() const
{
return hash_section_index;
}
//------------------------------------------------------------------------------
Elf_Half get_hash_table_index() const { return hash_section_index; }
//------------------------------------------------------------------------------
template< class T >
bool
generic_get_symbol( Elf_Xword index,
std::string& name, Elf64_Addr& value,
Elf_Xword& size,
unsigned char& bind, unsigned char& type,
Elf_Half& section_index,
unsigned char& other ) const
{
//------------------------------------------------------------------------------
template <class T>
bool generic_get_symbol(Elf_Xword index, std::string& name, Elf64_Addr& value, Elf_Xword& size,
unsigned char& bind, unsigned char& type, Elf_Half& section_index,
unsigned char& other) const {
bool ret = false;
if ( index < get_symbols_num() ) {
const T* pSym = reinterpret_cast<const T*>(
symbol_section->get_data() +
index * symbol_section->get_entry_size() );
if (index < get_symbols_num()) {
const T* pSym = reinterpret_cast<const T*>(symbol_section->get_data() +
index * symbol_section->get_entry_size());
const endianess_convertor& convertor = elf_file.get_convertor();
section* string_section = elf_file.sections[get_string_table_index()];
string_section_accessor str_reader( string_section );
const char* pStr = str_reader.get_string( convertor( pSym->st_name ) );
if ( 0 != pStr ) {
string_section_accessor str_reader(string_section);
const char* pStr = str_reader.get_string(convertor(pSym->st_name));
if (0 != pStr) {
name = pStr;
}
value = convertor( pSym->st_value );
size = convertor( pSym->st_size );
bind = ELF_ST_BIND( pSym->st_info );
type = ELF_ST_TYPE( pSym->st_info );
section_index = convertor( pSym->st_shndx );
other = pSym->st_other;
value = convertor(pSym->st_value);
size = convertor(pSym->st_size);
bind = ELF_ST_BIND(pSym->st_info);
type = ELF_ST_TYPE(pSym->st_info);
section_index = convertor(pSym->st_shndx);
other = pSym->st_other;
ret = true;
}
@@ -238,41 +184,37 @@ class symbol_section_accessor
return ret;
}
//------------------------------------------------------------------------------
template< class T >
Elf_Word
generic_add_symbol( Elf_Word name, Elf64_Addr value, Elf_Xword size,
unsigned char info, unsigned char other,
Elf_Half shndx )
{
//------------------------------------------------------------------------------
template <class T>
Elf_Word generic_add_symbol(Elf_Word name, Elf64_Addr value, Elf_Xword size, unsigned char info,
unsigned char other, Elf_Half shndx) {
const endianess_convertor& convertor = elf_file.get_convertor();
T entry;
entry.st_name = convertor( name );
entry.st_name = convertor(name);
entry.st_value = value;
entry.st_value = convertor( entry.st_value );
entry.st_size = size;
entry.st_size = convertor( entry.st_size );
entry.st_info = convertor( info );
entry.st_other = convertor( other );
entry.st_shndx = convertor( shndx );
entry.st_value = convertor(entry.st_value);
entry.st_size = size;
entry.st_size = convertor(entry.st_size);
entry.st_info = convertor(info);
entry.st_other = convertor(other);
entry.st_shndx = convertor(shndx);
symbol_section->append_data( reinterpret_cast<char*>( &entry ),
sizeof( entry ) );
symbol_section->append_data(reinterpret_cast<char*>(&entry), sizeof(entry));
Elf_Word nRet = symbol_section->get_size() / sizeof( entry ) - 1;
Elf_Word nRet = symbol_section->get_size() / sizeof(entry) - 1;
return nRet;
}
//------------------------------------------------------------------------------
private:
const elfio& elf_file;
section* symbol_section;
Elf_Half hash_section_index;
//------------------------------------------------------------------------------
private:
const elfio& elf_file;
section* symbol_section;
Elf_Half hash_section_index;
const section* hash_section;
};
} // namespace ELFIO
} // namespace ELFIO
#endif // ELFIO_SYMBOLS_HPP
#endif // ELFIO_SYMBOLS_HPP
+68 -120
View File
@@ -23,187 +23,135 @@ THE SOFTWARE.
#ifndef ELFIO_UTILS_HPP
#define ELFIO_UTILS_HPP
#define ELFIO_GET_ACCESS( TYPE, NAME, FIELD ) \
TYPE get_##NAME() const \
{ \
return (*convertor)( FIELD ); \
#define ELFIO_GET_ACCESS(TYPE, NAME, FIELD) \
TYPE get_##NAME() const { return (*convertor)(FIELD); }
#define ELFIO_SET_ACCESS(TYPE, NAME, FIELD) \
void set_##NAME(TYPE value) { \
FIELD = value; \
FIELD = (*convertor)(FIELD); \
}
#define ELFIO_SET_ACCESS( TYPE, NAME, FIELD ) \
void set_##NAME( TYPE value ) \
{ \
FIELD = value; \
FIELD = (*convertor)( FIELD ); \
}
#define ELFIO_GET_SET_ACCESS( TYPE, NAME, FIELD ) \
TYPE get_##NAME() const \
{ \
return (*convertor)( FIELD ); \
} \
void set_##NAME( TYPE value ) \
{ \
FIELD = value; \
FIELD = (*convertor)( FIELD ); \
#define ELFIO_GET_SET_ACCESS(TYPE, NAME, FIELD) \
TYPE get_##NAME() const { return (*convertor)(FIELD); } \
void set_##NAME(TYPE value) { \
FIELD = value; \
FIELD = (*convertor)(FIELD); \
}
#define ELFIO_GET_ACCESS_DECL( TYPE, NAME ) \
virtual TYPE get_##NAME() const = 0
#define ELFIO_GET_ACCESS_DECL(TYPE, NAME) virtual TYPE get_##NAME() const = 0
#define ELFIO_SET_ACCESS_DECL( TYPE, NAME ) \
virtual void set_##NAME( TYPE value ) = 0
#define ELFIO_SET_ACCESS_DECL(TYPE, NAME) virtual void set_##NAME(TYPE value) = 0
#define ELFIO_GET_SET_ACCESS_DECL( TYPE, NAME ) \
virtual TYPE get_##NAME() const = 0; \
virtual void set_##NAME( TYPE value ) = 0
#define ELFIO_GET_SET_ACCESS_DECL(TYPE, NAME) \
virtual TYPE get_##NAME() const = 0; \
virtual void set_##NAME(TYPE value) = 0
namespace ELFIO {
//------------------------------------------------------------------------------
class endianess_convertor {
public:
//------------------------------------------------------------------------------
endianess_convertor()
{
need_conversion = false;
public:
//------------------------------------------------------------------------------
endianess_convertor() { need_conversion = false; }
//------------------------------------------------------------------------------
void setup(unsigned char elf_file_encoding) {
need_conversion = (elf_file_encoding != get_host_encoding());
}
//------------------------------------------------------------------------------
void
setup( unsigned char elf_file_encoding )
{
need_conversion = ( elf_file_encoding != get_host_encoding() );
}
//------------------------------------------------------------------------------
uint64_t
operator()( uint64_t value ) const
{
if ( !need_conversion ) {
//------------------------------------------------------------------------------
uint64_t operator()(uint64_t value) const {
if (!need_conversion) {
return value;
}
value =
( ( value & 0x00000000000000FFull ) << 56 ) |
( ( value & 0x000000000000FF00ull ) << 40 ) |
( ( value & 0x0000000000FF0000ull ) << 24 ) |
( ( value & 0x00000000FF000000ull ) << 8 ) |
( ( value & 0x000000FF00000000ull ) >> 8 ) |
( ( value & 0x0000FF0000000000ull ) >> 24 ) |
( ( value & 0x00FF000000000000ull ) >> 40 ) |
( ( value & 0xFF00000000000000ull ) >> 56 );
value = ((value & 0x00000000000000FFull) << 56) | ((value & 0x000000000000FF00ull) << 40) |
((value & 0x0000000000FF0000ull) << 24) | ((value & 0x00000000FF000000ull) << 8) |
((value & 0x000000FF00000000ull) >> 8) | ((value & 0x0000FF0000000000ull) >> 24) |
((value & 0x00FF000000000000ull) >> 40) | ((value & 0xFF00000000000000ull) >> 56);
return value;
}
//------------------------------------------------------------------------------
int64_t
operator()( int64_t value ) const
{
if ( !need_conversion ) {
//------------------------------------------------------------------------------
int64_t operator()(int64_t value) const {
if (!need_conversion) {
return value;
}
return (int64_t)(*this)( (uint64_t)value );
return (int64_t)(*this)((uint64_t)value);
}
//------------------------------------------------------------------------------
uint32_t
operator()( uint32_t value ) const
{
if ( !need_conversion ) {
//------------------------------------------------------------------------------
uint32_t operator()(uint32_t value) const {
if (!need_conversion) {
return value;
}
value =
( ( value & 0x000000FF ) << 24 ) |
( ( value & 0x0000FF00 ) << 8 ) |
( ( value & 0x00FF0000 ) >> 8 ) |
( ( value & 0xFF000000 ) >> 24 );
value = ((value & 0x000000FF) << 24) | ((value & 0x0000FF00) << 8) |
((value & 0x00FF0000) >> 8) | ((value & 0xFF000000) >> 24);
return value;
}
//------------------------------------------------------------------------------
int32_t
operator()( int32_t value ) const
{
if ( !need_conversion ) {
//------------------------------------------------------------------------------
int32_t operator()(int32_t value) const {
if (!need_conversion) {
return value;
}
return (int32_t)(*this)( (uint32_t)value );
return (int32_t)(*this)((uint32_t)value);
}
//------------------------------------------------------------------------------
uint16_t
operator()( uint16_t value ) const
{
if ( !need_conversion ) {
//------------------------------------------------------------------------------
uint16_t operator()(uint16_t value) const {
if (!need_conversion) {
return value;
}
value =
( ( value & 0x00FF ) << 8 ) |
( ( value & 0xFF00 ) >> 8 );
value = ((value & 0x00FF) << 8) | ((value & 0xFF00) >> 8);
return value;
}
//------------------------------------------------------------------------------
int16_t
operator()( int16_t value ) const
{
if ( !need_conversion ) {
//------------------------------------------------------------------------------
int16_t operator()(int16_t value) const {
if (!need_conversion) {
return value;
}
return (int16_t)(*this)( (uint16_t)value );
return (int16_t)(*this)((uint16_t)value);
}
//------------------------------------------------------------------------------
int8_t
operator()( int8_t value ) const
{
return value;
}
//------------------------------------------------------------------------------
int8_t operator()(int8_t value) const { return value; }
//------------------------------------------------------------------------------
uint8_t
operator()( uint8_t value ) const
{
return value;
}
//------------------------------------------------------------------------------
uint8_t operator()(uint8_t value) const { return value; }
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
unsigned char
get_host_encoding() const
{
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
unsigned char get_host_encoding() const {
static const int tmp = 1;
if ( 1 == *(char*)&tmp ) {
if (1 == *(char*)&tmp) {
return ELFDATA2LSB;
}
else {
} else {
return ELFDATA2MSB;
}
}
//------------------------------------------------------------------------------
private:
//------------------------------------------------------------------------------
private:
bool need_conversion;
};
//------------------------------------------------------------------------------
inline
uint32_t
elf_hash( const unsigned char *name )
{
inline uint32_t elf_hash(const unsigned char* name) {
uint32_t h = 0, g;
while ( *name ) {
while (*name) {
h = (h << 4) + *name++;
g = h & 0xf0000000;
if ( g != 0 )
h ^= g >> 24;
if (g != 0) h ^= g >> 24;
h &= ~g;
}
return h;
}
} // namespace ELFIO
} // namespace ELFIO
#endif // ELFIO_UTILS_HPP
#endif // ELFIO_UTILS_HPP
+23 -27
View File
@@ -26,30 +26,30 @@ THE SOFTWARE.
//---
// Read environment variables.
void ihipReadEnv_I(int *var_ptr, const char *var_name1, const char *var_name2, const char *description)
{
char * env = getenv(var_name1);
void ihipReadEnv_I(int* var_ptr, const char* var_name1, const char* var_name2,
const char* description) {
char* env = getenv(var_name1);
// Check second name if first not defined, used to allow HIP_ or CUDA_ env vars.
if ((env == NULL) && strcmp(var_name2, "0")) {
env = getenv(var_name2);
}
// Default is set when variable is initialized (at top of this file), so only override if we find
// an environment variable.
// Default is set when variable is initialized (at top of this file), so only override if we
// find an environment variable.
if (env) {
long int v = strtol(env, NULL, 0);
*var_ptr = (int) (v);
*var_ptr = (int)(v);
}
if (HIP_PRINT_ENV) {
printf ("%-30s = %2d : %s\n", var_name1, *var_ptr, description);
printf("%-30s = %2d : %s\n", var_name1, *var_ptr, description);
}
}
void ihipReadEnv_S(std::string *var_ptr, const char *var_name1, const char *var_name2, const char *description)
{
char * env = getenv(var_name1);
void ihipReadEnv_S(std::string* var_ptr, const char* var_name1, const char* var_name2,
const char* description) {
char* env = getenv(var_name1);
// Check second name if first not defined, used to allow HIP_ or CUDA_ env vars.
if ((env == NULL) && strcmp(var_name2, "0")) {
@@ -60,14 +60,15 @@ void ihipReadEnv_S(std::string *var_ptr, const char *var_name1, const char *var_
*static_cast<std::string*>(var_ptr) = env;
}
if (HIP_PRINT_ENV) {
printf ("%-30s = %s : %s\n", var_name1, var_ptr->c_str(), description);
printf("%-30s = %s : %s\n", var_name1, var_ptr->c_str(), description);
}
}
void ihipReadEnv_Callback(void *var_ptr, const char *var_name1, const char *var_name2, const char *description, std::string (*setterCallback)(void * var_ptr, const char * env))
{
char * env = getenv(var_name1);
void ihipReadEnv_Callback(void* var_ptr, const char* var_name1, const char* var_name2,
const char* description,
std::string (*setterCallback)(void* var_ptr, const char* env)) {
char* env = getenv(var_name1);
// Check second name if first not defined, used to allow HIP_ or CUDA_ env vars.
if ((env == NULL) && strcmp(var_name2, "0")) {
@@ -79,35 +80,30 @@ void ihipReadEnv_Callback(void *var_ptr, const char *var_name1, const char *var_
var_string = setterCallback(var_ptr, env);
}
if (HIP_PRINT_ENV) {
printf ("%-30s = %s : %s\n", var_name1, var_string.c_str(), description);
printf("%-30s = %s : %s\n", var_name1, var_string.c_str(), description);
}
}
void tokenize(const std::string &s, char delim, std::vector<std::string> *tokens)
{
void tokenize(const std::string& s, char delim, std::vector<std::string>* tokens) {
std::stringstream ss;
ss.str(s);
std::string item;
while (getline(ss, item, delim)) {
item.erase (std::remove (item.begin(), item.end(), ' '), item.end()); // remove whitespace.
item.erase(std::remove(item.begin(), item.end(), ' '), item.end()); // remove whitespace.
tokens->push_back(item);
}
}
void trim(std::string *s)
{
void trim(std::string* s) {
// trim whitespace from beginning and end:
const char *t = "\t\n\r\f\v";
const char* t = "\t\n\r\f\v";
s->erase(0, s->find_first_not_of(t));
s->erase(s->find_last_not_of(t)+1);
s->erase(s->find_last_not_of(t) + 1);
}
static void ltrim(std::string *s)
{
static void ltrim(std::string* s) {
// trim whitespace from beginning
const char *t = "\t\n\r\f\v";
const char* t = "\t\n\r\f\v";
s->erase(0, s->find_first_not_of(t));
}
+12 -9
View File
@@ -3,22 +3,25 @@
extern void HipReadEnv();
#define READ_ENV_I(_build, _ENV_VAR, _ENV_VAR2, _description) \
#define READ_ENV_I(_build, _ENV_VAR, _ENV_VAR2, _description) \
ihipReadEnv_I(&_ENV_VAR, #_ENV_VAR, #_ENV_VAR2, _description);
#define READ_ENV_S(_build, _ENV_VAR, _ENV_VAR2, _description) \
#define READ_ENV_S(_build, _ENV_VAR, _ENV_VAR2, _description) \
ihipReadEnv_S(&_ENV_VAR, #_ENV_VAR, #_ENV_VAR2, _description);
#define READ_ENV_C(_build, _ENV_VAR, _ENV_VAR2, _description, _callback) \
#define READ_ENV_C(_build, _ENV_VAR, _ENV_VAR2, _description, _callback) \
ihipReadEnv_Callback(&_ENV_VAR, #_ENV_VAR, #_ENV_VAR2, _description, _callback);
extern void ihipReadEnv_I(int *var_ptr, const char *var_name1, const char *var_name2, const char *description);
extern void ihipReadEnv_S(std::string *var_ptr, const char *var_name1, const char *var_name2, const char *description);
extern void ihipReadEnv_Callback(void *var_ptr, const char *var_name1, const char *var_name2, const char *description, std::string (*setterCallback)(void * var_ptr, const char * env));
extern void ihipReadEnv_I(int* var_ptr, const char* var_name1, const char* var_name2,
const char* description);
extern void ihipReadEnv_S(std::string* var_ptr, const char* var_name1, const char* var_name2,
const char* description);
extern void ihipReadEnv_Callback(void* var_ptr, const char* var_name1, const char* var_name2,
const char* description,
std::string (*setterCallback)(void* var_ptr, const char* env));
// String functions:
extern void trim(std::string *s);
extern void tokenize(const std::string &s, char delim, std::vector<std::string> *tokens);
extern void trim(std::string* s);
extern void tokenize(const std::string& s, char delim, std::vector<std::string>* tokens);
+2 -2
View File
@@ -23,7 +23,7 @@ THE SOFTWARE.
#include "hip/hcc_detail/grid_launch_GGL.hpp"
#if __hcc_workweek__ >= 17481
#include "functional_grid_launch.inl"
#include "functional_grid_launch.inl"
#else
#include "macro_based_grid_launch.inl"
#include "macro_based_grid_launch.inl"
#endif
+64 -88
View File
@@ -30,18 +30,16 @@ THE SOFTWARE.
#include "trace_helper.h"
// Stack of contexts
thread_local std::stack<ihipCtx_t *> tls_ctxStack;
thread_local std::stack<ihipCtx_t*> tls_ctxStack;
thread_local bool tls_getPrimaryCtx = true;
void ihipCtxStackUpdate()
{
if(tls_ctxStack.empty()) {
tls_ctxStack.push(ihipGetTlsDefaultCtx());
void ihipCtxStackUpdate() {
if (tls_ctxStack.empty()) {
tls_ctxStack.push(ihipGetTlsDefaultCtx());
}
}
hipError_t hipInit(unsigned int flags)
{
hipError_t hipInit(unsigned int flags) {
HIP_INIT_API(flags);
hipError_t e = hipSuccess;
@@ -54,28 +52,26 @@ hipError_t hipInit(unsigned int flags)
return ihipLogStatus(e);
}
hipError_t hipCtxCreate(hipCtx_t *ctx, unsigned int flags, hipDevice_t device)
{
HIP_INIT_API(ctx, flags, device); // FIXME - review if we want to init
hipError_t hipCtxCreate(hipCtx_t* ctx, unsigned int flags, hipDevice_t device) {
HIP_INIT_API(ctx, flags, device); // FIXME - review if we want to init
hipError_t e = hipSuccess;
auto deviceHandle = ihipGetDevice(device);
{
// Obtain mutex access to the device critical data, release by destructor
LockedAccessor_DeviceCrit_t deviceCrit(deviceHandle->criticalData());
auto ictx = new ihipCtx_t(deviceHandle, g_deviceCnt, flags);
*ctx = ictx;
ihipSetTlsDefaultCtx(*ctx);
tls_ctxStack.push(*ctx);
// Obtain mutex access to the device critical data, release by destructor
LockedAccessor_DeviceCrit_t deviceCrit(deviceHandle->criticalData());
auto ictx = new ihipCtx_t(deviceHandle, g_deviceCnt, flags);
*ctx = ictx;
ihipSetTlsDefaultCtx(*ctx);
tls_ctxStack.push(*ctx);
tls_getPrimaryCtx = false;
deviceCrit->addContext(ictx);
deviceCrit->addContext(ictx);
}
return ihipLogStatus(e);
}
hipError_t hipDeviceGet(hipDevice_t *device, int deviceId)
{
HIP_INIT_API(device, deviceId); // FIXME - review if we want to init
hipError_t hipDeviceGet(hipDevice_t* device, int deviceId) {
HIP_INIT_API(device, deviceId); // FIXME - review if we want to init
auto deviceHandle = ihipGetDevice(deviceId);
@@ -89,8 +85,7 @@ hipError_t hipDeviceGet(hipDevice_t *device, int deviceId)
return ihipLogStatus(e);
};
hipError_t hipDriverGetVersion(int *driverVersion)
{
hipError_t hipDriverGetVersion(int* driverVersion) {
HIP_INIT_API(driverVersion);
hipError_t e = hipSuccess;
if (driverVersion) {
@@ -102,8 +97,7 @@ hipError_t hipDriverGetVersion(int *driverVersion)
return ihipLogStatus(e);
}
hipError_t hipRuntimeGetVersion(int *runtimeVersion)
{
hipError_t hipRuntimeGetVersion(int* runtimeVersion) {
HIP_INIT_API(runtimeVersion);
hipError_t e = hipSuccess;
if (runtimeVersion) {
@@ -115,58 +109,55 @@ hipError_t hipRuntimeGetVersion(int *runtimeVersion)
return ihipLogStatus(e);
}
hipError_t hipCtxDestroy(hipCtx_t ctx)
{
hipError_t hipCtxDestroy(hipCtx_t ctx) {
HIP_INIT_API(ctx);
hipError_t e = hipSuccess;
ihipCtx_t* currentCtx= ihipGetTlsDefaultCtx();
ihipCtx_t* primaryCtx= ((ihipDevice_t*)ctx->getDevice())->_primaryCtx;
if(primaryCtx== ctx)
{
ihipCtx_t* currentCtx = ihipGetTlsDefaultCtx();
ihipCtx_t* primaryCtx = ((ihipDevice_t*)ctx->getDevice())->_primaryCtx;
if (primaryCtx == ctx) {
e = hipErrorInvalidValue;
} else {
if(currentCtx == ctx) {
//need to destroy the ctx associated with calling thread
if (currentCtx == ctx) {
// need to destroy the ctx associated with calling thread
tls_ctxStack.pop();
}
{
auto deviceHandle = ctx->getWriteableDevice();
deviceHandle->locked_removeContext(ctx);
ctx->locked_reset();
}
delete ctx; //As per CUDA docs , attempting to access ctx from those threads which has this ctx as current, will result in the error HIP_ERROR_CONTEXT_IS_DESTROYED.
auto deviceHandle = ctx->getWriteableDevice();
deviceHandle->locked_removeContext(ctx);
ctx->locked_reset();
}
delete ctx; // As per CUDA docs , attempting to access ctx from those threads which has
// this ctx as current, will result in the error HIP_ERROR_CONTEXT_IS_DESTROYED.
}
return ihipLogStatus(e);
}
hipError_t hipCtxPopCurrent(hipCtx_t* ctx)
{
hipError_t hipCtxPopCurrent(hipCtx_t* ctx) {
HIP_INIT_API(ctx);
hipError_t e = hipSuccess;
ihipCtx_t* currentCtx = ihipGetTlsDefaultCtx();
auto deviceHandle = currentCtx->getDevice();
*ctx = currentCtx;
if(!tls_ctxStack.empty()) {
if (!tls_ctxStack.empty()) {
tls_ctxStack.pop();
}
if(!tls_ctxStack.empty()) {
currentCtx= tls_ctxStack.top();
if (!tls_ctxStack.empty()) {
currentCtx = tls_ctxStack.top();
} else {
currentCtx = deviceHandle->_primaryCtx;
}
ihipSetTlsDefaultCtx(currentCtx); //TOD0 - Shall check for NULL?
ihipSetTlsDefaultCtx(currentCtx); // TOD0 - Shall check for NULL?
return ihipLogStatus(e);
}
hipError_t hipCtxPushCurrent(hipCtx_t ctx)
{
hipError_t hipCtxPushCurrent(hipCtx_t ctx) {
HIP_INIT_API(ctx);
hipError_t e = hipSuccess;
if(ctx != NULL) { //TODO- is this check needed?
if (ctx != NULL) { // TODO- is this check needed?
ihipSetTlsDefaultCtx(ctx);
tls_ctxStack.push(ctx);
tls_getPrimaryCtx = false;
@@ -176,23 +167,21 @@ hipError_t hipCtxPushCurrent(hipCtx_t ctx)
return ihipLogStatus(e);
}
hipError_t hipCtxGetCurrent(hipCtx_t* ctx)
{
hipError_t hipCtxGetCurrent(hipCtx_t* ctx) {
HIP_INIT_API(ctx);
hipError_t e = hipSuccess;
if((tls_getPrimaryCtx) || tls_ctxStack.empty()) {
if ((tls_getPrimaryCtx) || tls_ctxStack.empty()) {
*ctx = ihipGetTlsDefaultCtx();
} else {
*ctx= tls_ctxStack.top();
*ctx = tls_ctxStack.top();
}
return ihipLogStatus(e);
}
hipError_t hipCtxSetCurrent(hipCtx_t ctx)
{
hipError_t hipCtxSetCurrent(hipCtx_t ctx) {
HIP_INIT_API(ctx);
hipError_t e = hipSuccess;
if(ctx == NULL) {
if (ctx == NULL) {
tls_ctxStack.pop();
} else {
ihipSetTlsDefaultCtx(ctx);
@@ -202,14 +191,13 @@ hipError_t hipCtxSetCurrent(hipCtx_t ctx)
return ihipLogStatus(e);
}
hipError_t hipCtxGetDevice(hipDevice_t *device)
{
hipError_t hipCtxGetDevice(hipDevice_t* device) {
HIP_INIT_API(device);
hipError_t e = hipSuccess;
ihipCtx_t *ctx = ihipGetTlsDefaultCtx();
ihipCtx_t* ctx = ihipGetTlsDefaultCtx();
if(ctx == nullptr) {
if (ctx == nullptr) {
e = hipErrorInvalidContext;
// TODO *device = nullptr;
} else {
@@ -219,8 +207,7 @@ hipError_t hipCtxGetDevice(hipDevice_t *device)
return ihipLogStatus(e);
}
hipError_t hipCtxGetApiVersion (hipCtx_t ctx,int *apiVersion)
{
hipError_t hipCtxGetApiVersion(hipCtx_t ctx, int* apiVersion) {
HIP_INIT_API(apiVersion);
if (apiVersion) {
@@ -230,8 +217,7 @@ hipError_t hipCtxGetApiVersion (hipCtx_t ctx,int *apiVersion)
return ihipLogStatus(hipSuccess);
}
hipError_t hipCtxGetCacheConfig ( hipFuncCache_t *cacheConfig )
{
hipError_t hipCtxGetCacheConfig(hipFuncCache_t* cacheConfig) {
HIP_INIT_API(cacheConfig);
*cacheConfig = hipFuncCachePreferNone;
@@ -239,8 +225,7 @@ hipError_t hipCtxGetCacheConfig ( hipFuncCache_t *cacheConfig )
return ihipLogStatus(hipSuccess);
}
hipError_t hipCtxSetCacheConfig ( hipFuncCache_t cacheConfig )
{
hipError_t hipCtxSetCacheConfig(hipFuncCache_t cacheConfig) {
HIP_INIT_API(cacheConfig);
// Nop, AMD does not support variable cache configs.
@@ -248,8 +233,7 @@ hipError_t hipCtxSetCacheConfig ( hipFuncCache_t cacheConfig )
return ihipLogStatus(hipSuccess);
}
hipError_t hipCtxSetSharedMemConfig ( hipSharedMemConfig config )
{
hipError_t hipCtxSetSharedMemConfig(hipSharedMemConfig config) {
HIP_INIT_API(config);
// Nop, AMD does not support variable shared mem configs.
@@ -257,8 +241,7 @@ hipError_t hipCtxSetSharedMemConfig ( hipSharedMemConfig config )
return ihipLogStatus(hipSuccess);
}
hipError_t hipCtxGetSharedMemConfig ( hipSharedMemConfig * pConfig )
{
hipError_t hipCtxGetSharedMemConfig(hipSharedMemConfig* pConfig) {
HIP_INIT_API(pConfig);
*pConfig = hipSharedMemBankSizeFourByte;
@@ -266,14 +249,12 @@ hipError_t hipCtxGetSharedMemConfig ( hipSharedMemConfig * pConfig )
return ihipLogStatus(hipSuccess);
}
hipError_t hipCtxSynchronize ( void )
{
hipError_t hipCtxSynchronize(void) {
HIP_INIT_API(1);
return ihipLogStatus(ihipSynchronize()); //TODP Shall check validity of ctx?
return ihipLogStatus(ihipSynchronize()); // TODP Shall check validity of ctx?
}
hipError_t hipCtxGetFlags ( unsigned int* flags )
{
hipError_t hipCtxGetFlags(unsigned int* flags) {
HIP_INIT_API(flags);
hipError_t e = hipSuccess;
ihipCtx_t* tempCtx;
@@ -282,8 +263,7 @@ hipError_t hipCtxGetFlags ( unsigned int* flags )
return ihipLogStatus(e);
}
hipError_t hipDevicePrimaryCtxGetState ( hipDevice_t dev, unsigned int* flags, int* active )
{
hipError_t hipDevicePrimaryCtxGetState(hipDevice_t dev, unsigned int* flags, int* active) {
HIP_INIT_API(dev, flags, active);
hipError_t e = hipSuccess;
auto deviceHandle = ihipGetDevice(dev);
@@ -295,18 +275,17 @@ hipError_t hipDevicePrimaryCtxGetState ( hipDevice_t dev, unsigned int* flags, i
ihipCtx_t* tempCtx;
tempCtx = ihipGetTlsDefaultCtx();
ihipCtx_t* primaryCtx = deviceHandle->_primaryCtx;
if(tempCtx == primaryCtx) {
if (tempCtx == primaryCtx) {
*active = 1;
*flags = tempCtx->_ctxFlags;
} else {
*active = 0;
*flags = primaryCtx->_ctxFlags;
}
return ihipLogStatus(e);
} else {
*active = 0;
*flags = primaryCtx->_ctxFlags;
}
return ihipLogStatus(e);
}
hipError_t hipDevicePrimaryCtxRelease ( hipDevice_t dev)
{
hipError_t hipDevicePrimaryCtxRelease(hipDevice_t dev) {
HIP_INIT_API(dev);
hipError_t e = hipSuccess;
auto deviceHandle = ihipGetDevice(dev);
@@ -317,8 +296,7 @@ hipError_t hipDevicePrimaryCtxRelease ( hipDevice_t dev)
return ihipLogStatus(e);
}
hipError_t hipDevicePrimaryCtxRetain ( hipCtx_t* pctx, hipDevice_t dev )
{
hipError_t hipDevicePrimaryCtxRetain(hipCtx_t* pctx, hipDevice_t dev) {
HIP_INIT_API(pctx, dev);
hipError_t e = hipSuccess;
auto deviceHandle = ihipGetDevice(dev);
@@ -330,8 +308,7 @@ hipError_t hipDevicePrimaryCtxRetain ( hipCtx_t* pctx, hipDevice_t dev )
return ihipLogStatus(e);
}
hipError_t hipDevicePrimaryCtxReset ( hipDevice_t dev )
{
hipError_t hipDevicePrimaryCtxReset(hipDevice_t dev) {
HIP_INIT_API(dev);
hipError_t e = hipSuccess;
auto deviceHandle = ihipGetDevice(dev);
@@ -344,8 +321,7 @@ hipError_t hipDevicePrimaryCtxReset ( hipDevice_t dev )
return ihipLogStatus(e);
}
hipError_t hipDevicePrimaryCtxSetFlags ( hipDevice_t dev, unsigned int flags )
{
hipError_t hipDevicePrimaryCtxSetFlags(hipDevice_t dev, unsigned int flags) {
HIP_INIT_API(dev, flags);
hipError_t e = hipSuccess;
auto deviceHandle = ihipGetDevice(dev);
+1 -8
View File
@@ -2,11 +2,4 @@
#include <hc_am.hpp>
void hipdbPrintMem(void *targetAddress)
{
hc::am_memtracker_print(targetAddress);
};
void hipdbPrintMem(void* targetAddress) { hc::am_memtracker_print(targetAddress); };
+208 -205
View File
@@ -26,25 +26,24 @@ THE SOFTWARE.
#include "device_util.h"
//-------------------------------------------------------------------------------------------------
//Devices
// Devices
//-------------------------------------------------------------------------------------------------
// TODO - does this initialize HIP runtime?
hipError_t hipGetDevice(int *deviceId)
{
hipError_t hipGetDevice(int* deviceId) {
HIP_INIT_API(deviceId);
hipError_t e = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if(deviceId != nullptr){
if (deviceId != nullptr) {
if (ctx == nullptr) {
e = hipErrorInvalidDevice; // TODO, check error code.
e = hipErrorInvalidDevice; // TODO, check error code.
*deviceId = -1;
} else {
*deviceId = ctx->getDevice()->_deviceId;
}
}else{
} else {
e = hipErrorInvalidValue;
}
@@ -52,11 +51,10 @@ hipError_t hipGetDevice(int *deviceId)
}
// TODO - does this initialize HIP runtime?
hipError_t ihipGetDeviceCount(int *count)
{
hipError_t ihipGetDeviceCount(int* count) {
hipError_t e = hipSuccess;
if(count != nullptr) {
if (count != nullptr) {
*count = g_deviceCnt;
if (*count > 0) {
@@ -70,14 +68,12 @@ hipError_t ihipGetDeviceCount(int *count)
return e;
}
hipError_t hipGetDeviceCount(int *count)
{
hipError_t hipGetDeviceCount(int* count) {
HIP_INIT_API(count);
return ihipLogStatus(ihipGetDeviceCount(count));
}
hipError_t hipDeviceSetCacheConfig(hipFuncCache_t cacheConfig)
{
hipError_t hipDeviceSetCacheConfig(hipFuncCache_t cacheConfig) {
HIP_INIT_API(cacheConfig);
// Nop, AMD does not support variable cache configs.
@@ -85,11 +81,10 @@ hipError_t hipDeviceSetCacheConfig(hipFuncCache_t cacheConfig)
return ihipLogStatus(hipSuccess);
}
hipError_t hipDeviceGetCacheConfig(hipFuncCache_t *cacheConfig)
{
hipError_t hipDeviceGetCacheConfig(hipFuncCache_t* cacheConfig) {
HIP_INIT_API(cacheConfig);
if(cacheConfig == nullptr) {
if (cacheConfig == nullptr) {
return ihipLogStatus(hipErrorInvalidValue);
}
@@ -98,22 +93,20 @@ hipError_t hipDeviceGetCacheConfig(hipFuncCache_t *cacheConfig)
return ihipLogStatus(hipSuccess);
}
hipError_t hipDeviceGetLimit (size_t *pValue, hipLimit_t limit)
{
hipError_t hipDeviceGetLimit(size_t* pValue, hipLimit_t limit) {
HIP_INIT_API(pValue, limit);
if(pValue == nullptr) {
if (pValue == nullptr) {
return ihipLogStatus(hipErrorInvalidValue);
}
if(limit == hipLimitMallocHeapSize) {
if (limit == hipLimitMallocHeapSize) {
*pValue = (size_t)SIZE_OF_HEAP;
return ihipLogStatus(hipSuccess);
}else{
} else {
return ihipLogStatus(hipErrorUnsupportedLimit);
}
}
hipError_t hipFuncSetCacheConfig (const void* func, hipFuncCache_t cacheConfig)
{
hipError_t hipFuncSetCacheConfig(const void* func, hipFuncCache_t cacheConfig) {
HIP_INIT_API(cacheConfig);
// Nop, AMD does not support variable cache configs.
@@ -121,8 +114,7 @@ hipError_t hipFuncSetCacheConfig (const void* func, hipFuncCache_t cacheConfig)
return ihipLogStatus(hipSuccess);
}
hipError_t hipDeviceSetSharedMemConfig (hipSharedMemConfig config)
{
hipError_t hipDeviceSetSharedMemConfig(hipSharedMemConfig config) {
HIP_INIT_API(config);
// Nop, AMD does not support variable shared mem configs.
@@ -130,8 +122,7 @@ hipError_t hipDeviceSetSharedMemConfig (hipSharedMemConfig config)
return ihipLogStatus(hipSuccess);
}
hipError_t hipDeviceGetSharedMemConfig (hipSharedMemConfig *pConfig)
{
hipError_t hipDeviceGetSharedMemConfig(hipSharedMemConfig* pConfig) {
HIP_INIT_API(pConfig);
*pConfig = hipSharedMemBankSizeFourByte;
@@ -139,8 +130,7 @@ hipError_t hipDeviceGetSharedMemConfig (hipSharedMemConfig *pConfig)
return ihipLogStatus(hipSuccess);
}
hipError_t hipSetDevice(int deviceId)
{
hipError_t hipSetDevice(int deviceId) {
HIP_INIT_API(deviceId);
if ((deviceId < 0) || (deviceId >= g_deviceCnt)) {
return ihipLogStatus(hipErrorInvalidDevice);
@@ -151,21 +141,20 @@ hipError_t hipSetDevice(int deviceId)
}
}
hipError_t hipDeviceSynchronize(void)
{
hipError_t hipDeviceSynchronize(void) {
HIP_INIT_SPECIAL_API(TRACE_SYNC);
return ihipLogStatus(ihipSynchronize());
}
hipError_t hipDeviceReset(void)
{
hipError_t hipDeviceReset(void) {
HIP_INIT_API();
auto *ctx = ihipGetTlsDefaultCtx();
auto* ctx = ihipGetTlsDefaultCtx();
// TODO-HCC
// This function currently does a user-level cleanup of known resources.
// It could benefit from KFD support to perform a more "nuclear" clean that would include any associated kernel resources and page table entries.
// It could benefit from KFD support to perform a more "nuclear" clean that would include any
// associated kernel resources and page table entries.
#if 0
if (ctx) {
@@ -173,24 +162,22 @@ hipError_t hipDeviceReset(void)
ctx->locked_reset();
}
#endif
if (ctx) {
ihipDevice_t *deviceHandle = ctx->getWriteableDevice();
deviceHandle->locked_reset();
}
if (ctx) {
ihipDevice_t* deviceHandle = ctx->getWriteableDevice();
deviceHandle->locked_reset();
}
return ihipLogStatus(hipSuccess);
}
hipError_t ihipDeviceSetState(void)
{
hipError_t ihipDeviceSetState(void) {
hipError_t e = hipErrorInvalidContext;
auto *ctx = ihipGetTlsDefaultCtx();
auto* ctx = ihipGetTlsDefaultCtx();
if (ctx) {
ihipDevice_t *deviceHandle = ctx->getWriteableDevice();
if(deviceHandle->_state == 0)
{
ihipDevice_t* deviceHandle = ctx->getWriteableDevice();
if (deviceHandle->_state == 0) {
deviceHandle->_state = 1;
}
e = hipSuccess;
@@ -200,70 +187,95 @@ hipError_t ihipDeviceSetState(void)
}
hipError_t ihipDeviceGetAttribute(int* pi, hipDeviceAttribute_t attr, int device)
{
hipError_t ihipDeviceGetAttribute(int* pi, hipDeviceAttribute_t attr, int device) {
hipError_t e = hipSuccess;
if(pi == nullptr) {
if (pi == nullptr) {
return hipErrorInvalidValue;
}
auto * hipDevice = ihipGetDevice(device);
hipDeviceProp_t *prop = &hipDevice->_props;
auto* hipDevice = ihipGetDevice(device);
hipDeviceProp_t* prop = &hipDevice->_props;
if (hipDevice) {
switch (attr) {
case hipDeviceAttributeMaxThreadsPerBlock:
*pi = prop->maxThreadsPerBlock; break;
case hipDeviceAttributeMaxBlockDimX:
*pi = prop->maxThreadsDim[0]; break;
case hipDeviceAttributeMaxBlockDimY:
*pi = prop->maxThreadsDim[1]; break;
case hipDeviceAttributeMaxBlockDimZ:
*pi = prop->maxThreadsDim[2]; break;
case hipDeviceAttributeMaxGridDimX:
*pi = prop->maxGridSize[0]; break;
case hipDeviceAttributeMaxGridDimY:
*pi = prop->maxGridSize[1]; break;
case hipDeviceAttributeMaxGridDimZ:
*pi = prop->maxGridSize[2]; break;
case hipDeviceAttributeMaxSharedMemoryPerBlock:
*pi = prop->sharedMemPerBlock; break;
case hipDeviceAttributeTotalConstantMemory:
*pi = prop->totalConstMem; break;
case hipDeviceAttributeWarpSize:
*pi = prop->warpSize; break;
case hipDeviceAttributeMaxRegistersPerBlock:
*pi = prop->regsPerBlock; break;
case hipDeviceAttributeClockRate:
*pi = prop->clockRate; break;
case hipDeviceAttributeMemoryClockRate:
*pi = prop->memoryClockRate; break;
case hipDeviceAttributeMemoryBusWidth:
*pi = prop->memoryBusWidth; break;
case hipDeviceAttributeMultiprocessorCount:
*pi = prop->multiProcessorCount; break;
case hipDeviceAttributeComputeMode:
*pi = prop->computeMode; break;
case hipDeviceAttributeL2CacheSize:
*pi = prop->l2CacheSize; break;
case hipDeviceAttributeMaxThreadsPerMultiProcessor:
*pi = prop->maxThreadsPerMultiProcessor; break;
case hipDeviceAttributeComputeCapabilityMajor:
*pi = prop->major; break;
case hipDeviceAttributeComputeCapabilityMinor:
*pi = prop->minor; break;
case hipDeviceAttributePciBusId:
*pi = prop->pciBusID; break;
case hipDeviceAttributeConcurrentKernels:
*pi = prop->concurrentKernels; break;
case hipDeviceAttributePciDeviceId:
*pi = prop->pciDeviceID; break;
case hipDeviceAttributeMaxSharedMemoryPerMultiprocessor:
*pi = prop->maxSharedMemoryPerMultiProcessor; break;
case hipDeviceAttributeIsMultiGpuBoard:
*pi = prop->isMultiGpuBoard; break;
default:
e = hipErrorInvalidValue; break;
case hipDeviceAttributeMaxThreadsPerBlock:
*pi = prop->maxThreadsPerBlock;
break;
case hipDeviceAttributeMaxBlockDimX:
*pi = prop->maxThreadsDim[0];
break;
case hipDeviceAttributeMaxBlockDimY:
*pi = prop->maxThreadsDim[1];
break;
case hipDeviceAttributeMaxBlockDimZ:
*pi = prop->maxThreadsDim[2];
break;
case hipDeviceAttributeMaxGridDimX:
*pi = prop->maxGridSize[0];
break;
case hipDeviceAttributeMaxGridDimY:
*pi = prop->maxGridSize[1];
break;
case hipDeviceAttributeMaxGridDimZ:
*pi = prop->maxGridSize[2];
break;
case hipDeviceAttributeMaxSharedMemoryPerBlock:
*pi = prop->sharedMemPerBlock;
break;
case hipDeviceAttributeTotalConstantMemory:
*pi = prop->totalConstMem;
break;
case hipDeviceAttributeWarpSize:
*pi = prop->warpSize;
break;
case hipDeviceAttributeMaxRegistersPerBlock:
*pi = prop->regsPerBlock;
break;
case hipDeviceAttributeClockRate:
*pi = prop->clockRate;
break;
case hipDeviceAttributeMemoryClockRate:
*pi = prop->memoryClockRate;
break;
case hipDeviceAttributeMemoryBusWidth:
*pi = prop->memoryBusWidth;
break;
case hipDeviceAttributeMultiprocessorCount:
*pi = prop->multiProcessorCount;
break;
case hipDeviceAttributeComputeMode:
*pi = prop->computeMode;
break;
case hipDeviceAttributeL2CacheSize:
*pi = prop->l2CacheSize;
break;
case hipDeviceAttributeMaxThreadsPerMultiProcessor:
*pi = prop->maxThreadsPerMultiProcessor;
break;
case hipDeviceAttributeComputeCapabilityMajor:
*pi = prop->major;
break;
case hipDeviceAttributeComputeCapabilityMinor:
*pi = prop->minor;
break;
case hipDeviceAttributePciBusId:
*pi = prop->pciBusID;
break;
case hipDeviceAttributeConcurrentKernels:
*pi = prop->concurrentKernels;
break;
case hipDeviceAttributePciDeviceId:
*pi = prop->pciDeviceID;
break;
case hipDeviceAttributeMaxSharedMemoryPerMultiprocessor:
*pi = prop->maxSharedMemoryPerMultiProcessor;
break;
case hipDeviceAttributeIsMultiGpuBoard:
*pi = prop->isMultiGpuBoard;
break;
default:
e = hipErrorInvalidValue;
break;
}
} else {
e = hipErrorInvalidDevice;
@@ -271,37 +283,34 @@ hipError_t ihipDeviceGetAttribute(int* pi, hipDeviceAttribute_t attr, int device
return e;
}
hipError_t hipDeviceGetAttribute(int* pi, hipDeviceAttribute_t attr, int device)
{
hipError_t hipDeviceGetAttribute(int* pi, hipDeviceAttribute_t attr, int device) {
HIP_INIT_API(pi, attr, device);
if ((device < 0) || (device >= g_deviceCnt)) {
return ihipLogStatus(hipErrorInvalidDevice);
}
return ihipLogStatus(ihipDeviceGetAttribute(pi,attr,device));
}
return ihipLogStatus(ihipDeviceGetAttribute(pi, attr, device));
}
hipError_t ihipGetDeviceProperties(hipDeviceProp_t* props, int device)
{
hipError_t ihipGetDeviceProperties(hipDeviceProp_t* props, int device) {
hipError_t e;
if(props != nullptr){
auto * hipDevice = ihipGetDevice(device);
if (props != nullptr) {
auto* hipDevice = ihipGetDevice(device);
if (hipDevice) {
// copy saved props
// copy saved props
*props = hipDevice->_props;
e = hipSuccess;
} else {
e = hipErrorInvalidDevice;
}
}else{
} else {
e = hipErrorInvalidDevice;
}
return e;
}
hipError_t hipGetDeviceProperties(hipDeviceProp_t* props, int device)
{
hipError_t hipGetDeviceProperties(hipDeviceProp_t* props, int device) {
HIP_INIT_API(props, device);
if ((device < 0) || (device >= g_deviceCnt)) {
return ihipLogStatus(hipErrorInvalidDevice);
@@ -309,54 +318,53 @@ hipError_t hipGetDeviceProperties(hipDeviceProp_t* props, int device)
return ihipLogStatus(ihipGetDeviceProperties(props, device));
}
hipError_t hipSetDeviceFlags( unsigned int flags)
{
hipError_t hipSetDeviceFlags(unsigned int flags) {
HIP_INIT_API(flags);
hipError_t e = hipSuccess;
auto * ctx = ihipGetTlsDefaultCtx();
auto* ctx = ihipGetTlsDefaultCtx();
// TODO : does this really OR in the flags or replaces previous flags:
// TODO : Review error handling behavior for this function, it often returns ErrorSetOnActiveProcess
// TODO : Review error handling behavior for this function, it often returns
// ErrorSetOnActiveProcess
if (ctx) {
auto *deviceHandle = ctx->getDevice();
if(deviceHandle->_state == 0)
{
ctx->_ctxFlags = ctx->_ctxFlags | flags;
if (flags & hipDeviceScheduleMask) {
switch (hipDeviceScheduleMask) {
case hipDeviceScheduleAuto:
case hipDeviceScheduleSpin:
case hipDeviceScheduleYield:
case hipDeviceScheduleBlockingSync:
e = hipSuccess;
break;
default:
e = hipSuccess; // TODO - should this be error? Map to Auto?
//e = hipErrorInvalidValue;
break;
}
}
auto* deviceHandle = ctx->getDevice();
if (deviceHandle->_state == 0) {
ctx->_ctxFlags = ctx->_ctxFlags | flags;
if (flags & hipDeviceScheduleMask) {
switch (hipDeviceScheduleMask) {
case hipDeviceScheduleAuto:
case hipDeviceScheduleSpin:
case hipDeviceScheduleYield:
case hipDeviceScheduleBlockingSync:
e = hipSuccess;
break;
default:
e = hipSuccess; // TODO - should this be error? Map to Auto?
// e = hipErrorInvalidValue;
break;
}
}
unsigned supportedFlags = hipDeviceScheduleMask | hipDeviceMapHost | hipDeviceLmemResizeToMax;
unsigned supportedFlags =
hipDeviceScheduleMask | hipDeviceMapHost | hipDeviceLmemResizeToMax;
if (flags & (~supportedFlags)) {
e = hipErrorInvalidValue;
}
if (flags & (~supportedFlags)) {
e = hipErrorInvalidValue;
}
} else {
e = hipErrorSetOnActiveProcess;
e = hipErrorSetOnActiveProcess;
}
} else {
e = hipErrorInvalidDevice;
} else {
e = hipErrorInvalidDevice;
}
return ihipLogStatus(e);
};
hipError_t hipDeviceComputeCapability(int *major, int *minor, hipDevice_t device)
{
HIP_INIT_API(major,minor, device);
hipError_t hipDeviceComputeCapability(int* major, int* minor, hipDevice_t device) {
HIP_INIT_API(major, minor, device);
hipError_t e = hipSuccess;
if ((device < 0) || (device >= g_deviceCnt)) {
e = hipErrorInvalidDevice;
@@ -367,34 +375,33 @@ hipError_t hipDeviceComputeCapability(int *major, int *minor, hipDevice_t device
return ihipLogStatus(e);
}
hipError_t hipDeviceGetName(char *name,int len,hipDevice_t device)
{
hipError_t hipDeviceGetName(char* name, int len, hipDevice_t device) {
// Cast to void* here to avoid printing garbage in debug modes.
HIP_INIT_API((void*)name,len, device);
HIP_INIT_API((void*)name, len, device);
hipError_t e = hipSuccess;
if ((device < 0) || (device >= g_deviceCnt)) {
e = hipErrorInvalidDevice;
} else {
auto deviceHandle = ihipGetDevice(device);
int nameLen = strlen(deviceHandle->_props.name);
if(nameLen <= len)
memcpy(name,deviceHandle->_props.name,nameLen);
if (nameLen <= len) memcpy(name, deviceHandle->_props.name, nameLen);
}
return ihipLogStatus(e);
}
hipError_t hipDeviceGetPCIBusId (char *pciBusId,int len, int device)
{
hipError_t hipDeviceGetPCIBusId(char* pciBusId, int len, int device) {
// Cast to void* here to avoid printing garbage in debug modes.
HIP_INIT_API((void*)pciBusId, len, device);
hipError_t e = hipErrorInvalidValue;
if ((device < 0) || (device >= g_deviceCnt)) {
e = hipErrorInvalidDevice;
} else {
if((pciBusId != nullptr) && (len > 0)) {
if ((pciBusId != nullptr) && (len > 0)) {
auto deviceHandle = ihipGetDevice(device);
int retVal = snprintf(pciBusId,len, "%04x:%02x:%02x.0",deviceHandle->_props.pciDomainID,deviceHandle->_props.pciBusID,deviceHandle->_props.pciDeviceID);
if( retVal > 0 && retVal < len) {
int retVal =
snprintf(pciBusId, len, "%04x:%02x:%02x.0", deviceHandle->_props.pciDomainID,
deviceHandle->_props.pciBusID, deviceHandle->_props.pciDeviceID);
if (retVal > 0 && retVal < len) {
e = hipSuccess;
}
}
@@ -402,117 +409,114 @@ hipError_t hipDeviceGetPCIBusId (char *pciBusId,int len, int device)
return ihipLogStatus(e);
}
hipError_t hipDeviceTotalMem (size_t *bytes,hipDevice_t device)
{
hipError_t hipDeviceTotalMem(size_t* bytes, hipDevice_t device) {
HIP_INIT_API(bytes, device);
hipError_t e = hipSuccess;
if ((device < 0) || (device >= g_deviceCnt)) {
e = hipErrorInvalidDevice;
} else {
auto deviceHandle = ihipGetDevice(device);
*bytes= deviceHandle->_props.totalGlobalMem;
*bytes = deviceHandle->_props.totalGlobalMem;
}
return ihipLogStatus(e);
}
hipError_t hipDeviceGetByPCIBusId (int* device, const char* pciBusId )
{
HIP_INIT_API(device,pciBusId);
hipDeviceProp_t tempProp;
int deviceCount = 0 ;
hipError_t hipDeviceGetByPCIBusId(int* device, const char* pciBusId) {
HIP_INIT_API(device, pciBusId);
hipDeviceProp_t tempProp;
int deviceCount = 0;
hipError_t e = hipErrorInvalidValue;
if((device != nullptr) && (pciBusId != nullptr)) {
if ((device != nullptr) && (pciBusId != nullptr)) {
int pciBusID = -1;
int pciDeviceID = -1;
int pciDomainID = -1;
int len = 0;
len = sscanf (pciBusId,"%04x:%02x:%02x",&pciDomainID,&pciBusID,&pciDeviceID);
if(len == 3) {
ihipGetDeviceCount( &deviceCount );
for (int i = 0; i< deviceCount; i++) {
ihipGetDeviceProperties( &tempProp, i );
if(tempProp.pciBusID == pciBusID) {
*device = i;
e = hipSuccess;
break;
}
}
}
len = sscanf(pciBusId, "%04x:%02x:%02x", &pciDomainID, &pciBusID, &pciDeviceID);
if (len == 3) {
ihipGetDeviceCount(&deviceCount);
for (int i = 0; i < deviceCount; i++) {
ihipGetDeviceProperties(&tempProp, i);
if (tempProp.pciBusID == pciBusID) {
*device = i;
e = hipSuccess;
break;
}
}
}
}
return ihipLogStatus(e);
}
hipError_t hipChooseDevice( int* device, const hipDeviceProp_t* prop )
{
HIP_INIT_API(device,prop);
hipDeviceProp_t tempProp;
hipError_t hipChooseDevice(int* device, const hipDeviceProp_t* prop) {
HIP_INIT_API(device, prop);
hipDeviceProp_t tempProp;
hipError_t e = hipSuccess;
if((device == NULL) || (prop == NULL)) {
if ((device == NULL) || (prop == NULL)) {
e = hipErrorInvalidValue;
}
if(e == hipSuccess) {
if (e == hipSuccess) {
int deviceCount;
int inPropCount = 0;
int matchedPropCount = 0;
ihipGetDeviceCount( &deviceCount );
ihipGetDeviceCount(&deviceCount);
*device = 0;
for (int i = 0; i < deviceCount; i++) {
ihipGetDeviceProperties( &tempProp, i );
if(prop->major != 0) {
ihipGetDeviceProperties(&tempProp, i);
if (prop->major != 0) {
inPropCount++;
if(tempProp.major >= prop->major) {
if (tempProp.major >= prop->major) {
matchedPropCount++;
}
if(prop->minor != 0) {
if (prop->minor != 0) {
inPropCount++;
if(tempProp.minor >= prop->minor) {
if (tempProp.minor >= prop->minor) {
matchedPropCount++;
}
}
}
if(prop->totalGlobalMem != 0) {
if (prop->totalGlobalMem != 0) {
inPropCount++;
if(tempProp.totalGlobalMem >= prop->totalGlobalMem) {
if (tempProp.totalGlobalMem >= prop->totalGlobalMem) {
matchedPropCount++;
}
}
if(prop->sharedMemPerBlock != 0) {
if (prop->sharedMemPerBlock != 0) {
inPropCount++;
if(tempProp.sharedMemPerBlock >= prop->sharedMemPerBlock) {
if (tempProp.sharedMemPerBlock >= prop->sharedMemPerBlock) {
matchedPropCount++;
}
}
if(prop->maxThreadsPerBlock != 0) {
if (prop->maxThreadsPerBlock != 0) {
inPropCount++;
if(tempProp.maxThreadsPerBlock >= prop->maxThreadsPerBlock ) {
if (tempProp.maxThreadsPerBlock >= prop->maxThreadsPerBlock) {
matchedPropCount++;
}
}
if(prop->totalConstMem != 0) {
if (prop->totalConstMem != 0) {
inPropCount++;
if(tempProp.totalConstMem >= prop->totalConstMem ) {
if (tempProp.totalConstMem >= prop->totalConstMem) {
matchedPropCount++;
}
}
if(prop->multiProcessorCount != 0) {
if (prop->multiProcessorCount != 0) {
inPropCount++;
if(tempProp.multiProcessorCount >= prop->multiProcessorCount ) {
if (tempProp.multiProcessorCount >= prop->multiProcessorCount) {
matchedPropCount++;
}
}
if(prop->maxThreadsPerMultiProcessor != 0) {
if (prop->maxThreadsPerMultiProcessor != 0) {
inPropCount++;
if(tempProp.maxThreadsPerMultiProcessor >= prop->maxThreadsPerMultiProcessor ) {
if (tempProp.maxThreadsPerMultiProcessor >= prop->maxThreadsPerMultiProcessor) {
matchedPropCount++;
}
}
if(prop->memoryClockRate != 0) {
if (prop->memoryClockRate != 0) {
inPropCount++;
if(tempProp.memoryClockRate >= prop->memoryClockRate ) {
if (tempProp.memoryClockRate >= prop->memoryClockRate) {
matchedPropCount++;
}
}
if(inPropCount == matchedPropCount) {
if (inPropCount == matchedPropCount) {
*device = i;
}
#if 0
@@ -524,4 +528,3 @@ hipError_t hipChooseDevice( int* device, const hipDeviceProp_t* prop )
}
return ihipLogStatus(e);
}
+6 -9
View File
@@ -29,8 +29,7 @@ THE SOFTWARE.
// Error Handling
//---
hipError_t hipGetLastError()
{
hipError_t hipGetLastError() {
HIP_INIT_API();
// Return last error, but then reset the state:
@@ -39,26 +38,24 @@ hipError_t hipGetLastError()
return e;
}
hipError_t hipPeekAtLastError()
{
hipError_t hipPeekAtLastError() {
HIP_INIT_API();
// peek at last error, but don't reset it.
return ihipLogStatus(tls_lastHipError);
}
const char *hipGetErrorName(hipError_t hip_error)
{
const char* hipGetErrorName(hipError_t hip_error) {
HIP_INIT_API(hip_error);
return ihipErrorString(hip_error);
}
const char *hipGetErrorString(hipError_t hip_error)
{
const char* hipGetErrorString(hipError_t hip_error) {
HIP_INIT_API(hip_error);
// TODO - return a message explaining the error.
// TODO - This should be set up to return the same string reported in the the doxygen comments, somehow.
// TODO - This should be set up to return the same string reported in the the doxygen comments,
// somehow.
return hipGetErrorName(hip_error);
}
+61 -78
View File
@@ -30,74 +30,60 @@ THE SOFTWARE.
//---
ihipEvent_t::ihipEvent_t(unsigned flags)
: _criticalData(this)
{
_flags = flags;
};
ihipEvent_t::ihipEvent_t(unsigned flags) : _criticalData(this) { _flags = flags; };
// Attach to an existing completion future:
void ihipEvent_t::attachToCompletionFuture(const hc::completion_future *cf,
hipStream_t stream, ihipEventType_t eventType)
{
void ihipEvent_t::attachToCompletionFuture(const hc::completion_future* cf, hipStream_t stream,
ihipEventType_t eventType) {
LockedAccessor_EventCrit_t crit(_criticalData);
crit->_eventData.marker(*cf);
crit->_eventData._type = eventType;
crit->_eventData._type = eventType;
crit->_eventData._stream = stream;
crit->_eventData._state = hipEventStatusRecording;
crit->_eventData._state = hipEventStatusRecording;
}
std::pair<hipEventStatus_t, uint64_t>
ihipEvent_t::refreshEventStatus()
{
std::pair<hipEventStatus_t, uint64_t> ihipEvent_t::refreshEventStatus() {
auto ecd = locked_copyCrit();
if (ecd._state == hipEventStatusRecording) {
bool isReady1 = ecd._stream->locked_eventIsReady(this);
if (isReady1) {
LockedAccessor_EventCrit_t eCrit(_criticalData);
if ((eCrit->_eventData._type == hipEventTypeIndependent) ||
if ((eCrit->_eventData._type == hipEventTypeIndependent) ||
(eCrit->_eventData._type == hipEventTypeStopCommand)) {
eCrit->_eventData._timestamp = eCrit->_eventData.marker().get_end_tick();
eCrit->_eventData._timestamp = eCrit->_eventData.marker().get_end_tick();
} else if (eCrit->_eventData._type == hipEventTypeStartCommand) {
eCrit->_eventData._timestamp = eCrit->_eventData.marker().get_begin_tick();
eCrit->_eventData._timestamp = eCrit->_eventData.marker().get_begin_tick();
} else {
eCrit->_eventData._timestamp = 0;
assert(0); // TODO - move to debug assert
eCrit->_eventData._timestamp = 0;
assert(0); // TODO - move to debug assert
}
eCrit->_eventData._state = hipEventStatusComplete;
return std::pair<hipEventStatus_t, uint64_t> (eCrit->_eventData._state, eCrit->_eventData._timestamp);
return std::pair<hipEventStatus_t, uint64_t>(eCrit->_eventData._state,
eCrit->_eventData._timestamp);
}
}
}
// Not complete path here:
return std::pair<hipEventStatus_t, uint64_t> (ecd._state, ecd._timestamp);
return std::pair<hipEventStatus_t, uint64_t>(ecd._state, ecd._timestamp);
}
hipError_t ihipEventCreate(hipEvent_t* event, unsigned flags)
{
hipError_t ihipEventCreate(hipEvent_t* event, unsigned flags) {
hipError_t e = hipSuccess;
// TODO-IPC - support hipEventInterprocess.
unsigned supportedFlags = hipEventDefault
| hipEventBlockingSync
| hipEventDisableTiming
| hipEventReleaseToDevice
| hipEventReleaseToSystem
;
unsigned supportedFlags = hipEventDefault | hipEventBlockingSync | hipEventDisableTiming |
hipEventReleaseToDevice | hipEventReleaseToSystem;
const unsigned releaseFlags = (hipEventReleaseToDevice | hipEventReleaseToSystem);
const bool illegalFlags = (flags & ~supportedFlags) || // can't set any unsupported flags.
(flags & releaseFlags) == releaseFlags; // can't set both release flags
const bool illegalFlags =
(flags & ~supportedFlags) || // can't set any unsupported flags.
(flags & releaseFlags) == releaseFlags; // can't set both release flags
if (!illegalFlags) {
*event = new ihipEvent_t(flags);
@@ -108,40 +94,37 @@ hipError_t ihipEventCreate(hipEvent_t* event, unsigned flags)
return e;
}
hipError_t hipEventCreateWithFlags(hipEvent_t* event, unsigned flags)
{
hipError_t hipEventCreateWithFlags(hipEvent_t* event, unsigned flags) {
HIP_INIT_API(event, flags);
return ihipLogStatus(ihipEventCreate(event, flags));
}
hipError_t hipEventCreate(hipEvent_t* event)
{
hipError_t hipEventCreate(hipEvent_t* event) {
HIP_INIT_API(event);
return ihipLogStatus(ihipEventCreate(event, 0));
}
hipError_t hipEventRecord(hipEvent_t event, hipStream_t stream)
{
hipError_t hipEventRecord(hipEvent_t event, hipStream_t stream) {
HIP_INIT_SPECIAL_API(TRACE_SYNC, event, stream);
auto ecd = event->locked_copyCrit();
if (event && ecd._state != hipEventStatusUnitialized) {
if (event && ecd._state != hipEventStatusUnitialized) {
stream = ihipSyncAndResolveStream(stream);
if (HIP_SYNC_NULL_STREAM && stream->isDefaultStream()) {
// TODO-HIP_SYNC_NULL_STREAM : can remove this code when HIP_SYNC_NULL_STREAM = 0
//
// If default stream , then wait on all queues.
ihipCtx_t *ctx = ihipGetTlsDefaultCtx();
ihipCtx_t* ctx = ihipGetTlsDefaultCtx();
ctx->locked_syncDefaultStream(true, true);
{
LockedAccessor_EventCrit_t eCrit(event->criticalData());
eCrit->_eventData.marker(hc::completion_future()); // reset event
eCrit->_eventData.marker(hc::completion_future()); // reset event
eCrit->_eventData._stream = stream;
eCrit->_eventData._timestamp = hc::get_system_ticks();
eCrit->_eventData._state = hipEventStatusComplete;
@@ -149,7 +132,8 @@ hipError_t hipEventRecord(hipEvent_t event, hipStream_t stream)
return ihipLogStatus(hipSuccess);
} else {
// Record the event in the stream:
// Keep a copy outside the critical section so we lock stream first, then event - to avoid deadlock
// Keep a copy outside the critical section so we lock stream first, then event - to
// avoid deadlock
hc::completion_future cf = stream->locked_recordEvent(event);
{
@@ -157,7 +141,7 @@ hipError_t hipEventRecord(hipEvent_t event, hipStream_t stream)
eCrit->_eventData.marker(cf);
eCrit->_eventData._stream = stream;
eCrit->_eventData._timestamp = 0;
eCrit->_eventData._state = hipEventStatusRecording;
eCrit->_eventData._state = hipEventStatusRecording;
}
return ihipLogStatus(hipSuccess);
@@ -168,8 +152,7 @@ hipError_t hipEventRecord(hipEvent_t event, hipStream_t stream)
}
hipError_t hipEventDestroy(hipEvent_t event)
{
hipError_t hipEventDestroy(hipEvent_t event) {
HIP_INIT_API(event);
if (event) {
@@ -181,31 +164,31 @@ hipError_t hipEventDestroy(hipEvent_t event)
}
}
hipError_t hipEventSynchronize(hipEvent_t event)
{
hipError_t hipEventSynchronize(hipEvent_t event) {
HIP_INIT_SPECIAL_API(TRACE_SYNC, event);
if (!(event->_flags & hipEventReleaseToSystem)) {
tprintf(DB_WARN, "hipEventSynchronize on event without system-scope fence ; consider creating with hipEventReleaseToSystem\n");
tprintf(DB_WARN,
"hipEventSynchronize on event without system-scope fence ; consider creating with "
"hipEventReleaseToSystem\n");
}
auto ecd = event->locked_copyCrit();
if (event) {
if (ecd._state == hipEventStatusUnitialized) {
return ihipLogStatus(hipErrorInvalidResourceHandle);
} else if (ecd._state == hipEventStatusCreated ) {
} else if (ecd._state == hipEventStatusCreated) {
// Created but not actually recorded on any device:
return ihipLogStatus(hipSuccess);
} else if (HIP_SYNC_NULL_STREAM && (ecd._stream->isDefaultStream() )) {
auto *ctx = ihipGetTlsDefaultCtx();
} else if (HIP_SYNC_NULL_STREAM && (ecd._stream->isDefaultStream())) {
auto* ctx = ihipGetTlsDefaultCtx();
// TODO-HIP_SYNC_NULL_STREAM - can remove this code
ctx->locked_syncDefaultStream(true, true);
return ihipLogStatus(hipSuccess);
} else {
ecd._stream->locked_eventWaitComplete(
ecd.marker(),
(event->_flags & hipEventBlockingSync) ?
hc::hcWaitModeBlocked : hc::hcWaitModeActive);
ecd.marker(), (event->_flags & hipEventBlockingSync) ? hc::hcWaitModeBlocked
: hc::hcWaitModeActive);
return ihipLogStatus(hipSuccess);
}
@@ -214,8 +197,7 @@ hipError_t hipEventSynchronize(hipEvent_t event)
}
}
hipError_t hipEventElapsedTime(float *ms, hipEvent_t start, hipEvent_t stop)
{
hipError_t hipEventElapsedTime(float* ms, hipEvent_t start, hipEvent_t stop) {
HIP_INIT_API(ms, start, stop);
hipError_t status = hipSuccess;
@@ -225,15 +207,14 @@ hipError_t hipEventElapsedTime(float *ms, hipEvent_t start, hipEvent_t stop)
if ((start == nullptr) || (stop == nullptr)) {
status = hipErrorInvalidResourceHandle;
} else {
auto startEcd = start->locked_copyCrit();
auto stopEcd = stop->locked_copyCrit();
auto stopEcd = stop->locked_copyCrit();
if ((start->_flags & hipEventDisableTiming) ||
(startEcd._state == hipEventStatusUnitialized) || (startEcd._state == hipEventStatusCreated) ||
(stop->_flags & hipEventDisableTiming) ||
(stopEcd._state == hipEventStatusUnitialized) || (stopEcd._state == hipEventStatusCreated)) {
(startEcd._state == hipEventStatusUnitialized) ||
(startEcd._state == hipEventStatusCreated) || (stop->_flags & hipEventDisableTiming) ||
(stopEcd._state == hipEventStatusUnitialized) ||
(stopEcd._state == hipEventStatusCreated)) {
// Both events must be at least recorded else return hipErrorInvalidResourceHandle
status = hipErrorInvalidResourceHandle;
@@ -242,42 +223,44 @@ hipError_t hipEventElapsedTime(float *ms, hipEvent_t start, hipEvent_t stop)
// Refresh status, if still recording...
auto startStatus = start->refreshEventStatus(); // pair < state, timestamp >
auto stopStatus = stop->refreshEventStatus(); // pair < state, timestamp >
auto stopStatus = stop->refreshEventStatus(); // pair < state, timestamp >
if ((startStatus.first == hipEventStatusComplete) && (stopStatus.first == hipEventStatusComplete)) {
// Common case, we have good information for both events. 'second" is the timestamp:
if ((startStatus.first == hipEventStatusComplete) &&
(stopStatus.first == hipEventStatusComplete)) {
// Common case, we have good information for both events. 'second" is the
// timestamp:
int64_t tickDiff = (stopStatus.second - startStatus.second);
uint64_t freqHz;
hsa_system_get_info(HSA_SYSTEM_INFO_TIMESTAMP_FREQUENCY, &freqHz);
if (freqHz) {
*ms = ((double)(tickDiff) / (double)(freqHz)) * 1000.0f;
*ms = ((double)(tickDiff) / (double)(freqHz)) * 1000.0f;
status = hipSuccess;
} else {
* ms = 0.0f;
status = hipErrorInvalidValue;
}
*ms = 0.0f;
status = hipErrorInvalidValue;
}
} else if ((startStatus.first == hipEventStatusRecording) ||
(stopStatus.first == hipEventStatusRecording)) {
(stopStatus.first == hipEventStatusRecording)) {
status = hipErrorNotReady;
} else {
assert(0);
}
}
}
}
return ihipLogStatus(status);
}
hipError_t hipEventQuery(hipEvent_t event)
{
hipError_t hipEventQuery(hipEvent_t event) {
HIP_INIT_SPECIAL_API(TRACE_QUERY, event);
if (!(event->_flags & hipEventReleaseToSystem)) {
tprintf(DB_WARN, "hipEventQuery on event without system-scope fence ; consider creating with hipEventReleaseToSystem\n");
tprintf(DB_WARN,
"hipEventQuery on event without system-scope fence ; consider creating with "
"hipEventReleaseToSystem\n");
}
auto ecd = event->locked_copyCrit();
+206 -360
View File
@@ -20,507 +20,355 @@ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.
*/
#include"hip/hcc_detail/hip_fp16.h"
#include "hip/hcc_detail/hip_fp16.h"
struct hipHalfHolder{
union {
__half h;
unsigned short s;
};
struct hipHalfHolder {
union {
__half h;
unsigned short s;
};
};
#define HINF 65504
__device__ static struct hipHalfHolder __hInfValue = {HINF};
__device__ __half __hadd(__half a, __half b) {
return a + b;
}
__device__ __half __hadd(__half a, __half b) { return a + b; }
__device__ __half __hadd_sat(__half a, __half b) {
return a + b;
}
__device__ __half __hadd_sat(__half a, __half b) { return a + b; }
__device__ __half __hfma(__half a, __half b, __half c) {
return a * b + c;
}
__device__ __half __hfma(__half a, __half b, __half c) { return a * b + c; }
__device__ __half __hfma_sat(__half a, __half b, __half c) {
return a * b + c;
}
__device__ __half __hfma_sat(__half a, __half b, __half c) { return a * b + c; }
__device__ __half __hmul(__half a, __half b) {
return a * b;
}
__device__ __half __hmul(__half a, __half b) { return a * b; }
__device__ __half __hmul_sat(__half a, __half b) {
return a * b;
}
__device__ __half __hmul_sat(__half a, __half b) { return a * b; }
__device__ __half __hneg(__half a) {
return -a;
}
__device__ __half __hneg(__half a) { return -a; }
__device__ __half __hsub(__half a, __half b) {
return a - b;
}
__device__ __half __hsub(__half a, __half b) { return a - b; }
__device__ __half __hsub_sat(__half a, __half b) {
return a - b;
}
__device__ __half __hsub_sat(__half a, __half b) { return a - b; }
__device__ __half hdiv(__half a, __half b) {
return a / b;
}
__device__ __half hdiv(__half a, __half b) { return a / b; }
/*
Half comparision Functions
*/
__device__ bool __heq(__half a, __half b) {
return a == b ? true : false;
}
__device__ bool __heq(__half a, __half b) { return a == b ? true : false; }
__device__ bool __hge(__half a, __half b) {
return a >= b ? true : false;
}
__device__ bool __hge(__half a, __half b) { return a >= b ? true : false; }
__device__ bool __hgt(__half a, __half b) {
return a > b ? true : false;
}
__device__ bool __hgt(__half a, __half b) { return a > b ? true : false; }
__device__ bool __hisinf(__half a) {
return a == HINF ? true : false;
}
__device__ bool __hisinf(__half a) { return a == HINF ? true : false; }
__device__ bool __hisnan(__half a) {
return a > HINF ? true : false;
}
__device__ bool __hisnan(__half a) { return a > HINF ? true : false; }
__device__ bool __hle(__half a, __half b) {
return a <= b ? true : false;
}
__device__ bool __hle(__half a, __half b) { return a <= b ? true : false; }
__device__ bool __hlt(__half a, __half b) {
return a < b ? true : false;
}
__device__ bool __hlt(__half a, __half b) { return a < b ? true : false; }
__device__ bool __hne(__half a, __half b) {
return a != b ? true : false;
}
__device__ bool __hne(__half a, __half b) { return a != b ? true : false; }
/*
Half2 Comparision Functions
*/
__device__ bool __hbeq2(__half2 a, __half2 b) {
return (a.x == b.x ? true : false) && (a.y == b.y ? true : false);
__device__ bool __hbeq2(__half2 a, __half2 b) {
return (a.x == b.x ? true : false) && (a.y == b.y ? true : false);
}
__device__ bool __hbge2(__half2 a, __half2 b) {
return (a.x >= b.x ? true : false) && (a.y >= b.y ? true : false);
__device__ bool __hbge2(__half2 a, __half2 b) {
return (a.x >= b.x ? true : false) && (a.y >= b.y ? true : false);
}
__device__ bool __hbgt2(__half2 a, __half2 b) {
return (a.x > b.x ? true : false) && (a.y > b.y ? true : false);
__device__ bool __hbgt2(__half2 a, __half2 b) {
return (a.x > b.x ? true : false) && (a.y > b.y ? true : false);
}
__device__ bool __hble2(__half2 a, __half2 b) {
return (a.x <= b.x ? true : false) && (a.y <= b.y ? true : false);
__device__ bool __hble2(__half2 a, __half2 b) {
return (a.x <= b.x ? true : false) && (a.y <= b.y ? true : false);
}
__device__ bool __hblt2(__half2 a, __half2 b) {
return (a.x < b.x ? true : false) && (a.y < b.y ? true : false);
__device__ bool __hblt2(__half2 a, __half2 b) {
return (a.x < b.x ? true : false) && (a.y < b.y ? true : false);
}
__device__ bool __hbne2(__half2 a, __half2 b) {
return (a.x != b.x ? true : false) && (a.y != b.y ? true : false);
__device__ bool __hbne2(__half2 a, __half2 b) {
return (a.x != b.x ? true : false) && (a.y != b.y ? true : false);
}
__device__ __half2 __heq2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x == b.x) ? (__half)1 : (__half)0;
c.y = (a.y == b.y) ? (__half)1 : (__half)0;
return c;
__device__ __half2 __heq2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x == b.x) ? (__half)1 : (__half)0;
c.y = (a.y == b.y) ? (__half)1 : (__half)0;
return c;
}
__device__ __half2 __hge2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x >= b.x) ? (__half)1 : (__half)0;
c.y = (a.y >= b.y) ? (__half)1 : (__half)0;
return c;
__device__ __half2 __hge2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x >= b.x) ? (__half)1 : (__half)0;
c.y = (a.y >= b.y) ? (__half)1 : (__half)0;
return c;
}
__device__ __half2 __hgt2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x > b.x) ? (__half)1 : (__half)0;
c.y = (a.y > b.y) ? (__half)1 : (__half)0;
return c;
__device__ __half2 __hgt2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x > b.x) ? (__half)1 : (__half)0;
c.y = (a.y > b.y) ? (__half)1 : (__half)0;
return c;
}
__device__ __half2 __hisnan2(__half2 a) {
__half2 c;
c.x = (a.x > HINF) ? (__half)1 : (__half)0;
c.y = (a.y > HINF) ? (__half)1 : (__half)0;
return c;
__device__ __half2 __hisnan2(__half2 a) {
__half2 c;
c.x = (a.x > HINF) ? (__half)1 : (__half)0;
c.y = (a.y > HINF) ? (__half)1 : (__half)0;
return c;
}
__device__ __half2 __hle2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x <= b.x) ? (__half)1 : (__half)0;
c.y = (a.y <= b.y) ? (__half)1 : (__half)0;
return c;
__device__ __half2 __hle2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x <= b.x) ? (__half)1 : (__half)0;
c.y = (a.y <= b.y) ? (__half)1 : (__half)0;
return c;
}
__device__ __half2 __hlt2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x < b.x) ? (__half)1 : (__half)0;
c.y = (a.y < b.y) ? (__half)1 : (__half)0;
return c;
__device__ __half2 __hlt2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x < b.x) ? (__half)1 : (__half)0;
c.y = (a.y < b.y) ? (__half)1 : (__half)0;
return c;
}
__device__ __half2 __hne2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x != b.x) ? (__half)1 : (__half)0;
c.y = (a.y != b.y) ? (__half)1 : (__half)0;
return c;
__device__ __half2 __hne2(__half2 a, __half2 b) {
__half2 c;
c.x = (a.x != b.x) ? (__half)1 : (__half)0;
c.y = (a.y != b.y) ? (__half)1 : (__half)0;
return c;
}
/*
Conversion instructions
*/
__device__ __half2 __float22half2_rn(const float2 a) {
__half2 b;
b.x = (__half)a.x;
b.y = (__half)a.y;
return b;
__device__ __half2 __float22half2_rn(const float2 a) {
__half2 b;
b.x = (__half)a.x;
b.y = (__half)a.y;
return b;
}
__device__ __half __float2half(const float a) {
return (__half)a;
}
__device__ __half __float2half(const float a) { return (__half)a; }
__device__ __half2 __float2half2_rn(const float a) {
__half2 b;
b.x = (__half)a;
b.y = (__half)a;
return b;
__device__ __half2 __float2half2_rn(const float a) {
__half2 b;
b.x = (__half)a;
b.y = (__half)a;
return b;
}
__device__ __half __float2half_rd(const float a) {
return (__half)a;
}
__device__ __half __float2half_rd(const float a) { return (__half)a; }
__device__ __half __float2half_rn(const float a) {
return (__half)a;
}
__device__ __half __float2half_rn(const float a) { return (__half)a; }
__device__ __half __float2half_ru(const float a) {
return (__half)a;
}
__device__ __half __float2half_ru(const float a) { return (__half)a; }
__device__ __half __float2half_rz(const float a) {
return (__half)a;
}
__device__ __half __float2half_rz(const float a) { return (__half)a; }
__device__ __half2 __floats2half2_rn(const float a, const float b) {
__half2 c;
c.x = (__half)a;
c.y = (__half)b;
return c;
__device__ __half2 __floats2half2_rn(const float a, const float b) {
__half2 c;
c.x = (__half)a;
c.y = (__half)b;
return c;
}
__device__ float2 __half22float2(const __half2 a) {
float2 b;
b.x = (float)a.x;
b.y = (float)a.y;
return b;
__device__ float2 __half22float2(const __half2 a) {
float2 b;
b.x = (float)a.x;
b.y = (float)a.y;
return b;
}
__device__ float __half2float(const __half a) {
return (float)a;
}
__device__ float __half2float(const __half a) { return (float)a; }
__device__ __half2 half2half2(const __half a) {
__half2 b;
b.x = a;
b.y = a;
return b;
__device__ __half2 half2half2(const __half a) {
__half2 b;
b.x = a;
b.y = a;
return b;
}
__device__ int __half2int_rd(__half h) {
return (int)h;
}
__device__ int __half2int_rd(__half h) { return (int)h; }
__device__ int __half2int_rn(__half h) {
return (int)h;
}
__device__ int __half2int_rn(__half h) { return (int)h; }
__device__ int __half2int_ru(__half h) {
return (int)h;
}
__device__ int __half2int_ru(__half h) { return (int)h; }
__device__ int __half2int_rz(__half h) {
return (int)h;
}
__device__ int __half2int_rz(__half h) { return (int)h; }
__device__ long long int __half2ll_rd(__half h) {
return (long long int)h;
}
__device__ long long int __half2ll_rd(__half h) { return (long long int)h; }
__device__ long long int __half2ll_rn(__half h) {
return (long long int)h;
}
__device__ long long int __half2ll_rn(__half h) { return (long long int)h; }
__device__ long long int __half2ll_ru(__half h) {
return (long long int)h;
}
__device__ long long int __half2ll_ru(__half h) { return (long long int)h; }
__device__ long long int __half2ll_rz(__half h) {
return (long long int)h;
}
__device__ long long int __half2ll_rz(__half h) { return (long long int)h; }
__device__ short __half2short_rd(__half h) {
return (short)h;
}
__device__ short __half2short_rd(__half h) { return (short)h; }
__device__ short __half2short_rn(__half h) {
return (short)h;
}
__device__ short __half2short_rn(__half h) { return (short)h; }
__device__ short __half2short_ru(__half h) {
return (short)h;
}
__device__ short __half2short_ru(__half h) { return (short)h; }
__device__ short __half2short_rz(__half h) {
return (short)h;
}
__device__ short __half2short_rz(__half h) { return (short)h; }
__device__ unsigned int __half2uint_rd(__half h) {
return (unsigned int)h;
}
__device__ unsigned int __half2uint_rd(__half h) { return (unsigned int)h; }
__device__ unsigned int __half2uint_rn(__half h) {
return (unsigned int)h;
}
__device__ unsigned int __half2uint_rn(__half h) { return (unsigned int)h; }
__device__ unsigned int __half2uint_ru(__half h) {
return (unsigned int)h;
}
__device__ unsigned int __half2uint_ru(__half h) { return (unsigned int)h; }
__device__ unsigned int __half2uint_rz(__half h) {
return (unsigned int)h;
}
__device__ unsigned int __half2uint_rz(__half h) { return (unsigned int)h; }
__device__ unsigned long long int __half2ull_rd(__half h) {
return (unsigned long long)h;
}
__device__ unsigned long long int __half2ull_rd(__half h) { return (unsigned long long)h; }
__device__ unsigned long long int __half2ull_rn(__half h) {
return (unsigned long long)h;
}
__device__ unsigned long long int __half2ull_rn(__half h) { return (unsigned long long)h; }
__device__ unsigned long long int __half2ull_ru(__half h) {
return (unsigned long long)h;
}
__device__ unsigned long long int __half2ull_ru(__half h) { return (unsigned long long)h; }
__device__ unsigned long long int __half2ull_rz(__half h) {
return (unsigned long long)h;
}
__device__ unsigned long long int __half2ull_rz(__half h) { return (unsigned long long)h; }
__device__ unsigned short int __half2ushort_rd(__half h) {
return (unsigned short int)h;
}
__device__ unsigned short int __half2ushort_rd(__half h) { return (unsigned short int)h; }
__device__ unsigned short int __half2ushort_rn(__half h) {
return (unsigned short int)h;
}
__device__ unsigned short int __half2ushort_rn(__half h) { return (unsigned short int)h; }
__device__ unsigned short int __half2ushort_ru(__half h) {
return (unsigned short int)h;
}
__device__ unsigned short int __half2ushort_ru(__half h) { return (unsigned short int)h; }
__device__ unsigned short int __half2ushort_rz(__half h) {
return (unsigned short int)h;
}
__device__ unsigned short int __half2ushort_rz(__half h) { return (unsigned short int)h; }
__device__ short int __half_as_short(const __half h) {
hipHalfHolder hH;
hH.h = h;
return (short)hH.s;
__device__ short int __half_as_short(const __half h) {
hipHalfHolder hH;
hH.h = h;
return (short)hH.s;
}
__device__ unsigned short int __half_as_ushort(const __half h) {
hipHalfHolder hH;
hH.h = h;
return hH.s;
__device__ unsigned short int __half_as_ushort(const __half h) {
hipHalfHolder hH;
hH.h = h;
return hH.s;
}
__device__ __half2 __halves2half2(const __half a, const __half b) {
__half2 c;
c.x = a;
c.y = b;
return c;
__device__ __half2 __halves2half2(const __half a, const __half b) {
__half2 c;
c.x = a;
c.y = b;
return c;
}
__device__ float __high2float(const __half2 a) {
return (float)a.y;
}
__device__ float __high2float(const __half2 a) { return (float)a.y; }
__device__ __half __high2half(const __half2 a) {
return a.y;
}
__device__ __half __high2half(const __half2 a) { return a.y; }
__device__ __half2 __high2half2(const __half2 a) {
__half2 b;
b.x = a.y;
b.y = a.y;
return b;
__device__ __half2 __high2half2(const __half2 a) {
__half2 b;
b.x = a.y;
b.y = a.y;
return b;
}
__device__ __half2 __highs2half2(const __half2 a, const __half2 b) {
__half2 c;
c.x = a.y;
c.y = b.y;
return c;
__device__ __half2 __highs2half2(const __half2 a, const __half2 b) {
__half2 c;
c.x = a.y;
c.y = b.y;
return c;
}
__device__ __half __int2half_rd(int i) {
return (__half)i;
}
__device__ __half __int2half_rd(int i) { return (__half)i; }
__device__ __half __int2half_rn(int i) {
return (__half)i;
}
__device__ __half __int2half_rn(int i) { return (__half)i; }
__device__ __half __int2half_ru(int i) {
return (__half)i;
}
__device__ __half __int2half_ru(int i) { return (__half)i; }
__device__ __half __int2half_rz(int i) {
return (__half)i;
}
__device__ __half __int2half_rz(int i) { return (__half)i; }
__device__ __half __ll2half_rd(long long int i){
return (__half)i;
}
__device__ __half __ll2half_rd(long long int i) { return (__half)i; }
__device__ __half __ll2half_rn(long long int i){
return (__half)i;
}
__device__ __half __ll2half_rn(long long int i) { return (__half)i; }
__device__ __half __ll2half_ru(long long int i){
return (__half)i;
}
__device__ __half __ll2half_ru(long long int i) { return (__half)i; }
__device__ __half __ll2half_rz(long long int i){
return (__half)i;
}
__device__ __half __ll2half_rz(long long int i) { return (__half)i; }
__device__ float __low2float(const __half2 a) {
return (float)a.x;
}
__device__ float __low2float(const __half2 a) { return (float)a.x; }
__device__ __half __low2half(const __half2 a) {
return a.x;
}
__device__ __half __low2half(const __half2 a) { return a.x; }
__device__ __half2 __low2half2(const __half2 a, const __half2 b) {
__half2 c;
c.x = a.x;
c.y = b.x;
return c;
__device__ __half2 __low2half2(const __half2 a, const __half2 b) {
__half2 c;
c.x = a.x;
c.y = b.x;
return c;
}
__device__ __half2 __low2half2(const __half2 a) {
__half2 b;
b.x = a.x;
b.y = a.x;
return b;
__device__ __half2 __low2half2(const __half2 a) {
__half2 b;
b.x = a.x;
b.y = a.x;
return b;
}
__device__ __half2 __lowhigh2highlow(const __half2 a) {
__half2 b;
b.x = a.y;
b.y = a.x;
return b;
__device__ __half2 __lowhigh2highlow(const __half2 a) {
__half2 b;
b.x = a.y;
b.y = a.x;
return b;
}
__device__ __half2 __lows2half2(const __half2 a, const __half2 b) {
__half2 c;
c.x = a.x;
c.y = b.x;
return c;
__device__ __half2 __lows2half2(const __half2 a, const __half2 b) {
__half2 c;
c.x = a.x;
c.y = b.x;
return c;
}
__device__ __half __short2half_rd(short int i) {
return (__half)i;
}
__device__ __half __short2half_rd(short int i) { return (__half)i; }
__device__ __half __short2half_rn(short int i) {
return (__half)i;
}
__device__ __half __short2half_rn(short int i) { return (__half)i; }
__device__ __half __short2half_ru(short int i) {
return (__half)i;
}
__device__ __half __short2half_ru(short int i) { return (__half)i; }
__device__ __half __short2half_rz(short int i) {
return (__half)i;
}
__device__ __half __short2half_rz(short int i) { return (__half)i; }
__device__ __half __uint2half_rd(unsigned int i) {
return (__half)i;
}
__device__ __half __uint2half_rd(unsigned int i) { return (__half)i; }
__device__ __half __uint2half_rn(unsigned int i) {
return (__half)i;
}
__device__ __half __uint2half_rn(unsigned int i) { return (__half)i; }
__device__ __half __uint2half_ru(unsigned int i) {
return (__half)i;
}
__device__ __half __uint2half_ru(unsigned int i) { return (__half)i; }
__device__ __half __uint2half_rz(unsigned int i) {
return (__half)i;
}
__device__ __half __uint2half_rz(unsigned int i) { return (__half)i; }
__device__ __half __ull2half_rd(unsigned long long int i) {
return (__half)i;
}
__device__ __half __ull2half_rd(unsigned long long int i) { return (__half)i; }
__device__ __half __ull2half_rn(unsigned long long int i) {
return (__half)i;
}
__device__ __half __ull2half_rn(unsigned long long int i) { return (__half)i; }
__device__ __half __ull2half_ru(unsigned long long int i) {
return (__half)i;
}
__device__ __half __ull2half_ru(unsigned long long int i) { return (__half)i; }
__device__ __half __ull2half_rz(unsigned long long int i) {
return (__half)i;
}
__device__ __half __ull2half_rz(unsigned long long int i) { return (__half)i; }
__device__ __half __ushort2half_rd(unsigned short int i) {
return (__half)i;
}
__device__ __half __ushort2half_rd(unsigned short int i) { return (__half)i; }
__device__ __half __ushort2half_rn(unsigned short int i) {
return (__half)i;
}
__device__ __half __ushort2half_rn(unsigned short int i) { return (__half)i; }
__device__ __half __ushort2half_ru(unsigned short int i) {
return (__half)i;
}
__device__ __half __ushort2half_ru(unsigned short int i) { return (__half)i; }
__device__ __half __ushort2half_rz(unsigned short int i) {
return (__half)i;
}
__device__ __half __ushort2half_rz(unsigned short int i) { return (__half)i; }
__device__ __half __ushort_as_half(const unsigned short int i) {
hipHalfHolder hH;
hH.s = i;
return hH.h;
__device__ __half __ushort_as_half(const unsigned short int i) {
hipHalfHolder hH;
hH.s = i;
return hH.h;
}
@@ -535,11 +383,9 @@ static const __half __half_value_zero_float = {0x0};
static const unsigned __half_pos_inf = 0x7C00;
static const unsigned __half_neg_inf = 0xFC00;
typedef struct{
union{
float f;
unsigned u;
};
typedef struct {
union {
float f;
unsigned u;
};
} struct_float;
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+29 -116
View File
@@ -23,148 +23,61 @@ THE SOFTWARE.
#include "hip/hcc_detail/hip_ldg.h"
#include "hip/hcc_detail/hip_vector_types.h"
__device__ char __ldg(const char* ptr)
{
return *ptr;
}
__device__ char __ldg(const char* ptr) { return *ptr; }
__device__ char2 __ldg(const char2* ptr)
{
return *ptr;
}
__device__ char2 __ldg(const char2* ptr) { return *ptr; }
__device__ char4 __ldg(const char4* ptr)
{
return *ptr;
}
__device__ char4 __ldg(const char4* ptr) { return *ptr; }
__device__ signed char __ldg(const signed char* ptr)
{
return ptr[0];
}
__device__ signed char __ldg(const signed char* ptr) { return ptr[0]; }
__device__ unsigned char __ldg(const unsigned char* ptr)
{
return ptr[0];
}
__device__ unsigned char __ldg(const unsigned char* ptr) { return ptr[0]; }
__device__ short __ldg(const short* ptr)
{
return ptr[0];
}
__device__ short __ldg(const short* ptr) { return ptr[0]; }
__device__ short2 __ldg(const short2* ptr)
{
return ptr[0];
}
__device__ short2 __ldg(const short2* ptr) { return ptr[0]; }
__device__ short4 __ldg(const short4* ptr)
{
return ptr[0];
}
__device__ short4 __ldg(const short4* ptr) { return ptr[0]; }
__device__ unsigned short __ldg(const unsigned short* ptr)
{
return ptr[0];
}
__device__ unsigned short __ldg(const unsigned short* ptr) { return ptr[0]; }
__device__ int __ldg(const int* ptr)
{
return ptr[0];
}
__device__ int __ldg(const int* ptr) { return ptr[0]; }
__device__ int2 __ldg(const int2* ptr)
{
return ptr[0];
}
__device__ int2 __ldg(const int2* ptr) { return ptr[0]; }
__device__ int4 __ldg(const int4* ptr)
{
return ptr[0];
}
__device__ int4 __ldg(const int4* ptr) { return ptr[0]; }
__device__ unsigned int __ldg(const unsigned int* ptr)
{
return ptr[0];
}
__device__ unsigned int __ldg(const unsigned int* ptr) { return ptr[0]; }
__device__ long __ldg(const long* ptr)
{
return ptr[0];
}
__device__ long __ldg(const long* ptr) { return ptr[0]; }
__device__ unsigned long __ldg(const unsigned long* ptr)
{
return ptr[0];
}
__device__ unsigned long __ldg(const unsigned long* ptr) { return ptr[0]; }
__device__ long long __ldg(const long long* ptr)
{
return ptr[0];
}
__device__ long long __ldg(const long long* ptr) { return ptr[0]; }
__device__ longlong2 __ldg(const longlong2* ptr)
{
return ptr[0];
}
__device__ longlong2 __ldg(const longlong2* ptr) { return ptr[0]; }
__device__ unsigned long long __ldg(const unsigned long long* ptr)
{
return ptr[0];
}
__device__ unsigned long long __ldg(const unsigned long long* ptr) { return ptr[0]; }
__device__ uchar2 __ldg(const uchar2* ptr)
{
return ptr[0];
}
__device__ uchar2 __ldg(const uchar2* ptr) { return ptr[0]; }
__device__ uchar4 __ldg(const uchar4* ptr)
{
return ptr[0];
}
__device__ uchar4 __ldg(const uchar4* ptr) { return ptr[0]; }
__device__ ushort2 __ldg(const ushort2* ptr)
{
return ptr[0];
}
__device__ ushort2 __ldg(const ushort2* ptr) { return ptr[0]; }
__device__ uint2 __ldg(const uint2* ptr)
{
return ptr[0];
}
__device__ uint2 __ldg(const uint2* ptr) { return ptr[0]; }
__device__ uint4 __ldg(const uint4* ptr)
{
return ptr[0];
}
__device__ uint4 __ldg(const uint4* ptr) { return ptr[0]; }
__device__ ulonglong2 __ldg(const ulonglong2* ptr)
{
return ptr[0];
}
__device__ ulonglong2 __ldg(const ulonglong2* ptr) { return ptr[0]; }
__device__ float __ldg(const float* ptr)
{
return ptr[0];
}
__device__ float __ldg(const float* ptr) { return ptr[0]; }
__device__ float2 __ldg(const float2* ptr)
{
return ptr[0];
}
__device__ float2 __ldg(const float2* ptr) { return ptr[0]; }
__device__ float4 __ldg(const float4* ptr)
{
return ptr[0];
}
__device__ float4 __ldg(const float4* ptr) { return ptr[0]; }
__device__ double __ldg(const double* ptr)
{
return ptr[0];
}
__device__ double __ldg(const double* ptr) { return ptr[0]; }
__device__ double2 __ldg(const double2* ptr)
{
return ptr[0];
}
__device__ double2 __ldg(const double2* ptr) { return ptr[0]; }
File diff suppressed because it is too large Load Diff
+240 -309
View File
@@ -47,57 +47,55 @@ THE SOFTWARE.
#include <utility>
#include <vector>
#include "../include/hip/hcc_detail/code_object_bundle.hpp"
//TODO Use Pool APIs from HCC to get memory regions.
// TODO Use Pool APIs from HCC to get memory regions.
using namespace ELFIO;
using namespace hip_impl;
using namespace std;
inline uint64_t alignTo(uint64_t Value, uint64_t Align, uint64_t Skew = 0) {
assert(Align != 0u && "Align can't be 0.");
Skew %= Align;
return (Value + Align - 1 - Skew) / Align * Align + Skew;
assert(Align != 0u && "Align can't be 0.");
Skew %= Align;
return (Value + Align - 1 - Skew) / Align * Align + Skew;
}
struct ihipKernArgInfo{
vector<uint32_t> Size;
vector<uint32_t> Align;
vector<string> ArgType;
vector<string> ArgName;
uint32_t totalSize;
struct ihipKernArgInfo {
vector<uint32_t> Size;
vector<uint32_t> Align;
vector<string> ArgType;
vector<string> ArgName;
uint32_t totalSize;
};
map<string, ihipKernArgInfo> kernelArguments;
struct ihipModuleSymbol_t{
uint64_t _object; // The kernel object.
struct ihipModuleSymbol_t {
uint64_t _object; // The kernel object.
uint32_t _groupSegmentSize;
uint32_t _privateSegmentSize;
string _name; // TODO - review for performance cost. Name is just used for debug.
string _name; // TODO - review for performance cost. Name is just used for debug.
};
template <>
string ToString(hipFunction_t v)
{
string ToString(hipFunction_t v) {
std::ostringstream ss;
ss << "0x" << std::hex << v->_object;
return ss.str();
};
#define CHECK_HSA(hsaStatus, hipStatus) \
if (hsaStatus != HSA_STATUS_SUCCESS) {\
return hipStatus;\
}
#define CHECK_HSA(hsaStatus, hipStatus) \
if (hsaStatus != HSA_STATUS_SUCCESS) { \
return hipStatus; \
}
#define CHECKLOG_HSA(hsaStatus, hipStatus) \
if (hsaStatus != HSA_STATUS_SUCCESS) {\
return ihipLogStatus(hipStatus);\
}
#define CHECKLOG_HSA(hsaStatus, hipStatus) \
if (hsaStatus != HSA_STATUS_SUCCESS) { \
return ihipLogStatus(hipStatus); \
}
hipError_t hipModuleUnload(hipModule_t hmod)
{
hipError_t hipModuleUnload(hipModule_t hmod) {
HIP_INIT_API(hmod);
// TODO - improve this synchronization so it is thread-safe.
@@ -105,74 +103,75 @@ hipError_t hipModuleUnload(hipModule_t hmod)
// thread from launching new kernels before we finish this operation.
ihipSynchronize();
delete hmod; // The ihipModule_t dtor will clean everything up.
delete hmod; // The ihipModule_t dtor will clean everything up.
hmod = nullptr;
return ihipLogStatus(hipSuccess);
}
hipError_t ihipModuleLaunchKernel(hipFunction_t f,
uint32_t globalWorkSizeX, uint32_t globalWorkSizeY, uint32_t globalWorkSizeZ,
uint32_t localWorkSizeX, uint32_t localWorkSizeY, uint32_t localWorkSizeZ,
size_t sharedMemBytes, hipStream_t hStream,
void **kernelParams, void **extra,
hipEvent_t startEvent, hipEvent_t stopEvent)
{
hipError_t ihipModuleLaunchKernel(hipFunction_t f, uint32_t globalWorkSizeX,
uint32_t globalWorkSizeY, uint32_t globalWorkSizeZ,
uint32_t localWorkSizeX, uint32_t localWorkSizeY,
uint32_t localWorkSizeZ, size_t sharedMemBytes,
hipStream_t hStream, void** kernelParams, void** extra,
hipEvent_t startEvent, hipEvent_t stopEvent) {
auto ctx = ihipGetTlsDefaultCtx();
hipError_t ret = hipSuccess;
if(ctx == nullptr){
if (ctx == nullptr) {
ret = hipErrorInvalidDevice;
}else{
} else {
int deviceId = ctx->getDevice()->_deviceId;
ihipDevice_t *currentDevice = ihipGetDevice(deviceId);
ihipDevice_t* currentDevice = ihipGetDevice(deviceId);
hsa_agent_t gpuAgent = (hsa_agent_t)currentDevice->_hsaAgent;
void *config[5] = {0};
void* config[5] = {0};
size_t kernArgSize;
if(kernelParams != NULL){
std::string name = f->_name;
struct ihipKernArgInfo pl = kernelArguments[name];
char* argBuf = (char*)malloc(pl.totalSize);
memset(argBuf, 0, pl.totalSize);
int index = 0;
for(int i=0;i<pl.Size.size();i++){
memcpy(argBuf + index, kernelParams[i], pl.Size[i]);
index += pl.Align[i];
}
config[1] = (void*)argBuf;
kernArgSize = pl.totalSize;
} else if(extra != NULL){
memcpy(config, extra, sizeof(size_t)*5);
if(config[0] == HIP_LAUNCH_PARAM_BUFFER_POINTER && config[2] == HIP_LAUNCH_PARAM_BUFFER_SIZE && config[4] == HIP_LAUNCH_PARAM_END){
if (kernelParams != NULL) {
std::string name = f->_name;
struct ihipKernArgInfo pl = kernelArguments[name];
char* argBuf = (char*)malloc(pl.totalSize);
memset(argBuf, 0, pl.totalSize);
int index = 0;
for (int i = 0; i < pl.Size.size(); i++) {
memcpy(argBuf + index, kernelParams[i], pl.Size[i]);
index += pl.Align[i];
}
config[1] = (void*)argBuf;
kernArgSize = pl.totalSize;
} else if (extra != NULL) {
memcpy(config, extra, sizeof(size_t) * 5);
if (config[0] == HIP_LAUNCH_PARAM_BUFFER_POINTER &&
config[2] == HIP_LAUNCH_PARAM_BUFFER_SIZE && config[4] == HIP_LAUNCH_PARAM_END) {
kernArgSize = *(size_t*)(config[3]);
} else {
return hipErrorNotInitialized;
}
}else{
} else {
return hipErrorInvalidValue;
}
/*
Kernel argument preparation.
*/
grid_launch_parm lp;
lp.dynamic_group_mem_bytes = sharedMemBytes; // TODO - this should be part of preLaunchKernel.
hStream = ihipPreLaunchKernel(hStream, dim3(globalWorkSizeX, globalWorkSizeY, globalWorkSizeZ), dim3(localWorkSizeX, localWorkSizeY, localWorkSizeZ), &lp, f->_name.c_str());
lp.dynamic_group_mem_bytes =
sharedMemBytes; // TODO - this should be part of preLaunchKernel.
hStream = ihipPreLaunchKernel(
hStream, dim3(globalWorkSizeX, globalWorkSizeY, globalWorkSizeZ),
dim3(localWorkSizeX, localWorkSizeY, localWorkSizeZ), &lp, f->_name.c_str());
hsa_kernel_dispatch_packet_t aql;
memset(&aql, 0, sizeof(aql));
//aql.completion_signal._handle = 0;
//aql.kernarg_address = 0;
// aql.completion_signal._handle = 0;
// aql.kernarg_address = 0;
aql.workgroup_size_x = localWorkSizeX;
aql.workgroup_size_y = localWorkSizeY;
@@ -184,8 +183,9 @@ hipError_t ihipModuleLaunchKernel(hipFunction_t f,
aql.private_segment_size = f->_privateSegmentSize;
aql.kernel_object = f->_object;
aql.setup = 3 << HSA_KERNEL_DISPATCH_PACKET_SETUP_DIMENSIONS;
aql.header = (HSA_PACKET_TYPE_KERNEL_DISPATCH << HSA_PACKET_HEADER_TYPE) |
(1 << HSA_PACKET_HEADER_BARRIER); // TODO - honor queue setting for execute_in_order
aql.header =
(HSA_PACKET_TYPE_KERNEL_DISPATCH << HSA_PACKET_HEADER_TYPE) |
(1 << HSA_PACKET_HEADER_BARRIER); // TODO - honor queue setting for execute_in_order
if (HCC_OPT_FLUSH) {
aql.header |= (HSA_FENCE_SCOPE_AGENT << HSA_PACKET_HEADER_ACQUIRE_FENCE_SCOPE) |
@@ -199,24 +199,24 @@ hipError_t ihipModuleLaunchKernel(hipFunction_t f,
hc::completion_future cf;
lp.av->dispatch_hsa_kernel(&aql, config[1] /* kernarg*/, kernArgSize,
(startEvent || stopEvent) ? &cf : nullptr
(startEvent || stopEvent) ? &cf : nullptr
#if (__hcc_workweek__ > 17312)
, f->_name.c_str()
,
f->_name.c_str()
#endif
);
);
if (startEvent) {
startEvent->attachToCompletionFuture(&cf, hStream, hipEventTypeStartCommand);
}
if (stopEvent) {
stopEvent->attachToCompletionFuture (&cf, hStream, hipEventTypeStopCommand);
stopEvent->attachToCompletionFuture(&cf, hStream, hipEventTypeStopCommand);
}
if(kernelParams != NULL){
free(config[1]);
if (kernelParams != NULL) {
free(config[1]);
}
ihipPostLaunchKernel(f->_name.c_str(), hStream, lp);
}
@@ -224,266 +224,211 @@ hipError_t ihipModuleLaunchKernel(hipFunction_t f,
return ret;
}
hipError_t hipModuleLaunchKernel(hipFunction_t f,
uint32_t gridDimX, uint32_t gridDimY, uint32_t gridDimZ,
uint32_t blockDimX, uint32_t blockDimY, uint32_t blockDimZ,
uint32_t sharedMemBytes, hipStream_t hStream,
void **kernelParams, void **extra)
{
HIP_INIT_API(f, gridDimX, gridDimY, gridDimZ,
blockDimX, blockDimY, blockDimZ,
sharedMemBytes, hStream,
kernelParams, extra);
return ihipLogStatus(ihipModuleLaunchKernel(f,
blockDimX * gridDimX, blockDimY * gridDimY, gridDimZ * blockDimZ,
blockDimX, blockDimY, blockDimZ,
sharedMemBytes, hStream, kernelParams, extra,
nullptr, nullptr));
hipError_t hipModuleLaunchKernel(hipFunction_t f, uint32_t gridDimX, uint32_t gridDimY,
uint32_t gridDimZ, uint32_t blockDimX, uint32_t blockDimY,
uint32_t blockDimZ, uint32_t sharedMemBytes, hipStream_t hStream,
void** kernelParams, void** extra) {
HIP_INIT_API(f, gridDimX, gridDimY, gridDimZ, blockDimX, blockDimY, blockDimZ, sharedMemBytes,
hStream, kernelParams, extra);
return ihipLogStatus(ihipModuleLaunchKernel(
f, blockDimX * gridDimX, blockDimY * gridDimY, gridDimZ * blockDimZ, blockDimX, blockDimY,
blockDimZ, sharedMemBytes, hStream, kernelParams, extra, nullptr, nullptr));
}
hipError_t hipHccModuleLaunchKernel(hipFunction_t f,
uint32_t globalWorkSizeX, uint32_t globalWorkSizeY, uint32_t globalWorkSizeZ,
uint32_t localWorkSizeX, uint32_t localWorkSizeY, uint32_t localWorkSizeZ,
size_t sharedMemBytes, hipStream_t hStream,
void **kernelParams, void **extra,
hipEvent_t startEvent, hipEvent_t stopEvent)
{
HIP_INIT_API(f, globalWorkSizeX, globalWorkSizeY, globalWorkSizeZ,
localWorkSizeX, localWorkSizeY, localWorkSizeZ,
sharedMemBytes, hStream,
kernelParams, extra);
return ihipLogStatus(ihipModuleLaunchKernel(f, globalWorkSizeX, globalWorkSizeY, globalWorkSizeZ,
localWorkSizeX, localWorkSizeY, localWorkSizeZ,
sharedMemBytes, hStream, kernelParams, extra, startEvent, stopEvent));
hipError_t hipHccModuleLaunchKernel(hipFunction_t f, uint32_t globalWorkSizeX,
uint32_t globalWorkSizeY, uint32_t globalWorkSizeZ,
uint32_t localWorkSizeX, uint32_t localWorkSizeY,
uint32_t localWorkSizeZ, size_t sharedMemBytes,
hipStream_t hStream, void** kernelParams, void** extra,
hipEvent_t startEvent, hipEvent_t stopEvent) {
HIP_INIT_API(f, globalWorkSizeX, globalWorkSizeY, globalWorkSizeZ, localWorkSizeX,
localWorkSizeY, localWorkSizeZ, sharedMemBytes, hStream, kernelParams, extra);
return ihipLogStatus(ihipModuleLaunchKernel(
f, globalWorkSizeX, globalWorkSizeY, globalWorkSizeZ, localWorkSizeX, localWorkSizeY,
localWorkSizeZ, sharedMemBytes, hStream, kernelParams, extra, startEvent, stopEvent));
}
namespace
{
struct Agent_global {
string name;
hipDeviceptr_t address;
uint32_t byte_cnt;
};
namespace {
struct Agent_global {
string name;
hipDeviceptr_t address;
uint32_t byte_cnt;
};
inline
void track(const Agent_global& x)
{
tprintf(
DB_MEM,
" add variable '%s' with ptr=%p size=%u to tracker\n",
x.name.c_str(),
x.address,
x.byte_cnt);
inline void track(const Agent_global& x) {
tprintf(DB_MEM, " add variable '%s' with ptr=%p size=%u to tracker\n", x.name.c_str(),
x.address, x.byte_cnt);
auto device = ihipGetTlsDefaultCtx()->getWriteableDevice();
auto device = ihipGetTlsDefaultCtx()->getWriteableDevice();
hc::AmPointerInfo ptr_info(
nullptr,
x.address,
x.address,
x.byte_cnt,
device->_acc,
true,
false);
hc::am_memtracker_add(x.address, ptr_info);
hc::am_memtracker_update(x.address, device->_deviceId, 0u);
hc::AmPointerInfo ptr_info(nullptr, x.address, x.address, x.byte_cnt, device->_acc, true,
false);
hc::am_memtracker_add(x.address, ptr_info);
hc::am_memtracker_update(x.address, device->_deviceId, 0u);
}
template <typename Container = vector<Agent_global>>
inline hsa_status_t copy_agent_global_variables(hsa_executable_t, hsa_agent_t,
hsa_executable_symbol_t x, void* out) {
assert(out);
hsa_symbol_kind_t t = {};
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_TYPE, &t);
if (t == HSA_SYMBOL_KIND_VARIABLE) {
static_cast<Container*>(out)->push_back(Agent_global{name(x), address(x), size(x)});
track(static_cast<Container*>(out)->back());
}
template<typename Container = vector<Agent_global>>
inline
hsa_status_t copy_agent_global_variables(
hsa_executable_t, hsa_agent_t, hsa_executable_symbol_t x, void* out)
{
assert(out);
return HSA_STATUS_SUCCESS;
}
hsa_symbol_kind_t t = {};
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_TYPE, &t);
inline hsa_agent_t this_agent() {
auto ctx = ihipGetTlsDefaultCtx();
if (t == HSA_SYMBOL_KIND_VARIABLE) {
static_cast<Container*>(out)->push_back(
Agent_global{name(x), address(x), size(x)});
if (!ctx) throw runtime_error{"No active HIP context."};
track(static_cast<Container*>(out)->back());
}
auto device = ctx->getDevice();
return HSA_STATUS_SUCCESS;
}
if (!device) throw runtime_error{"No device available for HIP."};
inline
hsa_agent_t this_agent()
{
auto ctx = ihipGetTlsDefaultCtx();
ihipDevice_t* currentDevice = ihipGetDevice(device->_deviceId);
if (!ctx) throw runtime_error{"No active HIP context."};
if (!currentDevice) throw runtime_error{"No active device for HIP."};
auto device = ctx->getDevice();
return currentDevice->_hsaAgent;
}
if (!device) throw runtime_error{"No device available for HIP."};
inline vector<Agent_global> read_agent_globals(hsa_agent_t agent, hsa_executable_t executable) {
vector<Agent_global> r;
ihipDevice_t *currentDevice = ihipGetDevice(device->_deviceId);
hsa_executable_iterate_agent_symbols(executable, agent, copy_agent_global_variables, &r);
if (!currentDevice) throw runtime_error{"No active device for HIP."};
return r;
}
return currentDevice->_hsaAgent;
}
template <typename ForwardIterator>
pair<hipDeviceptr_t, size_t> read_global_description(ForwardIterator f, ForwardIterator l,
const char* name) {
const auto it = std::find_if(f, l, [=](const Agent_global& x) { return x.name == name; });
inline
vector<Agent_global> read_agent_globals(
hsa_agent_t agent, hsa_executable_t executable)
{
vector<Agent_global> r;
return it == l ? make_pair(nullptr, 0u) : make_pair(it->address, it->byte_cnt);
}
hsa_executable_iterate_agent_symbols(
executable, agent, copy_agent_global_variables, &r);
hipError_t read_agent_global_from_module(hipDeviceptr_t* dptr, size_t* bytes, hipModule_t hmod,
const char* name) {
static unordered_map<hipModule_t, vector<Agent_global>> agent_globals;
return r;
}
// TODO: this is not particularly robust.
if (agent_globals.count(hmod) == 0) {
static mutex mtx;
lock_guard<mutex> lck{mtx};
template<typename ForwardIterator>
pair<hipDeviceptr_t, size_t> read_global_description(
ForwardIterator f, ForwardIterator l, const char* name)
{
const auto it = std::find_if(
f, l, [=](const Agent_global& x) { return x.name == name; });
return it == l ?
make_pair(nullptr, 0u) : make_pair(it->address, it->byte_cnt);
}
hipError_t read_agent_global_from_module(
hipDeviceptr_t *dptr,
size_t* bytes,
hipModule_t hmod,
const char* name)
{
static unordered_map<hipModule_t, vector<Agent_global>> agent_globals;
// TODO: this is not particularly robust.
if (agent_globals.count(hmod) == 0) {
static mutex mtx;
lock_guard<mutex> lck{mtx};
if (agent_globals.count(hmod) == 0) {
agent_globals.emplace(
hmod, read_agent_globals(this_agent(), hmod->executable));
}
agent_globals.emplace(hmod, read_agent_globals(this_agent(), hmod->executable));
}
// TODO: This is unsafe iff some other emplacement triggers rehashing.
// It will have to be properly fleshed out in the future.
const auto it0 = agent_globals.find(hmod);
if (it0 == agent_globals.cend()) {
throw runtime_error{"agent_globals data structure corrupted."};
}
tie(*dptr, *bytes) = read_global_description(
it0->second.cbegin(), it0->second.cend(), name);
return dptr ? hipSuccess : hipErrorNotFound;
}
hipError_t read_agent_global_from_process(
hipDeviceptr_t *dptr, size_t* bytes, const char* name)
{
static unordered_map<hsa_agent_t, vector<Agent_global>> agent_globals;
static std::once_flag f;
// TODO: This is unsafe iff some other emplacement triggers rehashing.
// It will have to be properly fleshed out in the future.
const auto it0 = agent_globals.find(hmod);
if (it0 == agent_globals.cend()) {
throw runtime_error{"agent_globals data structure corrupted."};
}
call_once(f, []() {
for (auto&& agent_executables : hip_impl::executables()) {
vector<Agent_global> tmp0;
for (auto&& executable : agent_executables.second) {
auto tmp1 = read_agent_globals(
agent_executables.first, executable);
tmp0.insert(
tmp0.end(),
make_move_iterator(tmp1.begin()),
make_move_iterator(tmp1.end()));
}
agent_globals.emplace(agent_executables.first, move(tmp0));
tie(*dptr, *bytes) = read_global_description(it0->second.cbegin(), it0->second.cend(), name);
return dptr ? hipSuccess : hipErrorNotFound;
}
hipError_t read_agent_global_from_process(hipDeviceptr_t* dptr, size_t* bytes, const char* name) {
static unordered_map<hsa_agent_t, vector<Agent_global>> agent_globals;
static std::once_flag f;
call_once(f, []() {
for (auto&& agent_executables : hip_impl::executables()) {
vector<Agent_global> tmp0;
for (auto&& executable : agent_executables.second) {
auto tmp1 = read_agent_globals(agent_executables.first, executable);
tmp0.insert(tmp0.end(), make_move_iterator(tmp1.begin()),
make_move_iterator(tmp1.end()));
}
});
agent_globals.emplace(agent_executables.first, move(tmp0));
}
});
const auto it = agent_globals.find(this_agent());
const auto it = agent_globals.find(this_agent());
if (it == agent_globals.cend()) return hipErrorNotInitialized;
if (it == agent_globals.cend()) return hipErrorNotInitialized;
tie(*dptr, *bytes) = read_global_description(
it->second.cbegin(), it->second.cend(), name);
tie(*dptr, *bytes) = read_global_description(it->second.cbegin(), it->second.cend(), name);
return dptr ? hipSuccess : hipErrorNotFound;
}
return dptr ? hipSuccess : hipErrorNotFound;
}
hsa_executable_symbol_t find_kernel_by_name(
hsa_executable_t executable, const char* kname)
{
pair<const char*, hsa_executable_symbol_t> r{kname, {}};
hsa_executable_symbol_t find_kernel_by_name(hsa_executable_t executable, const char* kname) {
pair<const char*, hsa_executable_symbol_t> r{kname, {}};
hsa_executable_iterate_agent_symbols(
executable,
this_agent(),
[](hsa_executable_t, hsa_agent_t, hsa_executable_symbol_t x, void* s) {
auto p =
static_cast<pair<const char*, hsa_executable_symbol_t>*>(s);
hsa_executable_iterate_agent_symbols(
executable, this_agent(),
[](hsa_executable_t, hsa_agent_t, hsa_executable_symbol_t x, void* s) {
auto p = static_cast<pair<const char*, hsa_executable_symbol_t>*>(s);
if (type(x) != HSA_SYMBOL_KIND_KERNEL) {
return HSA_STATUS_SUCCESS;
}
if (name(x) != p->first) return HSA_STATUS_SUCCESS;
if (type(x) != HSA_SYMBOL_KIND_KERNEL) {
return HSA_STATUS_SUCCESS;
}
if (name(x) != p->first) return HSA_STATUS_SUCCESS;
p->second = x;
p->second = x;
return HSA_STATUS_INFO_BREAK;
}, &r);
return HSA_STATUS_INFO_BREAK;
},
&r);
return r.second;
}
return r.second;
}
string read_elf_file_as_string(const void* file)
{ // Precondition: file points to an ELF image that was BITWISE loaded
// into process accessible memory, and not one loaded by
// the loader. This is because in the latter case
// alignment may differ, which will break the size
// computation.
// the image is Elf64, and matches endianness i.e. it is
// Little Endian.
if (!file) return {};
string read_elf_file_as_string(
const void* file) { // Precondition: file points to an ELF image that was BITWISE loaded
// into process accessible memory, and not one loaded by
// the loader. This is because in the latter case
// alignment may differ, which will break the size
// computation.
// the image is Elf64, and matches endianness i.e. it is
// Little Endian.
if (!file) return {};
auto h = static_cast<const Elf64_Ehdr*>(file);
auto s = static_cast<const char*>(file);
// This assumes the common case of SHT being the last part of the ELF.
auto sz = sizeof(Elf64_Ehdr) + h->e_shoff + h->e_shentsize * h->e_shnum;
auto h = static_cast<const Elf64_Ehdr*>(file);
auto s = static_cast<const char*>(file);
// This assumes the common case of SHT being the last part of the ELF.
auto sz = sizeof(Elf64_Ehdr) + h->e_shoff + h->e_shentsize * h->e_shnum;
return string{s, s + sz};
}
return string{s, s + sz};
}
string code_object_blob_for_agent(
const void* maybe_bundled_code, hsa_agent_t agent)
{
if (!maybe_bundled_code) return {};
string code_object_blob_for_agent(const void* maybe_bundled_code, hsa_agent_t agent) {
if (!maybe_bundled_code) return {};
Bundled_code_header tmp{maybe_bundled_code};
Bundled_code_header tmp{maybe_bundled_code};
if (!valid(tmp)) return {};
if (!valid(tmp)) return {};
const auto agent_isa = isa(agent);
const auto agent_isa = isa(agent);
const auto it = find_if(
bundles(tmp).cbegin(),
bundles(tmp).cend(),
[=](const Bundled_code& x) {
return agent_isa == triple_to_hsa_isa(x.triple);;
});
const auto it = find_if(bundles(tmp).cbegin(), bundles(tmp).cend(), [=](const Bundled_code& x) {
return agent_isa == triple_to_hsa_isa(x.triple);
;
});
if (it == bundles(tmp).cend()) return {};
if (it == bundles(tmp).cend()) return {};
return string{it->blob.cbegin(), it->blob.cend()};
}
} // Anonymous namespace, internal linkage.
return string{it->blob.cbegin(), it->blob.cend()};
}
} // namespace
hipError_t ihipModuleGetFunction(
hipFunction_t *func, hipModule_t hmod, const char *name)
{
hipError_t ihipModuleGetFunction(hipFunction_t* func, hipModule_t hmod, const char* name) {
HIP_INIT_API(func, hmod, name);
if (!func || !name) return ihipLogStatus(hipErrorInvalidValue);
@@ -510,30 +455,26 @@ hipError_t ihipModuleGetFunction(
return ihipLogStatus(hipSuccess);
}
hipError_t hipModuleGetFunction(hipFunction_t *hfunc, hipModule_t hmod,
const char *name){
hipError_t hipModuleGetFunction(hipFunction_t* hfunc, hipModule_t hmod, const char* name) {
HIP_INIT_API(hfunc, hmod, name);
return ihipLogStatus(ihipModuleGetFunction(hfunc, hmod, name));
}
hipError_t hipModuleGetGlobal(hipDeviceptr_t *dptr, size_t *bytes,
hipModule_t hmod, const char* name)
{
hipError_t hipModuleGetGlobal(hipDeviceptr_t* dptr, size_t* bytes, hipModule_t hmod,
const char* name) {
HIP_INIT_API(dptr, bytes, hmod, name);
if(!dptr || !bytes) return ihipLogStatus(hipErrorInvalidValue);
if (!dptr || !bytes) return ihipLogStatus(hipErrorInvalidValue);
if(!name) return ihipLogStatus(hipErrorNotInitialized);
if (!name) return ihipLogStatus(hipErrorNotInitialized);
const auto r = hmod ?
read_agent_global_from_module(dptr, bytes, hmod, name) :
read_agent_global_from_process(dptr, bytes, name);
const auto r = hmod ? read_agent_global_from_module(dptr, bytes, hmod, name)
: read_agent_global_from_process(dptr, bytes, name);
return ihipLogStatus(r);
}
hipError_t hipModuleLoad(hipModule_t *module, const char *fname)
{
hipError_t hipModuleLoad(hipModule_t* module, const char* fname) {
HIP_INIT_API(module, fname);
if (!fname) return ihipLogStatus(hipErrorInvalidValue);
@@ -542,14 +483,12 @@ hipError_t hipModuleLoad(hipModule_t *module, const char *fname)
if (!file.is_open()) return ihipLogStatus(hipErrorFileNotFound);
vector<char> tmp{
istreambuf_iterator<char>{file}, istreambuf_iterator<char>{}};
vector<char> tmp{istreambuf_iterator<char>{file}, istreambuf_iterator<char>{}};
return hipModuleLoadData(module, tmp.data());
}
hipError_t hipModuleLoadData(hipModule_t *module, const void *image)
{
hipError_t hipModuleLoadData(hipModule_t* module, const void* image) {
HIP_INIT_API(module, image);
if (!module) return ihipLogStatus(hipErrorInvalidValue);
@@ -559,37 +498,29 @@ hipError_t hipModuleLoadData(hipModule_t *module, const void *image)
auto ctx = ihipGetTlsDefaultCtx();
if (!ctx) return ihipLogStatus(hipErrorInvalidContext);
hsa_executable_create_alt(
HSA_PROFILE_FULL,
HSA_DEFAULT_FLOAT_ROUNDING_MODE_DEFAULT,
nullptr,
&(*module)->executable);
hsa_executable_create_alt(HSA_PROFILE_FULL, HSA_DEFAULT_FLOAT_ROUNDING_MODE_DEFAULT, nullptr,
&(*module)->executable);
auto tmp = code_object_blob_for_agent(image, this_agent());
(*module)->executable = hip_impl::load_executable(
tmp.empty() ? read_elf_file_as_string(image) : tmp,
(*module)->executable,
this_agent());
tmp.empty() ? read_elf_file_as_string(image) : tmp, (*module)->executable, this_agent());
return ihipLogStatus(
(*module)->executable.handle ? hipSuccess : hipErrorUnknown);
return ihipLogStatus((*module)->executable.handle ? hipSuccess : hipErrorUnknown);
}
hipError_t hipModuleLoadDataEx(hipModule_t *module, const void *image, unsigned int numOptions, hipJitOption *options, void **optionValues)
{
hipError_t hipModuleLoadDataEx(hipModule_t* module, const void* image, unsigned int numOptions,
hipJitOption* options, void** optionValues) {
return hipModuleLoadData(module, image);
}
hipError_t hipModuleGetTexRef(
textureReference** texRef, hipModule_t hmod, const char* name)
{
hipError_t hipModuleGetTexRef(textureReference** texRef, hipModule_t hmod, const char* name) {
HIP_INIT_API(texRef, hmod, name);
hipError_t ret = hipErrorNotFound;
if(!texRef) return ihipLogStatus(hipErrorInvalidValue);
if (!texRef) return ihipLogStatus(hipErrorInvalidValue);
if(!hmod || !name) return ihipLogStatus(hipErrorNotInitialized);
if (!hmod || !name) return ihipLogStatus(hipErrorNotInitialized);
const auto it = globals().find(name);
if (it == globals().end()) return ihipLogStatus(hipErrorInvalidValue);
+43 -48
View File
@@ -30,27 +30,27 @@ THE SOFTWARE.
// Peer access functions.
// There are two flavors:
// - one where contexts are specified with hipCtx_t type.
// - one where contexts are specified with integer deviceIds, that are mapped to the primary context for that device.
// The implementation contains a set of internal ihip* functions which operate on contexts. Then the
// public APIs are thin wrappers which call into this internal implementations.
// TODO - actually not yet - currently the integer deviceId flavors just call the context APIs. need to fix.
// - one where contexts are specified with integer deviceIds, that are mapped to the primary
// context for that device.
// The implementation contains a set of internal ihip* functions which operate on contexts. Then
// the public APIs are thin wrappers which call into this internal implementations.
// TODO - actually not yet - currently the integer deviceId flavors just call the context APIs. need
// to fix.
hipError_t ihipDeviceCanAccessPeer (int* canAccessPeer, hipCtx_t thisCtx, hipCtx_t peerCtx)
{
hipError_t ihipDeviceCanAccessPeer(int* canAccessPeer, hipCtx_t thisCtx, hipCtx_t peerCtx) {
hipError_t err = hipSuccess;
if ((thisCtx != NULL) && (peerCtx != NULL)) {
if (thisCtx == peerCtx) {
*canAccessPeer = 0;
tprintf(DB_MEM, "Can't be peer to self. (this=%s, peer=%s)\n",
thisCtx->toString().c_str(), peerCtx->toString().c_str());
} else if (HIP_FORCE_P2P_HOST & 0x2) {
} else if (HIP_FORCE_P2P_HOST & 0x2) {
*canAccessPeer = false;
tprintf(DB_MEM, "HIP_FORCE_P2P_HOST denies peer access this=%s peer=%s canAccessPeer=%d\n",
tprintf(DB_MEM,
"HIP_FORCE_P2P_HOST denies peer access this=%s peer=%s canAccessPeer=%d\n",
thisCtx->toString().c_str(), peerCtx->toString().c_str(), *canAccessPeer);
} else {
*canAccessPeer = peerCtx->getDevice()->_acc.get_is_peer(thisCtx->getDevice()->_acc);
@@ -72,8 +72,7 @@ hipError_t ihipDeviceCanAccessPeer (int* canAccessPeer, hipCtx_t thisCtx, hipCtx
* HCC returns 0 in *canAccessPeer ; Need to update this function when RT supports P2P
*/
//---
hipError_t hipDeviceCanAccessPeer (int* canAccessPeer, hipCtx_t thisCtx, hipCtx_t peerCtx)
{
hipError_t hipDeviceCanAccessPeer(int* canAccessPeer, hipCtx_t thisCtx, hipCtx_t peerCtx) {
HIP_INIT_API(canAccessPeer, thisCtx, peerCtx);
return ihipLogStatus(ihipDeviceCanAccessPeer(canAccessPeer, thisCtx, peerCtx));
@@ -83,28 +82,28 @@ hipError_t hipDeviceCanAccessPeer (int* canAccessPeer, hipCtx_t thisCtx, hipCtx_
//---
// Disable visibility of this device into memory allocated on peer device.
// Remove this device from peer device peerlist.
hipError_t ihipDisablePeerAccess (hipCtx_t peerCtx)
{
hipError_t ihipDisablePeerAccess(hipCtx_t peerCtx) {
hipError_t err = hipSuccess;
auto thisCtx = ihipGetTlsDefaultCtx();
if ((thisCtx != NULL) && (peerCtx != NULL)) {
bool canAccessPeer = peerCtx->getDevice()->_acc.get_is_peer(thisCtx->getDevice()->_acc);
bool canAccessPeer = peerCtx->getDevice()->_acc.get_is_peer(thisCtx->getDevice()->_acc);
if (! canAccessPeer) {
if (!canAccessPeer) {
err = hipErrorInvalidDevice; // P2P not allowed between these devices.
} else if (thisCtx == peerCtx) {
} else if (thisCtx == peerCtx) {
err = hipErrorInvalidDevice; // Can't disable peer access to self.
} else {
LockedAccessor_CtxCrit_t peerCrit(peerCtx->criticalData());
bool changed = peerCrit->removePeerWatcher(peerCtx, thisCtx);
if (changed) {
tprintf(DB_MEM, "device %s disable access to memory allocated on peer:%s\n",
thisCtx->toString().c_str(), peerCtx->toString().c_str());
thisCtx->toString().c_str(), peerCtx->toString().c_str());
// Update the peers for all memory already saved in the tracker:
am_memtracker_update_peers(peerCtx->getDevice()->_acc, peerCrit->peerCnt(), peerCrit->peerAgents());
am_memtracker_update_peers(peerCtx->getDevice()->_acc, peerCrit->peerCnt(),
peerCrit->peerAgents());
} else {
err = hipErrorPeerAccessNotEnabled; // never enabled P2P access.
err = hipErrorPeerAccessNotEnabled; // never enabled P2P access.
}
}
} else {
@@ -118,24 +117,24 @@ hipError_t ihipDisablePeerAccess (hipCtx_t peerCtx)
//---
// Allow the current device to see all memory allocated on peerCtx.
// This should add this device to the peer-device peer list.
hipError_t ihipEnablePeerAccess (hipCtx_t peerCtx, unsigned int flags)
{
hipError_t ihipEnablePeerAccess(hipCtx_t peerCtx, unsigned int flags) {
hipError_t err = hipSuccess;
if (flags != 0) {
err = hipErrorInvalidValue;
} else {
auto thisCtx = ihipGetTlsDefaultCtx();
if (thisCtx == peerCtx) {
if (thisCtx == peerCtx) {
err = hipErrorInvalidDevice; // Can't enable peer access to self.
} else if ((thisCtx != NULL) && (peerCtx != NULL)) {
LockedAccessor_CtxCrit_t peerCrit(peerCtx->criticalData());
// Add thisCtx to peerCtx's access list so that new allocations on peer will be made visible to this device:
// Add thisCtx to peerCtx's access list so that new allocations on peer will be made
// visible to this device:
bool isNewPeer = peerCrit->addPeerWatcher(peerCtx, thisCtx);
if (isNewPeer) {
tprintf(DB_MEM, "device=%s can now see all memory allocated on peer=%s\n",
thisCtx->toString().c_str(), peerCtx->toString().c_str());
am_memtracker_update_peers(peerCtx->getDevice()->_acc, peerCrit->peerCnt(), peerCrit->peerAgents());
thisCtx->toString().c_str(), peerCtx->toString().c_str());
am_memtracker_update_peers(peerCtx->getDevice()->_acc, peerCrit->peerCnt(),
peerCrit->peerAgents());
} else {
err = hipErrorPeerAccessAlreadyEnabled;
}
@@ -149,8 +148,8 @@ hipError_t ihipEnablePeerAccess (hipCtx_t peerCtx, unsigned int flags)
//---
hipError_t hipMemcpyPeer (void* dst, hipCtx_t dstCtx, const void* src, hipCtx_t srcCtx, size_t sizeBytes)
{
hipError_t hipMemcpyPeer(void* dst, hipCtx_t dstCtx, const void* src, hipCtx_t srcCtx,
size_t sizeBytes) {
HIP_INIT_API(dst, dstCtx, src, srcCtx, sizeBytes);
// TODO - move to ihip memory copy implementaion.
@@ -160,8 +159,8 @@ hipError_t hipMemcpyPeer (void* dst, hipCtx_t dstCtx, const void* src, hipCtx_t
//---
hipError_t hipMemcpyPeerAsync (void* dst, hipCtx_t dstDevice, const void* src, hipCtx_t srcDevice, size_t sizeBytes, hipStream_t stream)
{
hipError_t hipMemcpyPeerAsync(void* dst, hipCtx_t dstDevice, const void* src, hipCtx_t srcDevice,
size_t sizeBytes, hipStream_t stream) {
HIP_INIT_API(dst, dstDevice, src, srcDevice, sizeBytes, stream);
// TODO - move to ihip memory copy implementaion.
@@ -170,57 +169,53 @@ hipError_t hipMemcpyPeerAsync (void* dst, hipCtx_t dstDevice, const void* src, h
};
//=============================================================================
// These are the flavors that accept integer deviceIDs.
// Implementations map these to primary contexts and call the internal functions above.
//=============================================================================
hipError_t hipDeviceCanAccessPeer (int* canAccessPeer, int deviceId, int peerDeviceId)
{
hipError_t hipDeviceCanAccessPeer(int* canAccessPeer, int deviceId, int peerDeviceId) {
HIP_INIT_API(canAccessPeer, deviceId, peerDeviceId);
return ihipLogStatus(ihipDeviceCanAccessPeer(canAccessPeer, ihipGetPrimaryCtx(deviceId), ihipGetPrimaryCtx(peerDeviceId)));
return ihipLogStatus(ihipDeviceCanAccessPeer(canAccessPeer, ihipGetPrimaryCtx(deviceId),
ihipGetPrimaryCtx(peerDeviceId)));
}
hipError_t hipDeviceDisablePeerAccess (int peerDeviceId)
{
hipError_t hipDeviceDisablePeerAccess(int peerDeviceId) {
HIP_INIT_API(peerDeviceId);
return ihipLogStatus(ihipDisablePeerAccess(ihipGetPrimaryCtx(peerDeviceId)));
}
hipError_t hipDeviceEnablePeerAccess (int peerDeviceId, unsigned int flags)
{
hipError_t hipDeviceEnablePeerAccess(int peerDeviceId, unsigned int flags) {
HIP_INIT_API(peerDeviceId, flags);
return ihipLogStatus(ihipEnablePeerAccess(ihipGetPrimaryCtx(peerDeviceId), flags));
}
hipError_t hipMemcpyPeer (void* dst, int dstDevice, const void* src, int srcDevice, size_t sizeBytes)
{
hipError_t hipMemcpyPeer(void* dst, int dstDevice, const void* src, int srcDevice,
size_t sizeBytes) {
HIP_INIT_API(dst, dstDevice, src, srcDevice, sizeBytes);
return ihipLogStatus(hipMemcpyPeer(dst, ihipGetPrimaryCtx(dstDevice), src, ihipGetPrimaryCtx(srcDevice), sizeBytes));
return ihipLogStatus(hipMemcpyPeer(dst, ihipGetPrimaryCtx(dstDevice), src,
ihipGetPrimaryCtx(srcDevice), sizeBytes));
}
hipError_t hipMemcpyPeerAsync (void* dst, int dstDevice, const void* src, int srcDevice, size_t sizeBytes, hipStream_t stream)
{
hipError_t hipMemcpyPeerAsync(void* dst, int dstDevice, const void* src, int srcDevice,
size_t sizeBytes, hipStream_t stream) {
HIP_INIT_API(dst, dstDevice, src, srcDevice, sizeBytes, stream);
return ihipLogStatus(hip_internal::memcpyAsync(dst, src, sizeBytes, hipMemcpyDefault, stream));
}
hipError_t hipCtxEnablePeerAccess (hipCtx_t peerCtx, unsigned int flags)
{
hipError_t hipCtxEnablePeerAccess(hipCtx_t peerCtx, unsigned int flags) {
HIP_INIT_API(peerCtx, flags);
return ihipLogStatus(ihipEnablePeerAccess(peerCtx, flags));
}
hipError_t hipCtxDisablePeerAccess (hipCtx_t peerCtx)
{
hipError_t hipCtxDisablePeerAccess(hipCtx_t peerCtx) {
HIP_INIT_API(peerCtx);
return ihipLogStatus(ihipDisablePeerAccess(peerCtx));
+29 -40
View File
@@ -33,27 +33,25 @@ THE SOFTWARE.
//
//---
hipError_t ihipStreamCreate(hipStream_t *stream, unsigned int flags)
{
ihipCtx_t *ctx = ihipGetTlsDefaultCtx();
hipError_t ihipStreamCreate(hipStream_t* stream, unsigned int flags) {
ihipCtx_t* ctx = ihipGetTlsDefaultCtx();
hipError_t e = hipSuccess;
if (ctx) {
if (HIP_FORCE_NULL_STREAM) {
*stream = 0;
*stream = 0;
} else {
hc::accelerator acc = ctx->getWriteableDevice()->_acc;
// TODO - se try-catch loop to detect memory exception?
//
//Note this is an execute_in_order queue, so all kernels submitted will atuomatically wait for prev to complete:
//This matches CUDA stream behavior:
// Note this is an execute_in_order queue, so all kernels submitted will atuomatically
// wait for prev to complete: This matches CUDA stream behavior:
{
// Obtain mutex access to the device critical data, release by destructor
LockedAccessor_CtxCrit_t ctxCrit(ctx->criticalData());
LockedAccessor_CtxCrit_t ctxCrit(ctx->criticalData());
auto istream = new ihipStream_t(ctx, acc.create_view(), flags);
@@ -72,25 +70,21 @@ hipError_t ihipStreamCreate(hipStream_t *stream, unsigned int flags)
//---
hipError_t hipStreamCreateWithFlags(hipStream_t *stream, unsigned int flags)
{
hipError_t hipStreamCreateWithFlags(hipStream_t* stream, unsigned int flags) {
HIP_INIT_API(stream, flags);
return ihipLogStatus(ihipStreamCreate(stream, flags));
}
//---
hipError_t hipStreamCreate(hipStream_t *stream)
{
hipError_t hipStreamCreate(hipStream_t* stream) {
HIP_INIT_API(stream);
return ihipLogStatus(ihipStreamCreate(stream, hipStreamDefault));
}
hipError_t hipStreamWaitEvent(hipStream_t stream, hipEvent_t event, unsigned int flags)
{
hipError_t hipStreamWaitEvent(hipStream_t stream, hipEvent_t event, unsigned int flags) {
HIP_INIT_SPECIAL_API(TRACE_SYNC, stream, event, flags);
hipError_t e = hipSuccess;
@@ -100,35 +94,34 @@ hipError_t hipStreamWaitEvent(hipStream_t stream, hipEvent_t event, unsigned int
if (event == nullptr) {
e = hipErrorInvalidResourceHandle;
} else if ((ecd._state != hipEventStatusUnitialized) &&
(ecd._state != hipEventStatusCreated)) {
} else if ((ecd._state != hipEventStatusUnitialized) && (ecd._state != hipEventStatusCreated)) {
if (HIP_SYNC_STREAM_WAIT || (HIP_SYNC_NULL_STREAM && (stream == 0))) {
// conservative wait on host for the specified event to complete:
// return _stream->locked_eventWaitComplete(this, waitMode);
//
ecd._stream->locked_eventWaitComplete(ecd.marker(), (event->_flags & hipEventBlockingSync) ? hc::hcWaitModeBlocked : hc::hcWaitModeActive);
ecd._stream->locked_eventWaitComplete(
ecd.marker(), (event->_flags & hipEventBlockingSync) ? hc::hcWaitModeBlocked
: hc::hcWaitModeActive);
} else {
stream = ihipSyncAndResolveStream(stream);
// This will use create_blocking_marker to wait on the specified queue.
stream->locked_streamWaitEvent(ecd);
}
} // else event not recorded, return immediately and don't create marker.
} // else event not recorded, return immediately and don't create marker.
return ihipLogStatus(e);
};
//---
hipError_t hipStreamQuery(hipStream_t stream)
{
hipError_t hipStreamQuery(hipStream_t stream) {
HIP_INIT_SPECIAL_API(TRACE_QUERY, stream);
// Use default stream if 0 specified:
if (stream == hipStreamNull) {
ihipCtx_t *device = ihipGetTlsDefaultCtx();
stream = device->_defaultStream;
ihipCtx_t* device = ihipGetTlsDefaultCtx();
stream = device->_defaultStream;
}
bool isEmpty = 0;
@@ -138,15 +131,14 @@ hipError_t hipStreamQuery(hipStream_t stream)
isEmpty = crit->_av.get_is_empty();
}
hipError_t e = isEmpty ? hipSuccess : hipErrorNotReady ;
hipError_t e = isEmpty ? hipSuccess : hipErrorNotReady;
return ihipLogStatus(e);
}
//---
hipError_t hipStreamSynchronize(hipStream_t stream)
{
hipError_t hipStreamSynchronize(hipStream_t stream) {
HIP_INIT_SPECIAL_API(TRACE_SYNC, stream);
return ihipLogStatus(ihipStreamSynchronize(stream));
@@ -157,8 +149,7 @@ hipError_t hipStreamSynchronize(hipStream_t stream)
/**
* @return #hipSuccess, #hipErrorInvalidResourceHandle
*/
hipError_t hipStreamDestroy(hipStream_t stream)
{
hipError_t hipStreamDestroy(hipStream_t stream) {
HIP_INIT_API(stream);
hipError_t e = hipSuccess;
@@ -166,12 +157,12 @@ hipError_t hipStreamDestroy(hipStream_t stream)
//--- Drain the stream:
if (stream == NULL) {
if (!HIP_FORCE_NULL_STREAM) {
e = hipErrorInvalidResourceHandle;
}
e = hipErrorInvalidResourceHandle;
}
} else {
stream->locked_wait();
ihipCtx_t *ctx = stream->getCtx();
ihipCtx_t* ctx = stream->getCtx();
if (ctx) {
ctx->locked_removeStream(stream);
@@ -186,8 +177,7 @@ hipError_t hipStreamDestroy(hipStream_t stream)
//---
hipError_t hipStreamGetFlags(hipStream_t stream, unsigned int *flags)
{
hipError_t hipStreamGetFlags(hipStream_t stream, unsigned int* flags) {
HIP_INIT_API(stream, flags);
if (flags == NULL) {
@@ -202,19 +192,18 @@ hipError_t hipStreamGetFlags(hipStream_t stream, unsigned int *flags)
//---
hipError_t hipStreamAddCallback(hipStream_t stream, hipStreamCallback_t callback, void *userData, unsigned int flags)
{
hipError_t hipStreamAddCallback(hipStream_t stream, hipStreamCallback_t callback, void* userData,
unsigned int flags) {
HIP_INIT_API(stream, callback, userData, flags);
hipError_t e = hipSuccess;
// Create a thread in detached mode to handle callback
ihipStreamCallback_t *cb = new ihipStreamCallback_t(stream, callback, userData);
std::thread (ihipStreamCallbackHandler, cb).detach();
ihipStreamCallback_t* cb = new ihipStreamCallback_t(stream, callback, userData);
std::thread(ihipStreamCallbackHandler, cb).detach();
// Wait for thread to be ready
cb->_mtx.lock();
while(cb->_ready != true)
{
while (cb->_ready != true) {
cb->_mtx.unlock();
std::this_thread::sleep_for(std::chrono::milliseconds(10));
cb->_mtx.lock();
+15 -18
View File
@@ -32,9 +32,7 @@ THE SOFTWARE.
static std::map<hipSurfaceObject_t, hipSurface*> surfaceHash;
void saveSurfaceInfo(const hipSurface* pSurface,
const hipResourceDesc* pResDesc)
{
void saveSurfaceInfo(const hipSurface* pSurface, const hipResourceDesc* pResDesc) {
if (pResDesc != nullptr) {
memcpy((void*)&(pSurface->resDesc), (void*)pResDesc, sizeof(hipResourceDesc));
}
@@ -42,41 +40,40 @@ void saveSurfaceInfo(const hipSurface* pSurface,
// Surface Object APIs
hipError_t hipCreateSurfaceObject(hipSurfaceObject_t* pSurfObject,
const hipResourceDesc* pResDesc)
{
const hipResourceDesc* pResDesc) {
HIP_INIT_API(pSurfObject, pResDesc);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
hipSurface* pSurface = (hipSurface*) malloc(sizeof(hipSurface));
hipSurface* pSurface = (hipSurface*)malloc(sizeof(hipSurface));
if (pSurface != nullptr) {
memset(pSurface, 0, sizeof(hipSurface));
saveSurfaceInfo(pSurface, pResDesc);
}
switch (pResDesc->resType) {
case hipResourceTypeArray:
pSurface->array = pResDesc->res.array.array;
break;
default:
break;
case hipResourceTypeArray:
pSurface->array = pResDesc->res.array.array;
break;
default:
break;
}
unsigned int* surfObj;
hipMalloc((void **) &surfObj, sizeof(hipArray));
hipMemcpy(surfObj, (void *)pResDesc->res.array.array, sizeof(hipArray), hipMemcpyHostToDevice);
*pSurfObject = (hipSurfaceObject_t) surfObj;
hipMalloc((void**)&surfObj, sizeof(hipArray));
hipMemcpy(surfObj, (void*)pResDesc->res.array.array, sizeof(hipArray),
hipMemcpyHostToDevice);
*pSurfObject = (hipSurfaceObject_t)surfObj;
surfaceHash[*pSurfObject] = pSurface;
}
return ihipLogStatus(hip_status);
}
hipError_t hipDestroySurfaceObject(hipSurfaceObject_t surfaceObject)
{
hipError_t hipDestroySurfaceObject(hipSurfaceObject_t surfaceObject) {
HIP_INIT_API(surfaceObject);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
+312 -334
View File
@@ -14,11 +14,8 @@
static std::map<hipTextureObject_t, hipTexture*> textureHash;
void saveTextureInfo(const hipTexture* pTexture,
const hipResourceDesc* pResDesc,
const hipTextureDesc* pTexDesc,
const hipResourceViewDesc* pResViewDesc)
{
void saveTextureInfo(const hipTexture* pTexture, const hipResourceDesc* pResDesc,
const hipTextureDesc* pTexDesc, const hipResourceViewDesc* pResViewDesc) {
if (pResDesc != nullptr) {
memcpy((void*)&(pTexture->resDesc), (void*)pResDesc, sizeof(hipResourceDesc));
}
@@ -32,12 +29,10 @@ void saveTextureInfo(const hipTexture* pTexture,
}
}
void getDrvChannelOrderAndType(const enum hipArray_Format Format,
unsigned int NumChannels,
hsa_ext_image_channel_order_t* channelOrder,
hsa_ext_image_channel_type_t* channelType)
{
switch(Format) {
void getDrvChannelOrderAndType(const enum hipArray_Format Format, unsigned int NumChannels,
hsa_ext_image_channel_order_t* channelOrder,
hsa_ext_image_channel_type_t* channelType) {
switch (Format) {
case HIP_AD_FORMAT_UNSIGNED_INT8:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT8;
break;
@@ -74,11 +69,9 @@ void getDrvChannelOrderAndType(const enum hipArray_Format Format,
*channelOrder = HSA_EXT_IMAGE_CHANNEL_ORDER_R;
}
}
void getChannelOrderAndType(const hipChannelFormatDesc& desc,
enum hipTextureReadMode readMode,
void getChannelOrderAndType(const hipChannelFormatDesc& desc, enum hipTextureReadMode readMode,
hsa_ext_image_channel_order_t* channelOrder,
hsa_ext_image_channel_type_t* channelType)
{
hsa_ext_image_channel_type_t* channelType) {
if (desc.x != 0 && desc.y != 0 && desc.z != 0 && desc.w != 0) {
*channelOrder = HSA_EXT_IMAGE_CHANNEL_ORDER_RGBA;
} else if (desc.x != 0 && desc.y != 0 && desc.z != 0 && desc.w == 0) {
@@ -91,65 +84,67 @@ void getChannelOrderAndType(const hipChannelFormatDesc& desc,
}
switch (desc.f) {
case hipChannelFormatKindUnsigned:
switch(desc.x) {
case 32:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT32;
case hipChannelFormatKindUnsigned:
switch (desc.x) {
case 32:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT32;
break;
case 16:
*channelType = readMode == hipReadModeNormalizedFloat
? HSA_EXT_IMAGE_CHANNEL_TYPE_UNORM_INT16
: HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT16;
break;
case 8:
*channelType = readMode == hipReadModeNormalizedFloat
? HSA_EXT_IMAGE_CHANNEL_TYPE_UNORM_INT8
: HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT8;
break;
default:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT32;
}
break;
case 16:
*channelType = readMode == hipReadModeNormalizedFloat ? HSA_EXT_IMAGE_CHANNEL_TYPE_UNORM_INT16 :
HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT16;
case hipChannelFormatKindSigned:
switch (desc.x) {
case 32:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_SIGNED_INT32;
break;
case 16:
*channelType = readMode == hipReadModeNormalizedFloat
? HSA_EXT_IMAGE_CHANNEL_TYPE_SNORM_INT16
: HSA_EXT_IMAGE_CHANNEL_TYPE_SIGNED_INT16;
break;
case 8:
*channelType = readMode == hipReadModeNormalizedFloat
? HSA_EXT_IMAGE_CHANNEL_TYPE_SNORM_INT8
: HSA_EXT_IMAGE_CHANNEL_TYPE_SIGNED_INT8;
break;
default:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_SIGNED_INT32;
}
break;
case 8:
*channelType = readMode == hipReadModeNormalizedFloat ? HSA_EXT_IMAGE_CHANNEL_TYPE_UNORM_INT8 :
HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT8;
case hipChannelFormatKindFloat:
switch (desc.x) {
case 32:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_FLOAT;
break;
case 16:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_HALF_FLOAT;
break;
case 8:
break;
default:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_FLOAT;
}
break;
case hipChannelFormatKindNone:
default:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_UNSIGNED_INT32;
}
break;
case hipChannelFormatKindSigned:
switch(desc.x) {
case 32:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_SIGNED_INT32;
break;
case 16:
*channelType = readMode == hipReadModeNormalizedFloat ? HSA_EXT_IMAGE_CHANNEL_TYPE_SNORM_INT16 :
HSA_EXT_IMAGE_CHANNEL_TYPE_SIGNED_INT16;
break;
case 8:
*channelType = readMode == hipReadModeNormalizedFloat ? HSA_EXT_IMAGE_CHANNEL_TYPE_SNORM_INT8 :
HSA_EXT_IMAGE_CHANNEL_TYPE_SIGNED_INT8;
break;
default:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_SIGNED_INT32;
}
break;
case hipChannelFormatKindFloat:
switch(desc.x) {
case 32:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_FLOAT;
break;
case 16:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_HALF_FLOAT;
break;
case 8:
break;
default:
*channelType = HSA_EXT_IMAGE_CHANNEL_TYPE_FLOAT;
}
break;
case hipChannelFormatKindNone:
default:
break;
}
}
void fillSamplerDescriptor(hsa_ext_sampler_descriptor_t& samplerDescriptor,
enum hipTextureAddressMode addressMode,
enum hipTextureFilterMode filterMode,
int normalizedCoords)
{
enum hipTextureFilterMode filterMode, int normalizedCoords) {
if (normalizedCoords) {
samplerDescriptor.coordinate_mode = HSA_EXT_SAMPLER_COORDINATE_MODE_NORMALIZED;
} else {
@@ -157,42 +152,42 @@ void fillSamplerDescriptor(hsa_ext_sampler_descriptor_t& samplerDescriptor,
}
switch (filterMode) {
case hipFilterModePoint:
samplerDescriptor.filter_mode = HSA_EXT_SAMPLER_FILTER_MODE_NEAREST;
break;
case hipFilterModeLinear:
samplerDescriptor.filter_mode = HSA_EXT_SAMPLER_FILTER_MODE_LINEAR;
break;
case hipFilterModePoint:
samplerDescriptor.filter_mode = HSA_EXT_SAMPLER_FILTER_MODE_NEAREST;
break;
case hipFilterModeLinear:
samplerDescriptor.filter_mode = HSA_EXT_SAMPLER_FILTER_MODE_LINEAR;
break;
}
switch (addressMode) {
case hipAddressModeWrap:
samplerDescriptor.address_mode = HSA_EXT_SAMPLER_ADDRESSING_MODE_REPEAT;
break;
case hipAddressModeClamp:
samplerDescriptor.address_mode = HSA_EXT_SAMPLER_ADDRESSING_MODE_CLAMP_TO_EDGE;
break;
case hipAddressModeMirror:
samplerDescriptor.address_mode = HSA_EXT_SAMPLER_ADDRESSING_MODE_MIRRORED_REPEAT;
break;
case hipAddressModeBorder:
samplerDescriptor.address_mode = HSA_EXT_SAMPLER_ADDRESSING_MODE_CLAMP_TO_BORDER;
break;
case hipAddressModeWrap:
samplerDescriptor.address_mode = HSA_EXT_SAMPLER_ADDRESSING_MODE_REPEAT;
break;
case hipAddressModeClamp:
samplerDescriptor.address_mode = HSA_EXT_SAMPLER_ADDRESSING_MODE_CLAMP_TO_EDGE;
break;
case hipAddressModeMirror:
samplerDescriptor.address_mode = HSA_EXT_SAMPLER_ADDRESSING_MODE_MIRRORED_REPEAT;
break;
case hipAddressModeBorder:
samplerDescriptor.address_mode = HSA_EXT_SAMPLER_ADDRESSING_MODE_CLAMP_TO_BORDER;
break;
}
}
bool getHipTextureObject(hipTextureObject_t* pTexObject,
hsa_ext_image_t& image,
hsa_ext_sampler_t sampler)
{
bool getHipTextureObject(hipTextureObject_t* pTexObject, hsa_ext_image_t& image,
hsa_ext_sampler_t sampler) {
unsigned int* texSRD;
hipMalloc((void **) &texSRD, HIP_TEXTURE_OBJECT_SIZE_DWORD * 4);
hipMemcpy(texSRD, (void *)image.handle, HIP_IMAGE_OBJECT_SIZE_DWORD * 4, hipMemcpyDeviceToDevice);
hipMemcpy(texSRD + HIP_SAMPLER_OBJECT_OFFSET_DWORD, (void *)sampler.handle, HIP_SAMPLER_OBJECT_SIZE_DWORD * 4, hipMemcpyDeviceToDevice);
*pTexObject = (hipTextureObject_t) texSRD;
hipMalloc((void**)&texSRD, HIP_TEXTURE_OBJECT_SIZE_DWORD * 4);
hipMemcpy(texSRD, (void*)image.handle, HIP_IMAGE_OBJECT_SIZE_DWORD * 4,
hipMemcpyDeviceToDevice);
hipMemcpy(texSRD + HIP_SAMPLER_OBJECT_OFFSET_DWORD, (void*)sampler.handle,
HIP_SAMPLER_OBJECT_SIZE_DWORD * 4, hipMemcpyDeviceToDevice);
*pTexObject = (hipTextureObject_t)texSRD;
#ifdef DEBUG
unsigned int* srd = (unsigned int*) malloc(HIP_TEXTURE_OBJECT_SIZE_DWORD * 4);
unsigned int* srd = (unsigned int*)malloc(HIP_TEXTURE_OBJECT_SIZE_DWORD * 4);
hipMemcpy(srd, texSRD, HIP_TEXTURE_OBJECT_SIZE_DWORD * 4, hipMemcpyDeviceToHost);
printf("New SRD: \n");
for (int i = 0; i < HIP_TEXTURE_OBJECT_SIZE_DWORD; i++) {
@@ -204,22 +199,20 @@ bool getHipTextureObject(hipTextureObject_t* pTexObject,
}
// Texture Object APIs
hipError_t hipCreateTextureObject(hipTextureObject_t* pTexObject,
const hipResourceDesc* pResDesc,
hipError_t hipCreateTextureObject(hipTextureObject_t* pTexObject, const hipResourceDesc* pResDesc,
const hipTextureDesc* pTexDesc,
const hipResourceViewDesc* pResViewDesc)
{
const hipResourceViewDesc* pResViewDesc) {
HIP_INIT_API(pTexObject, pResDesc, pTexDesc, pResViewDesc);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
hc::accelerator acc = ctx->getDevice()->_acc;
auto device = ctx->getWriteableDevice();
hsa_agent_t* agent =static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hsa_agent_t* agent = static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hipTexture* pTexture = (hipTexture*) malloc(sizeof(hipTexture));
hipTexture* pTexture = (hipTexture*)malloc(sizeof(hipTexture));
if (pTexture != nullptr) {
memset(pTexture, 0, sizeof(hipTexture));
saveTextureInfo(pTexture, pResDesc, pTexDesc, pResViewDesc);
@@ -231,72 +224,81 @@ hipError_t hipCreateTextureObject(hipTextureObject_t* pTexObject,
void* devPtr = nullptr;
switch (pResDesc->resType) {
case hipResourceTypeArray:
devPtr = pResDesc->res.array.array->data;
imageDescriptor.width = pResDesc->res.array.array->width;
imageDescriptor.height = pResDesc->res.array.array->height;
switch (pResDesc->res.array.array->type) {
case hipArrayLayered:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2DA;
imageDescriptor.depth = 0;
imageDescriptor.array_size = pResDesc->res.array.array->depth;
case hipResourceTypeArray:
devPtr = pResDesc->res.array.array->data;
imageDescriptor.width = pResDesc->res.array.array->width;
imageDescriptor.height = pResDesc->res.array.array->height;
switch (pResDesc->res.array.array->type) {
case hipArrayLayered:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2DA;
imageDescriptor.depth = 0;
imageDescriptor.array_size = pResDesc->res.array.array->depth;
break;
case hipArrayCubemap:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_3D;
imageDescriptor.depth = pResDesc->res.array.array->depth;
imageDescriptor.array_size = 0;
break;
case hipArraySurfaceLoadStore:
case hipArrayTextureGather:
case hipArrayDefault:
default:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2D;
imageDescriptor.depth = 0;
imageDescriptor.array_size = 0;
break;
}
getChannelOrderAndType(pResDesc->res.array.array->desc, pTexDesc->readMode,
&channelOrder, &channelType);
break;
case hipArrayCubemap:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_3D;
imageDescriptor.depth = pResDesc->res.array.array->depth;
case hipResourceTypeMipmappedArray:
devPtr = pResDesc->res.mipmap.mipmap->data;
imageDescriptor.width = pResDesc->res.mipmap.mipmap->width;
imageDescriptor.height = pResDesc->res.mipmap.mipmap->height;
imageDescriptor.depth = pResDesc->res.mipmap.mipmap->depth;
imageDescriptor.array_size = 0;
break;
case hipArraySurfaceLoadStore:
case hipArrayTextureGather:
case hipArrayDefault:
default:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2D;
getChannelOrderAndType(pResDesc->res.mipmap.mipmap->desc, pTexDesc->readMode,
&channelOrder, &channelType);
break;
case hipResourceTypeLinear:
devPtr = pResDesc->res.linear.devPtr;
imageDescriptor.width = pResDesc->res.linear.sizeInBytes;
imageDescriptor.height = 1;
imageDescriptor.depth = 0;
imageDescriptor.array_size = 0;
imageDescriptor.geometry =
HSA_EXT_IMAGE_GEOMETRY_1D; // ? HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR
getChannelOrderAndType(pResDesc->res.linear.desc, pTexDesc->readMode, &channelOrder,
&channelType);
break;
case hipResourceTypePitch2D:
devPtr = pResDesc->res.pitch2D.devPtr;
imageDescriptor.width = pResDesc->res.pitch2D.width;
imageDescriptor.height = pResDesc->res.pitch2D.height;
imageDescriptor.depth = 0;
imageDescriptor.array_size = 0;
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2D;
getChannelOrderAndType(pResDesc->res.pitch2D.desc, pTexDesc->readMode,
&channelOrder, &channelType);
break;
default:
break;
}
getChannelOrderAndType(pResDesc->res.array.array->desc, pTexDesc->readMode, &channelOrder, &channelType);
break;
case hipResourceTypeMipmappedArray:
devPtr = pResDesc->res.mipmap.mipmap->data;
imageDescriptor.width = pResDesc->res.mipmap.mipmap->width;
imageDescriptor.height = pResDesc->res.mipmap.mipmap->height;
imageDescriptor.depth = pResDesc->res.mipmap.mipmap->depth;
imageDescriptor.array_size = 0;
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2D;
getChannelOrderAndType(pResDesc->res.mipmap.mipmap->desc, pTexDesc->readMode, &channelOrder, &channelType);
break;
case hipResourceTypeLinear:
devPtr = pResDesc->res.linear.devPtr;
imageDescriptor.width = pResDesc->res.linear.sizeInBytes;
imageDescriptor.height = 1;
imageDescriptor.depth = 0;
imageDescriptor.array_size = 0;
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_1D; // ? HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR
getChannelOrderAndType(pResDesc->res.linear.desc, pTexDesc->readMode, &channelOrder, &channelType);
break;
case hipResourceTypePitch2D:
devPtr = pResDesc->res.pitch2D.devPtr;
imageDescriptor.width = pResDesc->res.pitch2D.width;
imageDescriptor.height = pResDesc->res.pitch2D.height;
imageDescriptor.depth = 0;
imageDescriptor.array_size = 0;
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2D;
getChannelOrderAndType(pResDesc->res.pitch2D.desc, pTexDesc->readMode, &channelOrder, &channelType);
break;
default:
break;
}
imageDescriptor.format.channel_order = channelOrder;
imageDescriptor.format.channel_type = channelType;
hsa_ext_sampler_descriptor_t samplerDescriptor;
fillSamplerDescriptor(samplerDescriptor, pTexDesc->addressMode[0], pTexDesc->filterMode, pTexDesc->normalizedCoords);
fillSamplerDescriptor(samplerDescriptor, pTexDesc->addressMode[0], pTexDesc->filterMode,
pTexDesc->normalizedCoords);
hsa_access_permission_t permission = HSA_ACCESS_PERMISSION_RW;
if (HSA_STATUS_SUCCESS != hsa_ext_image_create_with_layout(*agent, &imageDescriptor, devPtr, permission, HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR, 0, 0, &(pTexture->image)) ||
HSA_STATUS_SUCCESS != hsa_ext_sampler_create(*agent, &samplerDescriptor, &(pTexture->sampler))) {
if (HSA_STATUS_SUCCESS != hsa_ext_image_create_with_layout(
*agent, &imageDescriptor, devPtr, permission,
HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR, 0, 0, &(pTexture->image)) ||
HSA_STATUS_SUCCESS !=
hsa_ext_sampler_create(*agent, &samplerDescriptor, &(pTexture->sampler))) {
return ihipLogStatus(hipErrorRuntimeOther);
}
@@ -308,18 +310,17 @@ hipError_t hipCreateTextureObject(hipTextureObject_t* pTexObject,
return ihipLogStatus(hip_status);
}
hipError_t hipDestroyTextureObject(hipTextureObject_t textureObject)
{
hipError_t hipDestroyTextureObject(hipTextureObject_t textureObject) {
HIP_INIT_API(textureObject);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
hc::accelerator acc = ctx->getDevice()->_acc;
auto device = ctx->getWriteableDevice();
hsa_agent_t* agent =static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hsa_agent_t* agent = static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hipTexture* pTexture = textureHash[textureObject];
if (pTexture != nullptr) {
@@ -332,10 +333,10 @@ hipError_t hipDestroyTextureObject(hipTextureObject_t textureObject)
return ihipLogStatus(hip_status);
}
hipError_t hipGetTextureObjectResourceDesc(hipResourceDesc* pResDesc, hipTextureObject_t textureObject)
{
hipError_t hipGetTextureObjectResourceDesc(hipResourceDesc* pResDesc,
hipTextureObject_t textureObject) {
HIP_INIT_API(pResDesc, textureObject);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
@@ -347,26 +348,27 @@ hipError_t hipGetTextureObjectResourceDesc(hipResourceDesc* pResDesc, hipTexture
return ihipLogStatus(hip_status);
}
hipError_t hipGetTextureObjectResourceViewDesc(hipResourceViewDesc* pResViewDesc, hipTextureObject_t textureObject)
{
hipError_t hipGetTextureObjectResourceViewDesc(hipResourceViewDesc* pResViewDesc,
hipTextureObject_t textureObject) {
HIP_INIT_API(pResViewDesc, textureObject);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
hipTexture* pTexture = textureHash[textureObject];
if (pTexture != nullptr && pResViewDesc != nullptr) {
memcpy((void*)pResViewDesc, (void*)&(pTexture->resViewDesc), sizeof(hipResourceViewDesc));
memcpy((void*)pResViewDesc, (void*)&(pTexture->resViewDesc),
sizeof(hipResourceViewDesc));
}
}
return ihipLogStatus(hip_status);
}
hipError_t hipGetTextureObjectTextureDesc(hipTextureDesc* pTexDesc, hipTextureObject_t textureObject)
{
hipError_t hipGetTextureObjectTextureDesc(hipTextureDesc* pTexDesc,
hipTextureObject_t textureObject) {
HIP_INIT_API(pTexDesc, textureObject);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
@@ -379,14 +381,10 @@ hipError_t hipGetTextureObjectTextureDesc(hipTextureDesc* pTexDesc, hipTextureOb
}
// Texture Reference APIs
hipError_t ihipBindTextureImpl(int dim,
enum hipTextureReadMode readMode,
size_t *offset,
const void *devPtr,
const struct hipChannelFormatDesc* desc,
size_t size, textureReference* tex )
{
hipError_t hip_status = hipSuccess;
hipError_t ihipBindTextureImpl(int dim, enum hipTextureReadMode readMode, size_t* offset,
const void* devPtr, const struct hipChannelFormatDesc* desc,
size_t size, textureReference* tex) {
hipError_t hip_status = hipSuccess;
enum hipTextureAddressMode addressMode = tex->addressMode[0];
enum hipTextureFilterMode filterMode = tex->filterMode;
int normalizedCoords = tex->normalized;
@@ -396,9 +394,9 @@ hipError_t ihipBindTextureImpl(int dim,
hc::accelerator acc = ctx->getDevice()->_acc;
auto device = ctx->getWriteableDevice();
hsa_agent_t* agent =static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hsa_agent_t* agent = static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hipTexture* pTexture = (hipTexture*) malloc(sizeof(hipTexture));
hipTexture* pTexture = (hipTexture*)malloc(sizeof(hipTexture));
if (pTexture != nullptr) {
memset(pTexture, 0, sizeof(hipTexture));
}
@@ -415,11 +413,11 @@ hipError_t ihipBindTextureImpl(int dim,
hsa_ext_image_channel_order_t channelOrder;
hsa_ext_image_channel_type_t channelType;
if(NULL == desc) {
getDrvChannelOrderAndType(tex->format, tex->numChannels, &channelOrder, &channelType);
} else {
getChannelOrderAndType(*desc, readMode, &channelOrder, &channelType);
}
if (NULL == desc) {
getDrvChannelOrderAndType(tex->format, tex->numChannels, &channelOrder, &channelType);
} else {
getChannelOrderAndType(*desc, readMode, &channelOrder, &channelType);
}
imageDescriptor.format.channel_order = channelOrder;
imageDescriptor.format.channel_type = channelType;
@@ -428,8 +426,11 @@ hipError_t ihipBindTextureImpl(int dim,
hsa_access_permission_t permission = HSA_ACCESS_PERMISSION_RW;
if (HSA_STATUS_SUCCESS != hsa_ext_image_create_with_layout(*agent, &imageDescriptor, devPtr, permission, HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR, 0, 0, &(pTexture->image)) ||
HSA_STATUS_SUCCESS != hsa_ext_sampler_create(*agent, &samplerDescriptor, &(pTexture->sampler))) {
if (HSA_STATUS_SUCCESS != hsa_ext_image_create_with_layout(
*agent, &imageDescriptor, devPtr, permission,
HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR, 0, 0, &(pTexture->image)) ||
HSA_STATUS_SUCCESS !=
hsa_ext_sampler_create(*agent, &samplerDescriptor, &(pTexture->sampler))) {
return hipErrorRuntimeOther;
}
getHipTextureObject(&textureObject, pTexture->image, pTexture->sampler);
@@ -439,42 +440,32 @@ hipError_t ihipBindTextureImpl(int dim,
return hip_status;
}
hipError_t hipBindTexture(size_t* offset,
textureReference* tex,
const void* devPtr,
const hipChannelFormatDesc* desc,
size_t size)
{
HIP_INIT_API(offset, tex, devPtr, desc, size);
hipError_t hip_status = hipSuccess;
hipError_t hipBindTexture(size_t* offset, textureReference* tex, const void* devPtr,
const hipChannelFormatDesc* desc, size_t size) {
HIP_INIT_API(offset, tex, devPtr, desc, size);
hipError_t hip_status = hipSuccess;
// TODO: hipReadModeElementType is default.
hip_status = ihipBindTextureImpl(hipTextureType1D, hipReadModeElementType,
offset, devPtr, desc, size, tex);
hip_status = ihipBindTextureImpl(hipTextureType1D, hipReadModeElementType, offset, devPtr, desc,
size, tex);
return ihipLogStatus(hip_status);
}
hipError_t ihipBindTexture2DImpl(int dim,
enum hipTextureReadMode readMode,
size_t *offset,
const void *devPtr,
const struct hipChannelFormatDesc* desc,
size_t width,
size_t height,
textureReference* tex)
{
hipError_t hip_status = hipSuccess;
hipError_t ihipBindTexture2DImpl(int dim, enum hipTextureReadMode readMode, size_t* offset,
const void* devPtr, const struct hipChannelFormatDesc* desc,
size_t width, size_t height, textureReference* tex) {
hipError_t hip_status = hipSuccess;
enum hipTextureAddressMode addressMode = tex->addressMode[0];
enum hipTextureFilterMode filterMode = tex->filterMode;
int normalizedCoords = tex->normalized;
enum hipTextureFilterMode filterMode = tex->filterMode;
int normalizedCoords = tex->normalized;
hipTextureObject_t& textureObject = tex->textureObject;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
hc::accelerator acc = ctx->getDevice()->_acc;
auto device = ctx->getWriteableDevice();
hsa_agent_t* agent =static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hsa_agent_t* agent = static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hipTexture* pTexture = (hipTexture*) malloc(sizeof(hipTexture));
hipTexture* pTexture = (hipTexture*)malloc(sizeof(hipTexture));
if (pTexture != nullptr) {
memset(pTexture, 0, sizeof(hipTexture));
}
@@ -492,11 +483,11 @@ hipError_t ihipBindTexture2DImpl(int dim,
hsa_ext_image_channel_order_t channelOrder;
hsa_ext_image_channel_type_t channelType;
if(NULL == desc) {
getDrvChannelOrderAndType(tex->format, tex->numChannels, &channelOrder, &channelType);
} else {
getChannelOrderAndType(*desc, readMode, &channelOrder, &channelType);
}
if (NULL == desc) {
getDrvChannelOrderAndType(tex->format, tex->numChannels, &channelOrder, &channelType);
} else {
getChannelOrderAndType(*desc, readMode, &channelOrder, &channelType);
}
imageDescriptor.format.channel_order = channelOrder;
imageDescriptor.format.channel_type = channelType;
@@ -505,8 +496,11 @@ hipError_t ihipBindTexture2DImpl(int dim,
hsa_access_permission_t permission = HSA_ACCESS_PERMISSION_RW;
if (HSA_STATUS_SUCCESS != hsa_ext_image_create_with_layout(*agent, &imageDescriptor, devPtr, permission, HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR, 0, 0, &(pTexture->image)) ||
HSA_STATUS_SUCCESS != hsa_ext_sampler_create(*agent, &samplerDescriptor, &(pTexture->sampler))) {
if (HSA_STATUS_SUCCESS != hsa_ext_image_create_with_layout(
*agent, &imageDescriptor, devPtr, permission,
HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR, 0, 0, &(pTexture->image)) ||
HSA_STATUS_SUCCESS !=
hsa_ext_sampler_create(*agent, &samplerDescriptor, &(pTexture->sampler))) {
return hipErrorRuntimeOther;
}
getHipTextureObject(&textureObject, pTexture->image, pTexture->sampler);
@@ -516,30 +510,23 @@ hipError_t ihipBindTexture2DImpl(int dim,
return hip_status;
}
hipError_t hipBindTexture2D(size_t* offset,
textureReference* tex,
const void* devPtr,
const hipChannelFormatDesc* desc,
size_t width,
size_t height,
size_t pitch)
{
HIP_INIT_API(offset, tex, devPtr, desc, width, height, pitch);
hipError_t hip_status = hipSuccess;
hip_status = ihipBindTexture2DImpl(hipTextureType2D, hipReadModeElementType,
offset, devPtr, desc, width, height, tex);
hipError_t hipBindTexture2D(size_t* offset, textureReference* tex, const void* devPtr,
const hipChannelFormatDesc* desc, size_t width, size_t height,
size_t pitch) {
HIP_INIT_API(offset, tex, devPtr, desc, width, height, pitch);
hipError_t hip_status = hipSuccess;
hip_status = ihipBindTexture2DImpl(hipTextureType2D, hipReadModeElementType, offset, devPtr,
desc, width, height, tex);
return ihipLogStatus(hip_status);
}
hipError_t ihipBindTextureToArrayImpl(int dim,
enum hipTextureReadMode readMode,
hipError_t ihipBindTextureToArrayImpl(int dim, enum hipTextureReadMode readMode,
hipArray_const_t array,
const struct hipChannelFormatDesc& desc,
textureReference* tex)
{
hipError_t hip_status = hipSuccess;
textureReference* tex) {
hipError_t hip_status = hipSuccess;
enum hipTextureAddressMode addressMode = tex->addressMode[0];
enum hipTextureFilterMode filterMode = tex->filterMode;
enum hipTextureFilterMode filterMode = tex->filterMode;
int normalizedCoords = tex->normalized;
hipTextureObject_t& textureObject = tex->textureObject;
auto ctx = ihipGetTlsDefaultCtx();
@@ -547,9 +534,9 @@ hipError_t ihipBindTextureToArrayImpl(int dim,
hc::accelerator acc = ctx->getDevice()->_acc;
auto device = ctx->getWriteableDevice();
hsa_agent_t* agent =static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hsa_agent_t* agent = static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hipTexture* pTexture = (hipTexture*) malloc(sizeof(hipTexture));
hipTexture* pTexture = (hipTexture*)malloc(sizeof(hipTexture));
if (pTexture != nullptr) {
memset(pTexture, 0, sizeof(hipTexture));
}
@@ -562,41 +549,42 @@ hipError_t ihipBindTextureToArrayImpl(int dim,
imageDescriptor.array_size = 0;
switch (dim) {
case hipTextureType1D:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_1D;
imageDescriptor.height = 1;
imageDescriptor.depth = 1;
break;
case hipTextureType2D:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2D;
imageDescriptor.depth = 1;
break;
case hipTextureType3D:
case hipTextureTypeCubemap:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_3D;
break;
case hipTextureType1DLayered:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_1DA;
imageDescriptor.height = 1;
imageDescriptor.array_size = array->height;
break;
case hipTextureType2DLayered:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2DA;
imageDescriptor.depth = 1;
imageDescriptor.array_size = array->depth;
break;
case hipTextureTypeCubemapLayered:
default:
break;
case hipTextureType1D:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_1D;
imageDescriptor.height = 1;
imageDescriptor.depth = 1;
break;
case hipTextureType2D:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2D;
imageDescriptor.depth = 1;
break;
case hipTextureType3D:
case hipTextureTypeCubemap:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_3D;
break;
case hipTextureType1DLayered:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_1DA;
imageDescriptor.height = 1;
imageDescriptor.array_size = array->height;
break;
case hipTextureType2DLayered:
imageDescriptor.geometry = HSA_EXT_IMAGE_GEOMETRY_2DA;
imageDescriptor.depth = 1;
imageDescriptor.array_size = array->depth;
break;
case hipTextureTypeCubemapLayered:
default:
break;
}
hsa_ext_image_channel_order_t channelOrder;
hsa_ext_image_channel_type_t channelType;
if(array->isDrv) {
getDrvChannelOrderAndType(array->drvDesc.format, array->drvDesc.numChannels, &channelOrder, &channelType);
} else {
if (array->isDrv) {
getDrvChannelOrderAndType(array->drvDesc.format, array->drvDesc.numChannels,
&channelOrder, &channelType);
} else {
getChannelOrderAndType(desc, readMode, &channelOrder, &channelType);
}
}
imageDescriptor.format.channel_order = channelOrder;
imageDescriptor.format.channel_type = channelType;
@@ -605,8 +593,11 @@ hipError_t ihipBindTextureToArrayImpl(int dim,
hsa_access_permission_t permission = HSA_ACCESS_PERMISSION_RW;
if (HSA_STATUS_SUCCESS != hsa_ext_image_create_with_layout(*agent, &imageDescriptor, array->data, permission, HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR, 0, 0, &(pTexture->image)) ||
HSA_STATUS_SUCCESS != hsa_ext_sampler_create(*agent, &samplerDescriptor, &(pTexture->sampler))) {
if (HSA_STATUS_SUCCESS != hsa_ext_image_create_with_layout(
*agent, &imageDescriptor, array->data, permission,
HSA_EXT_IMAGE_DATA_LAYOUT_LINEAR, 0, 0, &(pTexture->image)) ||
HSA_STATUS_SUCCESS !=
hsa_ext_sampler_create(*agent, &samplerDescriptor, &(pTexture->sampler))) {
return hipErrorRuntimeOther;
}
getHipTextureObject(&textureObject, pTexture->image, pTexture->sampler);
@@ -616,39 +607,35 @@ hipError_t ihipBindTextureToArrayImpl(int dim,
return hip_status;
}
hipError_t hipBindTextureToArray(textureReference* tex,
hipArray_const_t array,
const hipChannelFormatDesc* desc)
{
HIP_INIT_API(tex, array, desc);
hipError_t hip_status = hipSuccess;
hipError_t hipBindTextureToArray(textureReference* tex, hipArray_const_t array,
const hipChannelFormatDesc* desc) {
HIP_INIT_API(tex, array, desc);
hipError_t hip_status = hipSuccess;
// TODO: hipReadModeElementType is default.
hip_status = ihipBindTextureToArrayImpl(array->textureType, hipReadModeElementType,
array, *desc, tex);
hip_status =
ihipBindTextureToArrayImpl(array->textureType, hipReadModeElementType, array, *desc, tex);
return ihipLogStatus(hip_status);
}
hipError_t hipBindTextureToMipmappedArray(textureReference* tex,
hipMipmappedArray_const_t mipmappedArray,
const hipChannelFormatDesc* desc)
{
HIP_INIT_API(tex, mipmappedArray, desc);
hipError_t hip_status = hipSuccess;
return ihipLogStatus(hip_status);
const hipChannelFormatDesc* desc) {
HIP_INIT_API(tex, mipmappedArray, desc);
hipError_t hip_status = hipSuccess;
return ihipLogStatus(hip_status);
}
hipError_t ihipUnbindTextureImpl(const hipTextureObject_t& textureObject)
{
hipError_t hip_status = hipSuccess;
hipError_t ihipUnbindTextureImpl(const hipTextureObject_t& textureObject) {
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
hc::accelerator acc = ctx->getDevice()->_acc;
auto device = ctx->getWriteableDevice();
hsa_agent_t* agent =static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hsa_agent_t* agent = static_cast<hsa_agent_t*>(acc.get_hsa_agent());
hipTexture* pTexture = textureHash[textureObject];
hipTexture* pTexture = textureHash[textureObject];
if (pTexture != nullptr) {
hsa_ext_image_destroy(*agent, pTexture->image);
hsa_ext_sampler_destroy(*agent, pTexture->sampler);
@@ -660,18 +647,16 @@ hipError_t ihipUnbindTextureImpl(const hipTextureObject_t& textureObject)
return hip_status;
}
hipError_t hipUnbindTexture(const textureReference* tex)
{
HIP_INIT_API(tex);
hipError_t hip_status = hipSuccess;
hipError_t hipUnbindTexture(const textureReference* tex) {
HIP_INIT_API(tex);
hipError_t hip_status = hipSuccess;
hip_status = ihipUnbindTextureImpl(tex->textureObject);
return ihipLogStatus(hip_status);
}
hipError_t hipGetChannelDesc(hipChannelFormatDesc* desc, hipArray_const_t array)
{
hipError_t hipGetChannelDesc(hipChannelFormatDesc* desc, hipArray_const_t array) {
HIP_INIT_API(desc, array);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
@@ -680,11 +665,10 @@ hipError_t hipGetChannelDesc(hipChannelFormatDesc* desc, hipArray_const_t array)
return ihipLogStatus(hip_status);
}
hipError_t hipGetTextureAlignmentOffset(size_t* offset, const textureReference* tex)
{
hipError_t hipGetTextureAlignmentOffset(size_t* offset, const textureReference* tex) {
HIP_INIT_API(offset, tex);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
@@ -692,11 +676,10 @@ hipError_t hipGetTextureAlignmentOffset(size_t* offset, const textureReference*
return ihipLogStatus(hip_status);
}
hipError_t hipGetTextureReference(const textureReference** tex, const void* symbol)
{
hipError_t hipGetTextureReference(const textureReference** tex, const void* symbol) {
HIP_INIT_API(tex, symbol);
hipError_t hip_status = hipSuccess;
hipError_t hip_status = hipSuccess;
auto ctx = ihipGetTlsDefaultCtx();
if (ctx) {
@@ -704,67 +687,62 @@ hipError_t hipGetTextureReference(const textureReference** tex, const void* symb
return ihipLogStatus(hip_status);
}
hipError_t hipTexRefSetFormat (textureReference* tex, hipArray_Format fmt, int NumPackedComponents )
{
HIP_INIT_API(tex, fmt, NumPackedComponents);
hipError_t hip_status = hipSuccess;
tex->format = fmt;
tex->numChannels = NumPackedComponents;
hipError_t hipTexRefSetFormat(textureReference* tex, hipArray_Format fmt, int NumPackedComponents) {
HIP_INIT_API(tex, fmt, NumPackedComponents);
hipError_t hip_status = hipSuccess;
tex->format = fmt;
tex->numChannels = NumPackedComponents;
return ihipLogStatus(hip_status);
}
hipError_t hipTexRefSetFlags ( textureReference* tex, unsigned int flags )
{
HIP_INIT_API(tex, flags);
hipError_t hip_status = hipSuccess;
tex->normalized = flags;
hipError_t hipTexRefSetFlags(textureReference* tex, unsigned int flags) {
HIP_INIT_API(tex, flags);
hipError_t hip_status = hipSuccess;
tex->normalized = flags;
return ihipLogStatus(hip_status);
}
hipError_t hipTexRefSetFilterMode ( textureReference* tex, hipTextureFilterMode fm )
{
HIP_INIT_API(tex, fm);
hipError_t hip_status = hipSuccess;
hipError_t hipTexRefSetFilterMode(textureReference* tex, hipTextureFilterMode fm) {
HIP_INIT_API(tex, fm);
hipError_t hip_status = hipSuccess;
tex->filterMode = fm;
return ihipLogStatus(hip_status);
}
hipError_t hipTexRefSetAddressMode ( textureReference* tex, int dim, hipTextureAddressMode am )
{
HIP_INIT_API(tex, dim, am);
hipError_t hip_status = hipSuccess;
tex->addressMode[dim] = am;
hipError_t hipTexRefSetAddressMode(textureReference* tex, int dim, hipTextureAddressMode am) {
HIP_INIT_API(tex, dim, am);
hipError_t hip_status = hipSuccess;
tex->addressMode[dim] = am;
return ihipLogStatus(hip_status);
}
hipError_t hipTexRefSetArray ( textureReference* tex, hipArray_const_t array, unsigned int flags )
{
HIP_INIT_API(tex, array, flags);
hipError_t hip_status = hipSuccess;
hipError_t hipTexRefSetArray(textureReference* tex, hipArray_const_t array, unsigned int flags) {
HIP_INIT_API(tex, array, flags);
hipError_t hip_status = hipSuccess;
hip_status = ihipBindTextureToArrayImpl(array->textureType, hipReadModeElementType,
array, array->desc,tex );
hip_status = ihipBindTextureToArrayImpl(array->textureType, hipReadModeElementType, array,
array->desc, tex);
return ihipLogStatus(hip_status);
}
hipError_t hipTexRefSetAddress( size_t* offset, textureReference* tex, hipDeviceptr_t devPtr, size_t size )
{
HIP_INIT_API(offset, tex, devPtr, size);
hipError_t hip_status = hipSuccess;
hipError_t hipTexRefSetAddress(size_t* offset, textureReference* tex, hipDeviceptr_t devPtr,
size_t size) {
HIP_INIT_API(offset, tex, devPtr, size);
hipError_t hip_status = hipSuccess;
// TODO: hipReadModeElementType is default.
hip_status = ihipBindTextureImpl(hipTextureType1D, hipReadModeElementType,
offset, devPtr, NULL, size, tex);
hip_status = ihipBindTextureImpl(hipTextureType1D, hipReadModeElementType, offset, devPtr, NULL,
size, tex);
return ihipLogStatus(hip_status);
}
hipError_t hipTexRefSetAddress2D( textureReference* tex, const HIP_ARRAY_DESCRIPTOR* desc, hipDeviceptr_t devPtr, size_t pitch )
{
HIP_INIT_API(tex, desc, devPtr, pitch);
size_t offset;
hipError_t hip_status = hipSuccess;
hipError_t hipTexRefSetAddress2D(textureReference* tex, const HIP_ARRAY_DESCRIPTOR* desc,
hipDeviceptr_t devPtr, size_t pitch) {
HIP_INIT_API(tex, desc, devPtr, pitch);
size_t offset;
hipError_t hip_status = hipSuccess;
// TODO: hipReadModeElementType is default.
hip_status = ihipBindTexture2DImpl(hipTextureType2D, hipReadModeElementType,
&offset, devPtr, NULL, desc->width, desc->height, tex);
hip_status = ihipBindTexture2DImpl(hipTextureType2D, hipReadModeElementType, &offset, devPtr,
NULL, desc->width, desc->height, tex);
return ihipLogStatus(hip_status);
}
+63 -96
View File
@@ -27,118 +27,85 @@ THE SOFTWARE.
#include <functional>
#include <string>
inline
constexpr
bool operator==(hsa_isa_t x, hsa_isa_t y)
{
return x.handle == y.handle;
inline constexpr bool operator==(hsa_isa_t x, hsa_isa_t y) { return x.handle == y.handle; }
namespace std {
template <>
struct hash<hsa_isa_t> {
size_t operator()(hsa_isa_t x) const { return hash<decltype(x.handle)>{}(x.handle); }
};
} // namespace std
namespace hip_impl {
inline void* address(hsa_executable_symbol_t x) {
void* r = nullptr;
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_VARIABLE_ADDRESS, &r);
return r;
}
namespace std
{
template<>
struct hash<hsa_isa_t> {
size_t operator()(hsa_isa_t x) const
{
return hash<decltype(x.handle)>{}(x.handle);
}
};
inline hsa_agent_t agent(hsa_executable_symbol_t x) {
hsa_agent_t r = {};
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_AGENT, &r);
return r;
}
namespace hip_impl
{
inline
void* address(hsa_executable_symbol_t x)
{
void* r = nullptr;
hsa_executable_symbol_get_info(
x, HSA_EXECUTABLE_SYMBOL_INFO_VARIABLE_ADDRESS, &r);
inline std::uint32_t group_size(hsa_executable_symbol_t x) {
std::uint32_t r = 0u;
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_KERNEL_GROUP_SEGMENT_SIZE, &r);
return r;
}
return r;
}
inline
hsa_agent_t agent(hsa_executable_symbol_t x)
{
hsa_agent_t r = {};
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_AGENT, &r);
inline hsa_isa_t isa(hsa_agent_t x) {
hsa_isa_t r = {};
hsa_agent_iterate_isas(x,
[](hsa_isa_t i, void* o) {
*static_cast<hsa_isa_t*>(o) = i; // Pick the first.
return r;
}
return HSA_STATUS_INFO_BREAK;
},
&r);
inline
std::uint32_t group_size(hsa_executable_symbol_t x)
{
std::uint32_t r = 0u;
hsa_executable_symbol_get_info(
x, HSA_EXECUTABLE_SYMBOL_INFO_KERNEL_GROUP_SEGMENT_SIZE, &r);
return r;
}
return r;
}
inline std::uint64_t kernel_object(hsa_executable_symbol_t x) {
std::uint64_t r = 0u;
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_KERNEL_OBJECT, &r);
inline
hsa_isa_t isa(hsa_agent_t x)
{
hsa_isa_t r = {};
hsa_agent_iterate_isas(x, [](hsa_isa_t i, void* o) {
*static_cast<hsa_isa_t*>(o) = i; // Pick the first.
return r;
}
return HSA_STATUS_INFO_BREAK;
}, &r);
inline std::string name(hsa_executable_symbol_t x) {
std::uint32_t sz = 0u;
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_NAME_LENGTH, &sz);
return r;
}
std::string r(sz, '\0');
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_NAME, &r.front());
inline
std::uint64_t kernel_object(hsa_executable_symbol_t x)
{
std::uint64_t r = 0u;
hsa_executable_symbol_get_info(
x, HSA_EXECUTABLE_SYMBOL_INFO_KERNEL_OBJECT, &r);
return r;
}
return r;
}
inline std::uint32_t private_size(hsa_executable_symbol_t x) {
std::uint32_t r = 0u;
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_KERNEL_PRIVATE_SEGMENT_SIZE, &r);
inline
std::string name(hsa_executable_symbol_t x)
{
std::uint32_t sz = 0u;
hsa_executable_symbol_get_info(
x, HSA_EXECUTABLE_SYMBOL_INFO_NAME_LENGTH, &sz);
return r;
}
std::string r(sz, '\0');
hsa_executable_symbol_get_info(
x, HSA_EXECUTABLE_SYMBOL_INFO_NAME, &r.front());
inline std::uint32_t size(hsa_executable_symbol_t x) {
std::uint32_t r = 0;
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_VARIABLE_SIZE, &r);
return r;
}
return r;
}
inline
std::uint32_t private_size(hsa_executable_symbol_t x)
{
std::uint32_t r = 0u;
hsa_executable_symbol_get_info(
x, HSA_EXECUTABLE_SYMBOL_INFO_KERNEL_PRIVATE_SEGMENT_SIZE, &r);
inline hsa_symbol_kind_t type(hsa_executable_symbol_t x) {
hsa_symbol_kind_t r = {};
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_TYPE, &r);
return r;
}
inline
std::uint32_t size(hsa_executable_symbol_t x)
{
std::uint32_t r = 0;
hsa_executable_symbol_get_info(
x, HSA_EXECUTABLE_SYMBOL_INFO_VARIABLE_SIZE, &r);
return r;
}
inline
hsa_symbol_kind_t type(hsa_executable_symbol_t x)
{
hsa_symbol_kind_t r = {};
hsa_executable_symbol_get_info(x, HSA_EXECUTABLE_SYMBOL_INFO_TYPE, &r);
return r;
}
}
return r;
}
} // namespace hip_impl
File diff suppressed because it is too large Load Diff
+320 -375
View File
@@ -30,76 +30,64 @@ using namespace ELFIO;
using namespace hip_impl;
using namespace std;
namespace
{
struct Symbol {
string name;
ELFIO::Elf64_Addr value = 0;
Elf_Xword size = 0;
Elf_Half sect_idx = 0;
uint8_t bind = 0;
uint8_t type = 0;
uint8_t other = 0;
};
namespace {
struct Symbol {
string name;
ELFIO::Elf64_Addr value = 0;
Elf_Xword size = 0;
Elf_Half sect_idx = 0;
uint8_t bind = 0;
uint8_t type = 0;
uint8_t other = 0;
};
inline
Symbol read_symbol(const symbol_section_accessor& section, unsigned int idx)
{
assert(idx < section.get_symbols_num());
inline Symbol read_symbol(const symbol_section_accessor& section, unsigned int idx) {
assert(idx < section.get_symbols_num());
Symbol r;
section.get_symbol(
idx, r.name, r.value, r.size, r.bind, r.type, r.sect_idx, r.other);
Symbol r;
section.get_symbol(idx, r.name, r.value, r.size, r.bind, r.type, r.sect_idx, r.other);
return r;
}
return r;
}
template<typename P>
inline
section* find_section_if(elfio& reader, P p)
{
const auto it = find_if(
reader.sections.begin(), reader.sections.end(), move(p));
template <typename P>
inline section* find_section_if(elfio& reader, P p) {
const auto it = find_if(reader.sections.begin(), reader.sections.end(), move(p));
return it != reader.sections.end() ? *it : nullptr;
}
return it != reader.sections.end() ? *it : nullptr;
}
vector<string> copy_names_of_undefined_symbols(
const symbol_section_accessor& section)
{
vector<string> r;
vector<string> copy_names_of_undefined_symbols(const symbol_section_accessor& section) {
vector<string> r;
for (auto i = 0u; i != section.get_symbols_num(); ++i) {
// TODO: this is boyscout code, caching the temporaries
// may be of worth.
for (auto i = 0u; i != section.get_symbols_num(); ++i) {
// TODO: this is boyscout code, caching the temporaries
// may be of worth.
auto tmp = read_symbol(section, i);
if (tmp.sect_idx == SHN_UNDEF && !tmp.name.empty()) {
r.push_back(std::move(tmp.name));
}
auto tmp = read_symbol(section, i);
if (tmp.sect_idx == SHN_UNDEF && !tmp.name.empty()) {
r.push_back(std::move(tmp.name));
}
return r;
}
const std::unordered_map<
std::string,
std::pair<ELFIO::Elf64_Addr, ELFIO::Elf_Xword>>& symbol_addresses()
{
static unordered_map<string, pair<Elf64_Addr, Elf_Xword>> r;
static once_flag f;
return r;
}
call_once(f, []() {
dl_iterate_phdr([](dl_phdr_info* info, size_t, void*) {
const std::unordered_map<std::string, std::pair<ELFIO::Elf64_Addr, ELFIO::Elf_Xword>>&
symbol_addresses() {
static unordered_map<string, pair<Elf64_Addr, Elf_Xword>> r;
static once_flag f;
call_once(f, []() {
dl_iterate_phdr(
[](dl_phdr_info* info, size_t, void*) {
static constexpr const char self[] = "/proc/self/exe";
elfio reader;
static unsigned int iter = 0u;
if (reader.load(!iter ? self : info->dlpi_name)) {
auto it = find_section_if(
reader, [](const class section* x) {
return x->get_type() == SHT_SYMTAB;
});
reader, [](const class section* x) { return x->get_type() == SHT_SYMTAB; });
if (it) {
const symbol_section_accessor symtab{reader, it};
@@ -107,12 +95,9 @@ namespace
for (auto i = 0u; i != symtab.get_symbols_num(); ++i) {
auto tmp = read_symbol(symtab, i);
if (tmp.type == STT_OBJECT &&
tmp.sect_idx != SHN_UNDEF) {
const auto addr =
tmp.value + (iter ? info->dlpi_addr : 0);
r.emplace(
move(tmp.name), make_pair(addr, tmp.size));
if (tmp.type == STT_OBJECT && tmp.sect_idx != SHN_UNDEF) {
const auto addr = tmp.value + (iter ? info->dlpi_addr : 0);
r.emplace(move(tmp.name), make_pair(addr, tmp.size));
}
}
}
@@ -121,367 +106,327 @@ namespace
}
return 0;
}, nullptr);
});
},
nullptr);
});
return r;
}
return r;
}
void associate_code_object_symbols_with_host_allocation(
const elfio& reader,
section* code_object_dynsym,
hsa_agent_t agent,
hsa_executable_t executable)
{
if (!code_object_dynsym) return;
void associate_code_object_symbols_with_host_allocation(const elfio& reader,
section* code_object_dynsym,
hsa_agent_t agent,
hsa_executable_t executable) {
if (!code_object_dynsym) return;
const auto undefined_symbols = copy_names_of_undefined_symbols(
symbol_section_accessor{reader, code_object_dynsym});
const auto undefined_symbols =
copy_names_of_undefined_symbols(symbol_section_accessor{reader, code_object_dynsym});
for (auto&& x : undefined_symbols) {
if (globals().find(x) != globals().cend()) return;
for (auto&& x : undefined_symbols) {
if (globals().find(x) != globals().cend()) return;
const auto it1 = symbol_addresses().find(x);
const auto it1 = symbol_addresses().find(x);
if (it1 == symbol_addresses().cend()) {
throw runtime_error{"Global symbol: " + x + " is undefined."};
}
static mutex mtx;
lock_guard<mutex> lck{mtx};
if (globals().find(x) != globals().cend()) return;
globals().emplace(x, (void*)(it1->second.first));
void* p = nullptr;
hsa_amd_memory_lock(
reinterpret_cast<void*>(it1->second.first),
it1->second.second,
nullptr, // All agents.
0,
&p);
hsa_executable_agent_global_variable_define(
executable, agent, x.c_str(), p);
}
}
vector<char> code_object_blob_for_process()
{
static constexpr const char self[] = "/proc/self/exe";
static constexpr const char kernel_section[] = ".kernel";
elfio reader;
if (!reader.load(self)) {
throw runtime_error{"Failed to load ELF file for current process."};
if (it1 == symbol_addresses().cend()) {
throw runtime_error{"Global symbol: " + x + " is undefined."};
}
auto kernels = find_section_if(reader, [](const section* x) {
return x->get_name() == kernel_section;
});
static mutex mtx;
lock_guard<mutex> lck{mtx};
vector<char> r;
if (kernels) {
r.insert(
r.end(),
kernels->get_data(),
kernels->get_data() + kernels->get_size());
}
if (globals().find(x) != globals().cend()) return;
globals().emplace(x, (void*)(it1->second.first));
void* p = nullptr;
hsa_amd_memory_lock(reinterpret_cast<void*>(it1->second.first), it1->second.second,
nullptr, // All agents.
0, &p);
return r;
}
const unordered_map<hsa_isa_t, vector<vector<char>>>& code_object_blobs()
{
static unordered_map<hsa_isa_t, vector<vector<char>>> r;
static once_flag f;
call_once(f, []() {
static vector<vector<char>> blobs{
code_object_blob_for_process()};
dl_iterate_phdr([](dl_phdr_info* info, std::size_t, void*) {
elfio tmp;
if (tmp.load(info->dlpi_name)) {
const auto it = find_section_if(tmp, [](const section* x) {
return x->get_name() == ".kernel";
});
if (it) blobs.emplace_back(
it->get_data(), it->get_data() + it->get_size());
}
return 0;
}, nullptr);
for (auto&& blob : blobs) {
Bundled_code_header tmp{blob};
if (valid(tmp)) {
for (auto&& bundle : bundles(tmp)) {
r[triple_to_hsa_isa(bundle.triple)].push_back(
bundle.blob);
}
}
}
});
return r;
}
vector<pair<uintptr_t, string>> function_names_for(
const elfio& reader, section* symtab)
{
vector<pair<uintptr_t, string>> r;
symbol_section_accessor symbols{reader, symtab};
for (auto i = 0u; i != symbols.get_symbols_num(); ++i) {
// TODO: this is boyscout code, caching the temporaries
// may be of worth.
auto tmp = read_symbol(symbols, i);
if (tmp.type == STT_FUNC &&
tmp.sect_idx != SHN_UNDEF &&
!tmp.name.empty()) {
r.emplace_back(tmp.value, tmp.name);
}
}
return r;
}
const vector<pair<uintptr_t, string>>& function_names_for_process()
{
static constexpr const char self[] = "/proc/self/exe";
static vector<pair<uintptr_t, string>> r;
static once_flag f;
call_once(f, []() {
elfio reader;
if (!reader.load(self)) {
throw runtime_error{
"Failed to load the ELF file for the current process."};
}
auto symtab = find_section_if(reader, [](const section* x) {
return x->get_type() == SHT_SYMTAB;
});
if (symtab) r = function_names_for(reader, symtab);
});
return r;
}
const unordered_map<string, vector<hsa_executable_symbol_t>>& kernels()
{
static unordered_map<string, vector<hsa_executable_symbol_t>> r;
static once_flag f;
call_once(f, []() {
static const auto copy_kernels = [](
hsa_executable_t, hsa_agent_t, hsa_executable_symbol_t s, void*) {
if (type(s) == HSA_SYMBOL_KIND_KERNEL) r[name(s)].push_back(s);
return HSA_STATUS_SUCCESS;
};
for (auto&& agent_executables : executables()) {
for (auto&& executable : agent_executables.second) {
hsa_executable_iterate_agent_symbols(
executable,
agent_executables.first,
copy_kernels,
nullptr);
}
}
});
return r;
}
void load_code_object_and_freeze_executable(
const string& file, hsa_agent_t agent, hsa_executable_t executable)
{ // TODO: the following sequence is inefficient, should be refactored
// into a single load of the file and subsequent ELFIO
// processing.
static const auto cor_deleter = [](hsa_code_object_reader_t* p) {
if (p) {
hsa_code_object_reader_destroy(*p);
delete p;
}
};
using RAII_code_reader = unique_ptr<
hsa_code_object_reader_t, decltype(cor_deleter)>;
if (!file.empty()) {
RAII_code_reader tmp{new hsa_code_object_reader_t, cor_deleter};
hsa_code_object_reader_create_from_memory(
file.data(), file.size(), tmp.get());
hsa_executable_load_agent_code_object(
executable, agent, *tmp, nullptr, nullptr);
hsa_executable_freeze(executable, nullptr);
static vector<RAII_code_reader> code_readers;
static mutex mtx;
lock_guard<mutex> lck{mtx};
code_readers.push_back(move(tmp));
}
hsa_executable_agent_global_variable_define(executable, agent, x.c_str(), p);
}
}
namespace hip_impl
{
const unordered_map<hsa_agent_t, vector<hsa_executable_t>>& executables()
{ // TODO: This leaks the hsa_executable_ts, it should use RAII.
static unordered_map<hsa_agent_t, vector<hsa_executable_t>> r;
static once_flag f;
vector<char> code_object_blob_for_process() {
static constexpr const char self[] = "/proc/self/exe";
static constexpr const char kernel_section[] = ".kernel";
call_once(f, []() {
static const auto accelerators = hc::accelerator::get_all();
elfio reader;
for (auto&& acc : accelerators) {
auto agent = static_cast<hsa_agent_t*>(acc.get_hsa_agent());
if (!agent || !acc.is_hsa_accelerator()) continue;
hsa_agent_iterate_isas(*agent, [](hsa_isa_t x, void* pa) {
const auto it = code_object_blobs().find(x);
if (it != code_object_blobs().cend()) {
hsa_agent_t a = *static_cast<hsa_agent_t*>(pa);
for (auto&& blob : it->second) {
hsa_executable_t tmp = {};
hsa_executable_create_alt(
HSA_PROFILE_FULL,
HSA_DEFAULT_FLOAT_ROUNDING_MODE_DEFAULT,
nullptr,
&tmp);
// TODO: this is massively inefficient and only
// meant for illustration.
string blob_to_str{blob.cbegin(), blob.cend()};
tmp = load_executable(blob_to_str, tmp, a);
if (tmp.handle) r[a].push_back(tmp);
}
}
return HSA_STATUS_SUCCESS;
}, agent);
}
});
return r;
if (!reader.load(self)) {
throw runtime_error{"Failed to load ELF file for current process."};
}
const unordered_map<uintptr_t, string>& function_names()
{
static unordered_map<uintptr_t, string> r{
function_names_for_process().cbegin(),
function_names_for_process().cend()};
static once_flag f;
auto kernels =
find_section_if(reader, [](const section* x) { return x->get_name() == kernel_section; });
call_once(f, []() {
dl_iterate_phdr([](dl_phdr_info* info, size_t, void*) {
vector<char> r;
if (kernels) {
r.insert(r.end(), kernels->get_data(), kernels->get_data() + kernels->get_size());
}
return r;
}
const unordered_map<hsa_isa_t, vector<vector<char>>>& code_object_blobs() {
static unordered_map<hsa_isa_t, vector<vector<char>>> r;
static once_flag f;
call_once(f, []() {
static vector<vector<char>> blobs{code_object_blob_for_process()};
dl_iterate_phdr(
[](dl_phdr_info* info, std::size_t, void*) {
elfio tmp;
if (tmp.load(info->dlpi_name)) {
const auto it = find_section_if(tmp, [](const section* x) {
return x->get_type() == SHT_SYMTAB;
});
const auto it = find_section_if(
tmp, [](const section* x) { return x->get_name() == ".kernel"; });
if (it) blobs.emplace_back(it->get_data(), it->get_data() + it->get_size());
}
return 0;
},
nullptr);
for (auto&& blob : blobs) {
Bundled_code_header tmp{blob};
if (valid(tmp)) {
for (auto&& bundle : bundles(tmp)) {
r[triple_to_hsa_isa(bundle.triple)].push_back(bundle.blob);
}
}
}
});
return r;
}
vector<pair<uintptr_t, string>> function_names_for(const elfio& reader, section* symtab) {
vector<pair<uintptr_t, string>> r;
symbol_section_accessor symbols{reader, symtab};
for (auto i = 0u; i != symbols.get_symbols_num(); ++i) {
// TODO: this is boyscout code, caching the temporaries
// may be of worth.
auto tmp = read_symbol(symbols, i);
if (tmp.type == STT_FUNC && tmp.sect_idx != SHN_UNDEF && !tmp.name.empty()) {
r.emplace_back(tmp.value, tmp.name);
}
}
return r;
}
const vector<pair<uintptr_t, string>>& function_names_for_process() {
static constexpr const char self[] = "/proc/self/exe";
static vector<pair<uintptr_t, string>> r;
static once_flag f;
call_once(f, []() {
elfio reader;
if (!reader.load(self)) {
throw runtime_error{"Failed to load the ELF file for the current process."};
}
auto symtab =
find_section_if(reader, [](const section* x) { return x->get_type() == SHT_SYMTAB; });
if (symtab) r = function_names_for(reader, symtab);
});
return r;
}
const unordered_map<string, vector<hsa_executable_symbol_t>>& kernels() {
static unordered_map<string, vector<hsa_executable_symbol_t>> r;
static once_flag f;
call_once(f, []() {
static const auto copy_kernels = [](hsa_executable_t, hsa_agent_t,
hsa_executable_symbol_t s, void*) {
if (type(s) == HSA_SYMBOL_KIND_KERNEL) r[name(s)].push_back(s);
return HSA_STATUS_SUCCESS;
};
for (auto&& agent_executables : executables()) {
for (auto&& executable : agent_executables.second) {
hsa_executable_iterate_agent_symbols(executable, agent_executables.first,
copy_kernels, nullptr);
}
}
});
return r;
}
void load_code_object_and_freeze_executable(
const string& file, hsa_agent_t agent,
hsa_executable_t
executable) { // TODO: the following sequence is inefficient, should be refactored
// into a single load of the file and subsequent ELFIO
// processing.
static const auto cor_deleter = [](hsa_code_object_reader_t* p) {
if (p) {
hsa_code_object_reader_destroy(*p);
delete p;
}
};
using RAII_code_reader = unique_ptr<hsa_code_object_reader_t, decltype(cor_deleter)>;
if (!file.empty()) {
RAII_code_reader tmp{new hsa_code_object_reader_t, cor_deleter};
hsa_code_object_reader_create_from_memory(file.data(), file.size(), tmp.get());
hsa_executable_load_agent_code_object(executable, agent, *tmp, nullptr, nullptr);
hsa_executable_freeze(executable, nullptr);
static vector<RAII_code_reader> code_readers;
static mutex mtx;
lock_guard<mutex> lck{mtx};
code_readers.push_back(move(tmp));
}
}
} // namespace
namespace hip_impl {
const unordered_map<hsa_agent_t, vector<hsa_executable_t>>&
executables() { // TODO: This leaks the hsa_executable_ts, it should use RAII.
static unordered_map<hsa_agent_t, vector<hsa_executable_t>> r;
static once_flag f;
call_once(f, []() {
static const auto accelerators = hc::accelerator::get_all();
for (auto&& acc : accelerators) {
auto agent = static_cast<hsa_agent_t*>(acc.get_hsa_agent());
if (!agent || !acc.is_hsa_accelerator()) continue;
hsa_agent_iterate_isas(*agent,
[](hsa_isa_t x, void* pa) {
const auto it = code_object_blobs().find(x);
if (it != code_object_blobs().cend()) {
hsa_agent_t a = *static_cast<hsa_agent_t*>(pa);
for (auto&& blob : it->second) {
hsa_executable_t tmp = {};
hsa_executable_create_alt(
HSA_PROFILE_FULL,
HSA_DEFAULT_FLOAT_ROUNDING_MODE_DEFAULT, nullptr,
&tmp);
// TODO: this is massively inefficient and only
// meant for illustration.
string blob_to_str{blob.cbegin(), blob.cend()};
tmp = load_executable(blob_to_str, tmp, a);
if (tmp.handle) r[a].push_back(tmp);
}
}
return HSA_STATUS_SUCCESS;
},
agent);
}
});
return r;
}
const unordered_map<uintptr_t, string>& function_names() {
static unordered_map<uintptr_t, string> r{function_names_for_process().cbegin(),
function_names_for_process().cend()};
static once_flag f;
call_once(f, []() {
dl_iterate_phdr(
[](dl_phdr_info* info, size_t, void*) {
elfio tmp;
if (tmp.load(info->dlpi_name)) {
const auto it = find_section_if(
tmp, [](const section* x) { return x->get_type() == SHT_SYMTAB; });
if (it) {
auto n = function_names_for(tmp, it);
for (auto&& f : n) f.first += info->dlpi_addr;
r.insert(
make_move_iterator(n.begin()),
make_move_iterator(n.end()));
r.insert(make_move_iterator(n.begin()), make_move_iterator(n.end()));
}
}
return 0;
}, nullptr);
});
},
nullptr);
});
return r;
}
return r;
}
const unordered_map<
uintptr_t, vector<pair<hsa_agent_t, Kernel_descriptor>>>& functions()
{
static unordered_map<
uintptr_t, vector<pair<hsa_agent_t, Kernel_descriptor>>> r;
static once_flag f;
const unordered_map<uintptr_t, vector<pair<hsa_agent_t, Kernel_descriptor>>>& functions() {
static unordered_map<uintptr_t, vector<pair<hsa_agent_t, Kernel_descriptor>>> r;
static once_flag f;
call_once(f, []() {
for (auto&& function : function_names()) {
const auto it = kernels().find(function.second);
call_once(f, []() {
for (auto&& function : function_names()) {
const auto it = kernels().find(function.second);
if (it != kernels().cend()) {
for (auto&& kernel_symbol : it->second) {
r[function.first].emplace_back(
agent(kernel_symbol),
Kernel_descriptor{
kernel_object(kernel_symbol),
group_size(kernel_symbol),
private_size(kernel_symbol),
it->first});
}
if (it != kernels().cend()) {
for (auto&& kernel_symbol : it->second) {
r[function.first].emplace_back(
agent(kernel_symbol),
Kernel_descriptor{kernel_object(kernel_symbol), group_size(kernel_symbol),
private_size(kernel_symbol), it->first});
}
}
});
}
});
return r;
}
return r;
}
unordered_map<string, void*>& globals()
{
static unordered_map<string, void*> r;
static once_flag f;
call_once(f, []() { r.reserve(symbol_addresses().size()); });
unordered_map<string, void*>& globals() {
static unordered_map<string, void*> r;
static once_flag f;
call_once(f, []() { r.reserve(symbol_addresses().size()); });
return r;
}
return r;
}
hsa_executable_t load_executable(
const string& file, hsa_executable_t executable, hsa_agent_t agent)
{
elfio reader;
stringstream tmp{file};
hsa_executable_t load_executable(const string& file, hsa_executable_t executable,
hsa_agent_t agent) {
elfio reader;
stringstream tmp{file};
if (!reader.load(tmp)) return hsa_executable_t{};
if (!reader.load(tmp)) return hsa_executable_t{};
const auto code_object_dynsym =
find_section_if(reader, [](const ELFIO::section* x) {
return x->get_type() == SHT_DYNSYM;
});
const auto code_object_dynsym = find_section_if(
reader, [](const ELFIO::section* x) { return x->get_type() == SHT_DYNSYM; });
associate_code_object_symbols_with_host_allocation(
reader, code_object_dynsym, agent, executable);
associate_code_object_symbols_with_host_allocation(reader, code_object_dynsym, agent,
executable);
load_code_object_and_freeze_executable(file, agent, executable);
load_code_object_and_freeze_executable(file, agent, executable);
return executable;
}
return executable;
}
// To force HIP to load the kernels and to setup the function
// symbol map on program startup
class startup_kernel_loader {
private:
startup_kernel_loader() { functions(); }
startup_kernel_loader(const startup_kernel_loader&) = delete;
startup_kernel_loader& operator= (const startup_kernel_loader&) = delete;
static startup_kernel_loader skl;
};
startup_kernel_loader startup_kernel_loader::skl;
// To force HIP to load the kernels and to setup the function
// symbol map on program startup
class startup_kernel_loader {
private:
startup_kernel_loader() { functions(); }
startup_kernel_loader(const startup_kernel_loader&) = delete;
startup_kernel_loader& operator=(const startup_kernel_loader&) = delete;
static startup_kernel_loader skl;
};
startup_kernel_loader startup_kernel_loader::skl;
} // Namespace hip_impl.
} // Namespace hip_impl.
+24 -30
View File
@@ -32,17 +32,19 @@ THE SOFTWARE.
// Helper functions to convert HIP function arguments into strings.
// Handles POD data types as well as enumerations (ie hipMemcpyKind).
// The implementation uses C++11 variadic templates and template specialization.
// The hipMemcpyKind example below is a good example that shows how to implement conversion for a new HSA type.
// The hipMemcpyKind example below is a good example that shows how to implement conversion for a
// new HSA type.
// Handy macro to convert an enumeration to a stringified version of same:
#define CASE_STR(x) case x: return #x;
#define CASE_STR(x) \
case x: \
return #x;
// Building block functions:
template <typename T>
inline std::string ToHexString(T v)
{
inline std::string ToHexString(T v) {
std::ostringstream ss;
ss << "0x" << std::hex << v;
return ss.str();
@@ -54,8 +56,7 @@ inline std::string ToHexString(T v)
// This is the default which works for most types:
template <typename T>
inline std::string ToString(T v)
{
inline std::string ToString(T v) {
std::ostringstream ss;
ss << v;
return ss.str();
@@ -64,16 +65,14 @@ inline std::string ToString(T v)
// hipEvent_t specialization. TODO - maybe add an event ID for debug?
template <>
inline std::string ToString(hipEvent_t v)
{
inline std::string ToString(hipEvent_t v) {
std::ostringstream ss;
ss << v;
return ss.str();
};
// hipStream_t
template <>
inline std::string ToString(hipStream_t v)
{
inline std::string ToString(hipStream_t v) {
std::ostringstream ss;
if (v == NULL) {
ss << "stream:<null>";
@@ -86,40 +85,35 @@ inline std::string ToString(hipStream_t v)
// hipMemcpyKind specialization
template <>
inline std::string ToString(hipMemcpyKind v)
{
switch(v) {
CASE_STR(hipMemcpyHostToHost);
CASE_STR(hipMemcpyHostToDevice);
CASE_STR(hipMemcpyDeviceToHost);
CASE_STR(hipMemcpyDeviceToDevice);
CASE_STR(hipMemcpyDefault);
default : return ToHexString(v);
inline std::string ToString(hipMemcpyKind v) {
switch (v) {
CASE_STR(hipMemcpyHostToHost);
CASE_STR(hipMemcpyHostToDevice);
CASE_STR(hipMemcpyDeviceToHost);
CASE_STR(hipMemcpyDeviceToDevice);
CASE_STR(hipMemcpyDefault);
default:
return ToHexString(v);
};
};
template <>
inline std::string ToString(hipError_t v)
{
inline std::string ToString(hipError_t v) {
return ihipErrorString(v);
};
// Catch empty arguments case
inline std::string ToString()
{
return ("");
}
inline std::string ToString() { return (""); }
//---
// C++11 variadic template - peels off first argument, converts to string, and calls itself again to peel the next arg.
// Strings are automatically separated by comma+space.
// C++11 variadic template - peels off first argument, converts to string, and calls itself again to
// peel the next arg. Strings are automatically separated by comma+space.
template <typename T, typename... Args>
inline std::string ToString(T first, Args... args)
{
return ToString(first) + ", " + ToString(args...) ;
inline std::string ToString(T first, Args... args) {
return ToString(first) + ", " + ToString(args...);
}
#endif