Initial Commit

Contributors:
Ammar ELWazir <aelwazir@amd.com>
AravindanC <aravindan.cheruvally@amd.com>
Benjamin Welton <bewelton@amd.com>
Ma, Bing <Bing.Ma@amd.com>
Chun Yang <chun.yang@amd.com>
Cole Nelson <cole.nelson@amd.com>
Ethan Stewart <ethan.stewart@amd.com>
Evgeny <evgeny.shcherbakov@amd.com>
Freddy Paul <Freddy.paul@amd.com>
Giovanni Baraldi <gbaraldi@amd.com>
Gopesh Bhardwaj <Gopesh.Bhardwaj@amd.com>
Icarus Sparry <icarus.sparry@amd.com>
itrowbri <Ian.Trowbridge@amd.com>
James Edwards <JamesAdrian.Edwards@amd.com>
jatang <jatang@amd.com>
Jeremy Newton <Jeremy.Newton@amd.com>
Jonathan Kim <jonathan.kim@amd.com>
Kent Russell <kent.russell@amd.com>
Kiumars Sabeti <kiumars.sabeti@amd.com>
Lang Yu <lang.yu@amd.com>
Laurent Morichetti <laurent.morichetti@amd.com>
Mallya, Ameya Keshava <AmeyaKeshava.Mallya@amd.com>
Manjunath Jakaraddi <manjunath.jakaraddi@amd.com>
Mark Laws <markdavid.laws@amd.com>
Mohan Kumar Mithur <Mohan.KumarMithur@amd.com>
Nicholas Curtis <nicurtis@amd.com>
Nirmal Unnikrishnan <Nirmal.Unnikrishnan@amd.com>
Parag Bhandari <parag.bhandari@amd.com>
Ranjith Ramakrishnan <Ranjith.Ramakrishnan@amd.com>
Robert Gregory <Robert.Gregory@amd.com>
Saravanan Solaiyappan <saravanan.solaiyappan@amd.com>
Saurabh Verma <saurabh.verma@amd.com>
Srihari Uttanur <srihari.u@amd.com>
Srinivasan Subramanian <srinivasan.subramanian@amd.com>
Sriraksha Nagaraj <Sriraksha.Nagaraj@amd.com>
Sushma Vaddireddy <svaddire@amd.com>
Xianwei Zhang <Xianwei.Zhang@amd.com>
This commit is contained in:
Evgeny
2017-06-20 17:43:27 -05:00
committed by Ammar ELWazir
parent 7c7d7b4022
commit 1ed169e30c
157 changed files with 371830 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
add_subdirectory(include)
+46
View File
@@ -0,0 +1,46 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#ifndef SRC_CORE_AMD_AQL_PM4_IB_PACKET_H_
#define SRC_CORE_AMD_AQL_PM4_IB_PACKET_H_
#include <hsa/hsa.h>
// Value of 'pm4_ib_format' field of amd_aql_pm4_ib_packet_t packet
static const uint32_t AMD_AQL_PM4_IB_FORMAT = 1;
// Value of 'dw_count_remain' field of amd_aql_pm4_ib_packet_t packet
static const uint32_t AMD_AQL_PM4_IB_DW_COUNT_REMAIN = 10;
// Size of 'reserved' array of amd_aql_pm4_ib_packet_t packet
static const uint32_t AMD_AQL_PM4_IB_RESERVED_COUNT = 8;
// AQL Vendor Specific Packet which carry PM4 IB command
typedef struct {
uint16_t header;
uint16_t pm4_ib_format;
uint32_t pm4_ib_command[4];
uint32_t dw_count_remain;
uint32_t reserved[AMD_AQL_PM4_IB_RESERVED_COUNT];
hsa_signal_t completion_signal;
} amd_aql_pm4_ib_packet_t;
#endif // SRC_CORE_AMD_AQL_PM4_IB_PACKET_H_
+799
View File
@@ -0,0 +1,799 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/aql_profile.hpp"
#include "core/include/aql_profile_v2.h"
#include <cstdint>
#include <future>
#include <map>
#include <string>
#include <vector>
#include <mutex>
#include "core/counter_dimensions.hpp"
#include "core/logger.h"
#include "core/pm4_factory.h"
#include "pm4/cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/spm_builder.h"
#include "pm4/sqtt_builder.h"
#include "core/commandbuffermgr.hpp"
#define CONSTRUCTOR_API __attribute__((constructor))
#define DESTRUCTOR_API __attribute__((destructor))
#define ERR_CHECK(cond, err, msg) \
{ \
if (cond) { \
ERR_LOGGING << msg; \
return err; \
} \
}
// Getting SPM data using driver API
namespace spm_kfd_namespace {
hsa_status_t spm_iterate_data(const hsa_ven_amd_aqlprofile_profile_t* profile,
hsa_ven_amd_aqlprofile_data_callback_t callback, void* data);
}
// PC sampling callback data
struct pcsmp_callback_data_t {
const char* kernel_name; // sampled kernel name
void* data_buffer; // host buffer for tracing data
uint64_t id; // sample id
uint64_t cycle; // sample cycle
uint64_t pc; // sample PC
};
std::atomic<int> ATT_TARGET_CU{0};
namespace aql_profile {
// Command buffer partitioning manager
// Supports Pre/Post commands partitioning
// and prefix control partition
static std::unordered_map<void*, pm4_builder::TraceConfig> configs;
static std::mutex config_mut;
static inline pm4_builder::counters_vector CountersVec(const profile_t* profile,
const Pm4Factory* pm4_factory) {
pm4_builder::counters_vector vec;
std::map<block_des_t, uint32_t, lt_block_des> index_map;
for (const hsa_ven_amd_aqlprofile_event_t* p = profile->events;
p < profile->events + profile->event_count; ++p) {
const GpuBlockInfo* block_info = pm4_factory->GetBlockInfo(p);
const block_des_t block_des = {pm4_factory->GetBlockInfo(p)->id, p->block_index};
// Counting counter register index per block
const auto ret = index_map.insert({block_des, 0});
uint32_t& reg_index = ret.first->second;
if (pm4_builder::SPISkip(block_info->attr, p->counter_id)) {
vec.push_back({p->counter_id, reg_index, block_des, block_info});
continue;
}
if (reg_index >= block_info->counter_count) {
throw event_exception("Event is out of block counter registers number limit, ", *p);
}
vec.push_back({p->counter_id, reg_index, block_des, block_info});
++reg_index;
}
if (pm4_factory->IsGFX10() && (vec.get_attr() & CounterBlockGRBMAttr) == 0 && !vec.empty()) {
event_t grbm_event{
.block_name = HSA_VEN_AMD_AQLPROFILE_BLOCK_NAME_GRBM, .block_index = 0, .counter_id = 0};
const GpuBlockInfo* block_info = pm4_factory->GetBlockInfo(&grbm_event);
if (block_info == nullptr) return vec;
const block_des_t block_des = {block_info->id, 0};
const auto ret = index_map.insert({block_des, 0});
uint32_t& reg_index = ret.first->second;
vec.push_back({0, reg_index, block_des, block_info});
reg_index++;
}
return vec;
}
static inline bool IsEventMatch(const event_t& event1, const event_t& event2) {
return (event1.block_name == event2.block_name) && (event1.block_index == event2.block_index) &&
(event1.counter_id == event2.counter_id);
}
hsa_status_t DefaultPmcdataCallback(hsa_ven_amd_aqlprofile_info_type_t info_type,
hsa_ven_amd_aqlprofile_info_data_t* info_data,
void* callback_data) {
hsa_status_t status = HSA_STATUS_SUCCESS;
hsa_ven_amd_aqlprofile_info_data_t* passed_data =
reinterpret_cast<hsa_ven_amd_aqlprofile_info_data_t*>(callback_data);
if (info_type == HSA_VEN_AMD_AQLPROFILE_INFO_PMC_DATA) {
if (IsEventMatch(info_data->pmc_data.event, passed_data->pmc_data.event)) {
if (passed_data->sample_id == UINT32_MAX) {
passed_data->pmc_data.result += info_data->pmc_data.result;
} else if (passed_data->sample_id == info_data->sample_id) {
passed_data->pmc_data.result = info_data->pmc_data.result;
status = HSA_STATUS_INFO_BREAK;
}
}
}
return status;
}
hsa_status_t DefaultTracedataCallback(hsa_ven_amd_aqlprofile_info_type_t info_type,
hsa_ven_amd_aqlprofile_info_data_t* info_data,
void* callback_data) {
hsa_status_t status = HSA_STATUS_SUCCESS;
hsa_ven_amd_aqlprofile_info_data_t* passed_data =
reinterpret_cast<hsa_ven_amd_aqlprofile_info_data_t*>(callback_data);
if (info_type == HSA_VEN_AMD_AQLPROFILE_INFO_TRACE_DATA) {
if (info_data->sample_id == passed_data->sample_id) {
passed_data->trace_data = info_data->trace_data;
status = HSA_STATUS_INFO_BREAK;
}
}
return status;
}
Logger::mutex_t Logger::mutex_;
Logger* Logger::instance_ = NULL;
bool Pm4Factory::concurrent_create_mode_ = false;
bool Pm4Factory::spm_kfd_mode_ = false;
Pm4Factory::mutex_t Pm4Factory::mutex_;
Pm4Factory::instances_t* Pm4Factory::instances_ = NULL;
bool read_api_enabled = true;
CONSTRUCTOR_API void constructor() {
const char* read_api_enabled_str = getenv("AQLPROFILE_READ_API");
if (read_api_enabled_str != NULL) {
if (atoi(read_api_enabled_str) == 0) read_api_enabled = false;
}
}
DESTRUCTOR_API void destructor() {
Logger::Destroy();
Pm4Factory::Destroy();
}
} // namespace aql_profile
extern "C" {
// Return library major/minor version
PUBLIC_API uint32_t hsa_ven_amd_aqlprofile_version_major() { return HSA_AQLPROFILE_VERSION_MAJOR; }
PUBLIC_API uint32_t hsa_ven_amd_aqlprofile_version_minor() { return HSA_AQLPROFILE_VERSION_MINOR; }
// Returns the last error message
PUBLIC_API hsa_status_t hsa_ven_amd_aqlprofile_error_string(const char** str) {
*str = aql_profile::Logger::LastMessage().c_str();
return HSA_STATUS_SUCCESS;
}
// Check if event is valid for the specific GPU
PUBLIC_API hsa_status_t hsa_ven_amd_aqlprofile_validate_event(
hsa_agent_t agent, const hsa_ven_amd_aqlprofile_event_t* event, bool* result) {
hsa_status_t status = HSA_STATUS_SUCCESS;
*result = false;
try {
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(agent);
if (pm4_factory->GetBlockInfo(event) != NULL) *result = true;
} catch (aql_profile::event_exception& e) {
INFO_LOGGING << e.what();
} catch (std::exception& e) {
ERR_LOGGING << e.what();
status = HSA_STATUS_ERROR;
}
return status;
}
// Method to populate the provided AQL packet with profiling start commands
PUBLIC_API hsa_status_t hsa_ven_amd_aqlprofile_start(hsa_ven_amd_aqlprofile_profile_t* profile,
aql_profile::packet_t* aql_start_packet) {
try {
pm4_builder::CmdBuffer commands;
aql_profile::CommandBufferMgr cmd_buffer_mgr(profile->command_buffer.ptr, UINT_MAX);
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile);
const bool is_concurrent = pm4_factory->IsConcurrent();
const pm4_builder::counters_vector countersVec = CountersVec(profile, pm4_factory);
if (profile->type == HSA_VEN_AMD_AQLPROFILE_EVENT_TYPE_PMC) {
pm4_builder::PmcBuilder* pmc_builder = pm4_factory->GetPmcBuilder();
// Generate read commands
auto data_size = pmc_builder->Read(&commands, countersVec, profile->output_buffer.ptr);
if (!aql_profile::read_api_enabled) commands.Clear();
cmd_buffer_mgr.SetRdSize(commands.Size());
// Copy generated read commands
if (profile->command_buffer.ptr != NULL) {
const aql_profile::descriptor_t rd_descr = cmd_buffer_mgr.GetRdDescr();
memcpy(rd_descr.ptr, commands.Data(), commands.Size());
commands.Clear();
}
// Generate start commands
pmc_builder->Start(&commands, countersVec);
cmd_buffer_mgr.SetPreSize(commands.Size());
// Generate stop commands
if (!aql_profile::read_api_enabled)
pmc_builder->Read(&commands, countersVec, profile->output_buffer.ptr);
pmc_builder->Stop(&commands, countersVec);
if (profile->output_buffer.size < data_size) {
profile->output_buffer.size = data_size;
if (profile->output_buffer.ptr != NULL) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
} else if (profile->type == HSA_VEN_AMD_AQLPROFILE_EVENT_TYPE_TRACE) {
pm4_builder::TraceConfig trace_config{};
const uint64_t se_number_total = pm4_factory->GetShaderEnginesNumber();
if (profile->parameters) {
for (const hsa_ven_amd_aqlprofile_parameter_t* p = profile->parameters;
p < (profile->parameters + profile->parameter_count); ++p) {
switch (p->parameter_name) {
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_SE_MASK:
trace_config.se_mask = p->value & ((1ull << se_number_total) - 1);
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_COMPUTE_UNIT_TARGET:
if (p->value > 15)
throw aql_profile::aql_profile_exc_val<uint32_t>(
"ThreadTraceConfig: CuId must be between 0 and 15, TargetCu", p->value);
trace_config.targetCu = p->value;
ATT_TARGET_CU.store(p->value);
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_VM_ID_MASK:
trace_config.vmIdMask = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_MASK:
if ((p->value & 0x50) != 0)
throw aql_profile::aql_profile_exc_val<uint32_t>(
"ThreadTraceConfig: Mask should have bits [4,6] set to Zero, Mask", p->value);
trace_config.deprecated_mask = p->value;
trace_config.targetCu = p->value & 0xF;
ATT_TARGET_CU.store(trace_config.targetCu);
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_TOKEN_MASK:
if ((p->value & 0xFF000000) != 0)
throw aql_profile::aql_profile_exc_val<uint32_t>(
"ThreadTraceConfig: TokenMask should have bits [31:25] set to Zero, TokenMask",
p->value);
trace_config.deprecated_tokenMask = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_TOKEN_MASK2:
trace_config.deprecated_tokenMask2 = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_SAMPLE_RATE:
trace_config.sampleRate = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_K_CONCURRENT:
trace_config.concurrent = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_SIMD_SELECTION:
trace_config.simd_sel = p->value & 0xF;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_OCCUPANCY_MODE:
trace_config.occupancy_mode = p->value ? 1 : 0;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_ATT_BUFFER_SIZE:
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_PERFCOUNTER_MASK:
trace_config.perfMASK = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_PERFCOUNTER_CTRL:
trace_config.perfCTRL = ((p->value & 0x1F) << 8) | 0xFFFF007F;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_PERFCOUNTER_NAME:
if (trace_config.perfcounters.size() < 8)
trace_config.perfcounters.push_back({p->value, 0xF});
break;
default:
ERR_LOGGING << "Bad trace parameter name (" << p->parameter_name << ")";
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
}
}
auto control_size = sizeof(pm4_builder::TraceControl) * se_number_total;
char* prefix_ptr = cmd_buffer_mgr.AddPrefix(control_size);
auto* control_ptr = reinterpret_cast<pm4_builder::TraceControl*>(prefix_ptr);
trace_config.control_buffer_ptr = control_ptr;
trace_config.control_buffer_size = control_size;
trace_config.data_buffer_ptr = profile->output_buffer.ptr;
trace_config.data_buffer_size = profile->output_buffer.size;
if (countersVec.size() == 0) {
pm4_builder::SqttBuilder* sqtt_builder = pm4_factory->GetSqttBuilder();
// Generate start commands
sqtt_builder->Begin(&commands, &trace_config);
cmd_buffer_mgr.SetPreSize(commands.Size());
// Generate stop commands
sqtt_builder->End(&commands, &trace_config);
} else {
const char* sz_sampling_rate = getenv("AQLPROFILE_SPM_SAMPLE_RATE");
if (sz_sampling_rate != NULL) trace_config.sampleRate = atoi(sz_sampling_rate);
pm4_builder::SpmBuilder* spm_builder = pm4_factory->GetSpmBuilder();
// Generate start commands
spm_builder->Begin(&commands, &trace_config, countersVec);
cmd_buffer_mgr.SetPreSize(commands.Size());
// Generate stop commands
spm_builder->End(&commands, &trace_config);
}
aql_profile::configs[profile->command_buffer.ptr] = trace_config;
} else {
ERR_LOGGING << "Bad profile type (" << profile->type << ")";
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
cmd_buffer_mgr.Finalize(commands.Size());
const uint32_t cmd_size = (cmd_buffer_mgr.GetSize() + 0x1800) & ~0xFFF;
if (profile->command_buffer.size < cmd_size) {
profile->command_buffer.size = cmd_size;
if (profile->command_buffer.ptr != NULL) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
if (profile->command_buffer.ptr != NULL) {
// Copy generated commands
const aql_profile::descriptor_t pre_descr = cmd_buffer_mgr.GetPreDescr();
const aql_profile::descriptor_t post_descr = cmd_buffer_mgr.GetPostDescr();
memcpy(pre_descr.ptr, commands.Data(), pre_descr.size);
memcpy(post_descr.ptr, reinterpret_cast<const char*>(commands.Data()) + pre_descr.size,
post_descr.size);
// Populate start aql packet
pm4_builder::CmdBuilder* cmd_writer = pm4_factory->GetCmdBuilder();
aql_profile::PopulateAql(pre_descr.ptr, pre_descr.size, cmd_writer, aql_start_packet);
}
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
// Method to populate the provided AQL packet with profiling stop commands
PUBLIC_API hsa_status_t hsa_ven_amd_aqlprofile_stop(const hsa_ven_amd_aqlprofile_profile_t* profile,
aql_profile::packet_t* aql_stop_packet) {
try {
// Populate stop aql packet
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile);
pm4_builder::CmdBuilder* cmd_writer = pm4_factory->GetCmdBuilder();
aql_profile::CommandBufferMgr cmd_buffer_mgr(profile);
const aql_profile::descriptor_t post_descr = cmd_buffer_mgr.GetPostDescr();
aql_profile::PopulateAql(post_descr.ptr, post_descr.size, cmd_writer, aql_stop_packet);
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
// Method to populate the provided AQL packet with profiling read commands
PUBLIC_API hsa_status_t hsa_ven_amd_aqlprofile_read(const hsa_ven_amd_aqlprofile_profile_t* profile,
aql_profile::packet_t* aql_read_packet) {
if (!aql_profile::read_api_enabled) return HSA_STATUS_ERROR;
try {
// Populate read aql packet
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile);
const bool is_concurrent = pm4_factory->IsConcurrent();
pm4_builder::CmdBuilder* cmd_writer = pm4_factory->GetCmdBuilder();
aql_profile::CommandBufferMgr cmd_buffer_mgr(profile);
const aql_profile::descriptor_t rd_descr =
(is_concurrent == false) ? cmd_buffer_mgr.GetRdDescr() : cmd_buffer_mgr.FetchRdDescr();
aql_profile::PopulateAql(rd_descr.ptr, rd_descr.size, cmd_writer, aql_read_packet);
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
// Legacy devices, converting of the profiling AQL packet to PM4 packet blob
PUBLIC_API hsa_status_t
hsa_ven_amd_aqlprofile_legacy_get_pm4(const aql_profile::packet_t* aql_packet, void* data) {
return HSA_STATUS_ERROR;
}
// Method for getting the profile info
PUBLIC_API hsa_status_t
hsa_ven_amd_aqlprofile_get_info(const hsa_ven_amd_aqlprofile_profile_t* profile,
hsa_ven_amd_aqlprofile_info_type_t attribute, void* value) {
hsa_status_t status = HSA_STATUS_SUCCESS;
const uint32_t attr_op = (uint32_t)attribute;
const uint32_t begin_op = (uint32_t)HSA_VEN_AMD_AQLPROFILE_INFO_ENABLE_CMD;
if (attr_op >= begin_op) attribute = (hsa_ven_amd_aqlprofile_info_type_t)begin_op;
if (profile == NULL) {
ERR_LOGGING << "NULL argument 'profile'";
return HSA_STATUS_ERROR;
}
if (attribute != HSA_VEN_AMD_AQLPROFILE_INFO_ENABLE_CMD) {
if (value == NULL) {
ERR_LOGGING << "NULL argument 'value'";
return HSA_STATUS_ERROR;
}
}
try {
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile);
switch (attribute) {
case HSA_VEN_AMD_AQLPROFILE_INFO_COMMAND_BUFFER_SIZE:
*(uint32_t*)value = 0x2000; // a current approximation as 4K is big enough
break;
case HSA_VEN_AMD_AQLPROFILE_INFO_PMC_DATA_SIZE:
*(uint32_t*)value = 0x1800; // a current approximation as 4K is big enough
break;
case HSA_VEN_AMD_AQLPROFILE_INFO_PMC_DATA:
reinterpret_cast<hsa_ven_amd_aqlprofile_info_data_t*>(value)->pmc_data.result = 0;
status = hsa_ven_amd_aqlprofile_iterate_data(profile, aql_profile::DefaultPmcdataCallback,
value);
break;
case HSA_VEN_AMD_AQLPROFILE_INFO_TRACE_DATA:
status = hsa_ven_amd_aqlprofile_iterate_data(profile, aql_profile::DefaultTracedataCallback,
value);
break;
case HSA_VEN_AMD_AQLPROFILE_INFO_BLOCK_COUNTERS:
*reinterpret_cast<uint32_t*>(value) =
pm4_factory->GetBlockInfo(&(profile->events[0]))->counter_count;
break;
case HSA_VEN_AMD_AQLPROFILE_INFO_BLOCK_ID: {
hsa_ven_amd_aqlprofile_id_query_t* query =
reinterpret_cast<hsa_ven_amd_aqlprofile_id_query_t*>(value);
const uint32_t block = pm4_factory->FindBlock(query->name);
const GpuBlockInfo* info = pm4_factory->GetBlockInfo(block);
status = (info == NULL) ? HSA_STATUS_ERROR : HSA_STATUS_SUCCESS;
if (status == HSA_STATUS_SUCCESS) {
query->id = block;
query->instance_count = info->instance_count;
}
break;
}
case HSA_VEN_AMD_AQLPROFILE_INFO_ENABLE_CMD: {
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile);
pm4_builder::PmcBuilder* pmc_builder = pm4_factory->GetPmcBuilder();
pm4_builder::CmdBuilder* cmd_writer = pm4_factory->GetCmdBuilder();
pm4_builder::CmdBuffer commands;
const uint32_t op = attr_op - begin_op;
switch (op) {
case 0:
pmc_builder->Enable(&commands);
break;
case 1:
pmc_builder->Disable(&commands);
break;
case 2:
pmc_builder->WaitIdle(&commands);
break;
default:
ERR_LOGGING << "get_info, not supported op (" << op << ")";
status = HSA_STATUS_ERROR;
}
if (profile->command_buffer.ptr == NULL) {
const_cast<hsa_ven_amd_aqlprofile_profile_t*>(profile)->command_buffer.size =
commands.Size();
break;
}
if (profile->command_buffer.size != commands.Size()) {
ERR_LOGGING << "get_info, wrong profile cmd size";
status = HSA_STATUS_ERROR;
break;
}
if (value == NULL) {
ERR_LOGGING << "NULL argument 'value'";
status = HSA_STATUS_ERROR;
break;
}
memcpy(profile->command_buffer.ptr, commands.Data(), profile->command_buffer.size);
aql_profile::PopulateAql(profile->command_buffer.ptr, profile->command_buffer.size,
cmd_writer, reinterpret_cast<aql_profile::packet_t*>(value));
break;
}
default:
status = HSA_STATUS_ERROR_INVALID_ARGUMENT;
ERR_LOGGING << "Invalid attribute (" << attribute << ")";
}
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
}
return status;
}
PUBLIC_API hsa_status_t
hsa_ven_amd_aqlprofile_iterate_event_ids(hsa_ven_amd_aqlprofile_eventname_callback_t callback) {
try {
EventDimension::init();
for (auto& [name, id] : EventDimension::dimension_table) callback(id, name.c_str());
} catch (...) {
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
PUBLIC_API hsa_status_t hsa_ven_amd_aqlprofile_iterate_event_coord(
hsa_agent_t agent, hsa_ven_amd_aqlprofile_event_t event, uint32_t sample_id,
hsa_ven_amd_aqlprofile_coordinate_callback_t callback, void* userdata) {
try {
const EventAttribDimension& attrib = EventAttribDimension::get(agent, event.block_name);
if (!attrib.get_num()) return HSA_STATUS_ERROR;
std::vector<uint8_t> coord;
coord.resize(attrib.get_num());
attrib.get_coordinates(coord.data(),
sample_id * attrib.get_num_instances() + event.block_index);
for (size_t i = 0; i < attrib.get_num(); i++) {
EventDimension dim = attrib.get_dim(i);
callback(i, dim.id, dim.extent, coord[i], dim.name.data(), userdata);
}
} catch (...) {
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
// Method for iterating the events output data
PUBLIC_API hsa_status_t
hsa_ven_amd_aqlprofile_iterate_data(const hsa_ven_amd_aqlprofile_profile_t* profile,
hsa_ven_amd_aqlprofile_data_callback_t callback, void* data) {
hsa_status_t status = HSA_STATUS_SUCCESS;
try {
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile);
const bool is_concurrent = pm4_factory->IsConcurrent();
const uint32_t xcc_num = pm4_factory->GetXccNumber();
const uint32_t se_number = pm4_factory->GetShaderEnginesNumber() / xcc_num;
if (profile->type == HSA_VEN_AMD_AQLPROFILE_EVENT_TYPE_PMC) {
uint64_t* samples = reinterpret_cast<uint64_t*>(profile->output_buffer.ptr);
for (const hsa_ven_amd_aqlprofile_event_t* p = profile->events;
p < profile->events + profile->event_count; ++p) {
if ((char*)samples >= (char*)profile->output_buffer.ptr + profile->output_buffer.size)
return HSA_STATUS_ERROR;
if (!(pm4_factory->GetBlockInfo(p)->attr & CounterBlockAidAttr)) continue;
// Process an MI300 UMC event for XCC 0 ONLY
auto sample_id = p->block_index; // sample id is the event block_index or the UMCCH id
hsa_ven_amd_aqlprofile_info_data_t sample_info;
sample_info.sample_id = sample_id;
sample_info.pmc_data.event = *p;
sample_info.pmc_data.result = *samples;
#if DEBUG_TRACE == 2
printf(
"DATA: sample index(%u) id(%u) bloc id(%u) index(%u) counter id(%u) "
"res(%lu)\n",
sample_index, sample_id, p->block_name, p->block_index, p->counter_id,
samples[sample_index]);
#endif
status = callback(HSA_VEN_AMD_AQLPROFILE_INFO_PMC_DATA, &sample_info, data);
if (status == HSA_STATUS_INFO_BREAK) {
status = HSA_STATUS_SUCCESS;
break;
}
if (status != HSA_STATUS_SUCCESS) break;
samples++;
}
for (uint32_t xcc_index = 0; xcc_index < xcc_num; xcc_index++) {
for (const hsa_ven_amd_aqlprofile_event_t* p = profile->events;
p < profile->events + profile->event_count; ++p) {
// this check needs to be the first check as it takes care of a corner case
// in which a UMC event is the last event in profile->events
if (pm4_factory->GetBlockInfo(p)->attr & CounterBlockAidAttr) continue;
if ((char*)samples > (char*)profile->output_buffer.ptr + profile->output_buffer.size)
return HSA_STATUS_ERROR;
// non-MI300A-AID counter event.
uint32_t block_samples_count = 1;
if (pm4_factory->GetBlockInfo(p)->attr & CounterBlockSeAttr)
block_samples_count *= se_number;
if (pm4_factory->GetBlockInfo(p)->attr & CounterBlockSaAttr)
block_samples_count *= 2;
if (pm4_factory->GetBlockInfo(p)->attr & CounterBlockWgpAttr)
block_samples_count *= pm4_factory->GetNumWGPs();
if (pm4_factory->GetBlockInfo(p)->attr & CounterBlockSqAttr && pm4_factory->IsGFX11())
block_samples_count *= pm4_factory->GetNumWGPs();
for (uint32_t blk = 0; blk < block_samples_count; ++blk) {
hsa_ven_amd_aqlprofile_info_data_t sample_info;
sample_info.sample_id = blk;
sample_info.pmc_data.event = *p;
#if DEBUG_TRACE == 2
printf("DATA: xcc(%u) id(%u) bloc id(%u) index(%u) counter id(%u) res(%lu)\n",
xcc_index, blk, p->block_name, p->block_index, p->counter_id, *samples);
#endif
sample_info.pmc_data.result = *samples;
status = callback(HSA_VEN_AMD_AQLPROFILE_INFO_PMC_DATA, &sample_info, data);
if (status == HSA_STATUS_INFO_BREAK) {
status = HSA_STATUS_SUCCESS;
break;
}
if (status != HSA_STATUS_SUCCESS) break;
samples++;
}
}
}
} else if (profile->type == HSA_VEN_AMD_AQLPROFILE_EVENT_TYPE_TRACE) {
uint32_t mode = 2;
switch (profile->event_count) {
case 0:
mode = 0;
break;
case UINT32_MAX:
const_cast<hsa_ven_amd_aqlprofile_profile_t*>(profile)->event_count = 0;
mode = 1;
break;
}
if (mode != 2) { // SQTT trace data, or SQTT pc sampling
auto& trace_config = aql_profile::configs.at(profile->command_buffer.ptr);
pm4_builder::SqttBuilder* sqttbuilder = pm4_factory->GetSqttBuilder();
const uint64_t se_number_total = pm4_factory->GetShaderEnginesNumber();
// Control buffer was allocated as the CmdBuffer prefix partition
aql_profile::CommandBufferMgr cmd_buffer_mgr(profile);
auto* control_ptr =
reinterpret_cast<pm4_builder::TraceControl*>(cmd_buffer_mgr.GetPrefix1());
// Check if SQTT buffer was wrapped
for (size_t se_index = 0; se_index < se_number_total; se_index++) {
if (control_ptr[se_index].status & sqttbuilder->GetUTCErrorMask()) {
ERR_LOGGING << "SQTT memory error received, SE(" << se_index << ")";
status = HSA_STATUS_ERROR_EXCEPTION;
} else if (control_ptr[se_index].status & sqttbuilder->GetBufferFullMask()) {
ERR2_LOGGING << "SQTT data buffer full, SE(" << se_index << ")";
if (status == HSA_STATUS_SUCCESS) status = HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
}
// The samples sizes are returned in the control buffer
for (size_t se_index = 0; se_index < se_number_total; se_index++) {
bool bMaskedIn = trace_config.GetTargetCU(se_index) >= 0;
uint64_t sample_capacity = trace_config.GetCapacity(se_index);
void* sample_ptr = reinterpret_cast<void*>(trace_config.GetSEBaseAddr(se_index));
// WPTR specifies the index in thread trace buffer where next token will be
// written by hardware. The index is incremented by size of 32 bytes.
size_t sample_size = (control_ptr[se_index].wptr & sqttbuilder->GetWritePtrMask()) *
sqttbuilder->GetWritePtrBlk();
if (pm4_factory->GetGpuId() == aql_profile::GFX11_GPU_ID) {
sample_size = sample_size - reinterpret_cast<uint64_t>(sample_ptr);
sample_size &= (1ull << 29) - 1;
}
if (sample_size >= sample_capacity) {
ERR_LOGGING << "SQTT data out of bounds, sample_id(" << se_index << ") size("
<< sample_size << "/" << sample_capacity << ")";
sample_size = sample_capacity;
if (status == HSA_STATUS_SUCCESS) status = HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
hsa_status_t call_status;
if (mode == 0) { // SQTT trace
if (bMaskedIn) {
hsa_ven_amd_aqlprofile_info_data_t info;
info.sample_id = se_index;
info.trace_data.ptr = sample_ptr;
info.trace_data.size = sample_size;
status = callback(HSA_VEN_AMD_AQLPROFILE_INFO_TRACE_DATA, &info, data);
}
} else { // PC sampling
pcsmp_callback_data_t* pcsmp_data = reinterpret_cast<pcsmp_callback_data_t*>(data);
pcsmp_data->id = se_index;
pcsmp_data->cycle = 333;
pcsmp_data->pc = 0x333;
call_status = callback(HSA_VEN_AMD_AQLPROFILE_INFO_TRACE_DATA, NULL, data);
}
}
} else { // SPM trace data
if (pm4_factory->SpmKfdMode() == false) {
const uint32_t tnumber = 1;
void* sample_ptr = profile->output_buffer.ptr;
const uint32_t sample_size = profile->output_buffer.size;
const uint32_t sample_capacity = (profile->output_buffer.size / tnumber);
for (unsigned i = 0; i < tnumber; ++i) {
hsa_ven_amd_aqlprofile_info_data_t sample_info;
sample_info.sample_id = i;
sample_info.trace_data.ptr = sample_ptr;
sample_info.trace_data.size = sample_size;
status = callback(HSA_VEN_AMD_AQLPROFILE_INFO_TRACE_DATA, &sample_info, data);
if (status == HSA_STATUS_INFO_BREAK) {
status = HSA_STATUS_SUCCESS;
break;
}
if (status != HSA_STATUS_SUCCESS) {
ERR_LOGGING << "SQTT data callback error, sample_id(" << i << ") status(" << status
<< ")";
break;
}
sample_ptr = reinterpret_cast<char*>(sample_ptr) + sample_capacity;
}
} else {
status = spm_kfd_namespace::spm_iterate_data(profile, callback, data);
}
}
} else {
ERR_LOGGING << "Bad profile type (" << profile->type << ")";
status = HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
}
return status;
}
// Method to populate the provided AQL packet with ATT Markers
PUBLIC_API hsa_status_t hsa_ven_amd_aqlprofile_att_marker(
hsa_ven_amd_aqlprofile_profile_t* profile, aql_profile::packet_t* aql_marker_packet,
uint32_t data, hsa_ven_amd_aqlprofile_att_marker_channel_t channel) {
assert(profile->type == HSA_VEN_AMD_AQLPROFILE_EVENT_TYPE_TRACE);
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile);
pm4_builder::SqttBuilder* sqtt_builder = pm4_factory->GetSqttBuilder();
pm4_builder::CmdBuilder* cmd_writer = pm4_factory->GetCmdBuilder();
pm4_builder::CmdBuffer commands;
// Generate start commands
auto status = sqtt_builder->InsertMarker(&commands, data, channel);
if (status != HSA_STATUS_SUCCESS) return status;
aql_profile::descriptor_t& cmdbuffer = profile->command_buffer;
size_t cmd_size = cmdbuffer.size;
cmdbuffer.size = commands.Size();
if (cmdbuffer.ptr == NULL) return HSA_STATUS_SUCCESS;
if (cmd_size < commands.Size()) return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
// Populate stop aql packet
memcpy(cmdbuffer.ptr, commands.Data(), commands.Size());
aql_profile::PopulateAql(cmdbuffer.ptr, commands.Size(), cmd_writer, aql_marker_packet);
return HSA_STATUS_SUCCESS;
}
} // extern "C"
+67
View File
@@ -0,0 +1,67 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#ifndef SRC_CORE_AQL_PROFILE_H_
#define SRC_CORE_AQL_PROFILE_H_
#include <hsa/hsa_ven_amd_aqlprofile.h>
#include <iostream>
#include <string>
#include "include/aql_profile_v2.h"
#include "core/aql_profile_exception.h"
#define PUBLIC_API __attribute__((visibility("default")))
namespace pm4_builder {
class CmdBuilder;
}
namespace aql_profile {
typedef hsa_ven_amd_aqlprofile_descriptor_t descriptor_t;
typedef hsa_ven_amd_aqlprofile_profile_t profile_t;
typedef hsa_ven_amd_aqlprofile_info_type_t info_type_t;
typedef hsa_ven_amd_aqlprofile_data_callback_t data_callback_t;
typedef hsa_ext_amd_aql_pm4_packet_t packet_t;
typedef hsa_ven_amd_aqlprofile_event_t event_t;
void PopulateAql(const void* cmd_buffer, uint32_t cmd_size, pm4_builder::CmdBuilder* cmd_writer,
packet_t* aql_packet);
void* LegacyAqlAcquire(const packet_t* aql_packet, void* data);
void* LegacyAqlRelease(const packet_t* aql_packet, void* data);
void* LegacyPm4(const packet_t* aql_packet, void* data);
class event_exception : public aql_profile_exc_val<event_t> {
public:
event_exception(const std::string& m, const event_t& ev) : aql_profile_exc_val(m, ev) {}
};
} // namespace aql_profile
static std::ostream& operator<<(std::ostream& os, const aql_profile::event_t& ev) {
os << "event( block(" << ev.block_name << "." << ev.block_index << "), Id(" << ev.counter_id
<< "))";
return os;
}
#endif // SRC_CORE_AQL_PROFILE_H_
+57
View File
@@ -0,0 +1,57 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#ifndef SRC_CORE_AQL_PROFILE_EXCEPTION_H_
#define SRC_CORE_AQL_PROFILE_EXCEPTION_H_
#include <string.h>
#include <sstream>
#include <string>
namespace aql_profile {
class aql_profile_exc_msg : public std::exception {
public:
explicit aql_profile_exc_msg(const std::string& msg) : str_(msg) {}
virtual const char* what() const throw() { return str_.c_str(); }
protected:
std::string str_;
};
template <typename T>
class aql_profile_exc_val : public std::exception {
public:
aql_profile_exc_val(const std::string& msg, const T& val) {
std::ostringstream oss;
oss << msg << "(" << val << ")";
str_ = oss.str();
}
virtual const char* what() const throw() { return str_.c_str(); }
protected:
std::string str_;
};
} // namespace aql_profile
#endif // SRC_CORE_AQL_PROFILE_EXCEPTION_H_
+187
View File
@@ -0,0 +1,187 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#pragma once
#include <cstdint>
#include <future>
#include <map>
#include <string>
#include <vector>
#include "core/aql_profile_exception.h"
#include "core/aql_profile_exception.h"
#include "core/aql_profile.hpp"
namespace aql_profile {
class CommandBufferMgr {
public:
struct info_t {
uint32_t prefix_size;
uint32_t rdcmds_size;
uint32_t rd2cmds_size;
uint32_t is_rd_fetch2;
uint32_t precmds_size;
uint32_t postcmds_size;
};
CommandBufferMgr(void* ptr, const uint32_t& size) { Init(descriptor_t{ptr, size}, false); }
explicit CommandBufferMgr(const profile_t* profile) { Init(profile->command_buffer, true); }
char* GetPrefix() { return reinterpret_cast<char*>(buffer_.ptr); }
char* GetPrefix1() { return reinterpret_cast<char*>(buffer_.ptr) + sizeof(info_t); }
char* AddPrefix(const uint32_t& delta) {
const uint32_t size = Align(delta);
char* ptr = (buffer_.ptr != NULL) ? GetPrefix() + info_.prefix_size : NULL;
info_.prefix_size += delta;
buffer_.size -= (size < buffer_.size) ? size : buffer_.size;
if (buffer_.size == 0)
throw aql_profile_exc_msg("CommandBufferMgr::AddPrefix(): buffer size set to zero");
return (buffer_.size != 0) ? ptr : NULL;
}
bool SetRdSize(const uint32_t& rd_data_size) {
const uint32_t size = Align(rd_data_size);
const bool suc = (size <= buffer_.size);
if (suc) {
info_.rdcmds_size = rd_data_size;
buffer_.size -= size;
}
if (!suc)
throw aql_profile_exc_msg("CommandBufferMgr::SetRdSize(): size set out of the buffer");
return suc;
}
bool SetRd2Size(const uint32_t& rd_data_size) {
const uint32_t size = Align(rd_data_size);
const bool suc = SetRdSize(Align(size));
if (suc) {
info_.rd2cmds_size = rd_data_size;
info_.rdcmds_size = 2 * size;
}
if (!suc)
throw aql_profile_exc_msg("CommandBufferMgr::SetRd2Size(): size set out of the buffer");
return suc;
}
bool SetPreSize(const uint32_t& pre_data_size) {
const uint32_t size = Align(pre_data_size);
const bool suc = (size <= buffer_.size);
if (suc) {
info_.precmds_size = pre_data_size;
buffer_.size -= size;
}
if (!suc)
throw aql_profile_exc_msg("CommandBufferMgr::SetPreSize(): size set out of the buffer");
return suc;
}
bool Finalize(const uint32_t& data_size) {
bool suc = (data_size > info_.precmds_size);
if (suc) {
const uint32_t post_data_size = data_size - info_.precmds_size;
const uint32_t size = Align(post_data_size);
suc = (size <= buffer_.size);
if (suc) {
info_.postcmds_size = post_data_size;
buffer_.size -= size;
}
if (!suc)
throw aql_profile_exc_msg("CommandBufferMgr::Finalize(): postcmd size is out of cmdbuffer");
}
if (!suc) throw aql_profile_exc_msg("CommandBufferMgr::Finalize(): postcmd size is zero");
if (info_slot_) *info_slot_ = info_;
return suc;
}
uint32_t GetSize() const { return GetEndOffset(); }
descriptor_t GetRdDescr() const {
descriptor_t descr;
descr.ptr = reinterpret_cast<char*>(buffer_.ptr) + GetRdOffset();
descr.size = info_.rdcmds_size;
return descr;
}
descriptor_t FetchRdDescr() {
descriptor_t descr;
if (info_.is_rd_fetch2 == 0) {
info_.is_rd_fetch2 = 1;
descr.ptr = reinterpret_cast<char*>(buffer_.ptr) + GetRdOffset();
} else {
descr.ptr = reinterpret_cast<char*>(buffer_.ptr) + GetRdOffset() + (info_.rdcmds_size / 2);
}
descr.size = info_.rd2cmds_size;
return descr;
}
descriptor_t GetPreDescr() const {
descriptor_t descr;
descr.ptr = reinterpret_cast<char*>(buffer_.ptr) + GetPreOffset();
descr.size = info_.precmds_size;
return descr;
}
descriptor_t GetPostDescr() const {
descriptor_t descr;
descr.ptr = reinterpret_cast<char*>(buffer_.ptr) + GetPostOffset();
descr.size = info_.postcmds_size;
return descr;
}
static uint32_t Align(const uint32_t& size) { return (size + align_mask_) & ~align_mask_; }
private:
void Init(const descriptor_t& buffer, const bool& import) {
buffer_ = buffer;
info_ = {};
info_slot_ = NULL;
uint32_t prefix_size = sizeof(info_t);
if (buffer_.ptr != NULL) {
info_slot_ = reinterpret_cast<info_t*>(GetPrefix());
if (import) {
prefix_size = info_slot_->prefix_size;
info_ = *info_slot_;
info_.prefix_size = 0;
}
} else {
buffer_.size = UINT_MAX;
}
AddPrefix(prefix_size);
}
uint32_t GetRdOffset() const { return Align(info_.prefix_size); }
uint32_t GetPreOffset() const { return GetRdOffset() + Align(info_.rdcmds_size); }
uint32_t GetPostOffset() const { return GetPreOffset() + Align(info_.precmds_size); }
uint32_t GetEndOffset() const { return GetPostOffset() + Align(info_.postcmds_size); }
static const uint32_t align_size_ = 0x100;
static const uint32_t align_mask_ = align_size_ - 1;
descriptor_t buffer_;
info_t info_;
info_t* info_slot_;
};
} // namespace aql_profile
+204
View File
@@ -0,0 +1,204 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#pragma once
#include <hsa/hsa_ven_amd_aqlprofile.h>
#include "def/gpu_block_info.h"
#include "core/aql_profile.hpp"
#include "core/pm4_factory.h"
#include <cstdint>
#include <string>
#include <vector>
#include <unordered_map>
#include <memory>
#include <array>
struct EventDimension {
EventDimension(const EventDimension& other) = default;
EventDimension(std::string_view _name, size_t _extent)
: id(dimension_table.at(std::string(_name))), name(_name), extent(_extent) {}
uint64_t id;
uint64_t extent;
std::string_view name;
static std::vector<std::string> dimension_list;
static std::unordered_map<std::string, size_t> dimension_table;
static void init() {
if (dimension_list.size()) return;
dimension_list.push_back("XCD");
dimension_list.push_back("AID");
dimension_list.push_back("SE");
dimension_list.push_back("SA");
dimension_list.push_back("WGP");
dimension_list.push_back("INSTANCE");
for (size_t i = 0; i < dimension_list.size(); i++) dimension_table[dimension_list[i]] = i;
}
};
class EventKey {
public:
uint64_t agent;
uint64_t block;
bool operator==(const EventKey& other) const {
return agent == other.agent && block == other.block;
}
bool operator!=(const EventKey& other) const { return !(*this == other); }
};
template <>
struct std::hash<EventKey> {
uint64_t operator()(const EventKey& ev) const {
return ev.agent | (ev.block << 56) | (ev.block >> 8);
}
};
class EventAttribDimension {
public:
static constexpr size_t event_id_bit = 24;
template <typename AgentType>
EventAttribDimension(AgentType agent, hsa_ven_amd_aqlprofile_block_name_t block_name)
: key({agent.handle, (uint64_t)block_name}) {
EventDimension::init();
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(agent);
this->block_info = pm4_factory->GetBlockInfo(block_name);
bIsGFX12 = pm4_factory->IsGFX12();
bIsGFX11 = pm4_factory->IsGFX11();
bIsGFX9 = pm4_factory->IsGFX9();
num_xccs = pm4_factory->GetXccNumber();
if (num_xccs > 1 && HasAttr(CounterBlockUmcAttr)) { // For MI300 AID only
num_xccs = 1;
num_aid = 4;
}
shader_engine = HasAttr(CounterBlockSeAttr);
shader_array = HasAttr(CounterBlockSaAttr);
if (bIsGFX9)
compute_unit = HasAttr(CounterBlockTcAttr) && shader_engine;
else if (bIsGFX11 || bIsGFX12)
workgroup_processor = HasAttr(CounterBlockSqAttr);
se_num = pm4_factory->GetShaderEnginesNumber();
sarrays = pm4_factory->GetShaderArraysNumber() * se_num;
cu_num = (pm4_factory->GetComputeUnitNumber() + sarrays - 1) / sarrays;
wgp_num = (pm4_factory->GetComputeUnitNumber() / 2 + sarrays - 1) / sarrays;
if (HasAttr(CounterBlockUmcAttr))
block_instance_count = block_info->instance_count / num_aid;
else if (compute_unit)
block_instance_count = std::min<size_t>(block_info->instance_count, cu_num + 1);
else
block_instance_count = block_info->instance_count;
if (num_xccs > 1) dimensions.push_back({"XCD", num_xccs});
if (num_aid > 1) dimensions.push_back({"AID", num_aid});
if (workgroup_processor)
dimensions.push_back({"WGP", wgp_num});
else
dimensions.push_back({"INSTANCE", block_instance_count});
if (shader_engine)
dimensions.push_back(
{"SE", pm4_factory->GetShaderEnginesNumber() / (num_xccs > 0 ? num_xccs : 1)});
if (shader_array) dimensions.push_back({"SA", pm4_factory->GetShaderArraysNumber()});
}
size_t get_num_xccs() const { return num_xccs; };
size_t get_total_elements() const {
size_t acc = 1;
for (auto& d : dimensions) acc *= d.extent;
return acc;
}
uint64_t get_num() const { return dimensions.size(); };
EventDimension get_dim(uint64_t index) const { return dimensions.at(index); };
hsa_status_t get_coordinates(uint8_t* coordinates, int64_t cumulative_id) const {
const int end = static_cast<int>(get_num()) - 1;
for (int i = end; i >= 0; i--) {
coordinates[i] = static_cast<uint8_t>(cumulative_id % dimensions.at(i).extent);
cumulative_id /= dimensions.at(i).extent;
}
if (cumulative_id != 0) return HSA_STATUS_ERROR_INVALID_INDEX;
return HSA_STATUS_SUCCESS;
}
size_t get_num_instances() const { return block_instance_count; }
private:
bool HasAttr(CounterBlockAttr attr) const { return (block_info->attr & attr) != 0; }
EventKey key;
const GpuBlockInfo* block_info = nullptr;
hsa_ven_amd_aqlprofile_event_t event{};
bool bIsGFX12;
bool bIsGFX11;
bool bIsGFX9;
bool shader_engine = false;
bool shader_array = false;
bool compute_unit = false;
bool workgroup_processor = false;
size_t num_xccs = 1;
size_t num_aid = 1;
size_t se_num = 1;
size_t sarrays = 1;
size_t cu_num = 1;
size_t wgp_num = 1;
size_t block_instance_count = 1;
std::vector<EventDimension> dimensions;
public:
template <typename AgentType>
static const EventAttribDimension& get(AgentType agent,
hsa_ven_amd_aqlprofile_block_name_t block_name) {
thread_local std::unordered_map<EventKey, std::shared_ptr<EventAttribDimension>> event_map{};
thread_local std::shared_ptr<EventAttribDimension> event_cache{nullptr};
EventKey key{agent.handle, (uint64_t)block_name};
if (!event_cache || event_cache->key != key) {
auto it = event_map.find(key);
if (auto it = event_map.find(key); it != event_map.end())
event_cache = it->second;
else
event_cache =
event_map.emplace(key, std::make_shared<EventAttribDimension>(agent, block_name))
.first->second;
}
return *event_cache;
}
};
+447
View File
@@ -0,0 +1,447 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/aql_profile.hpp"
#include "core/include/aql_profile_v2.h"
#include <array>
#include <cstdint>
#include <future>
#include <map>
#include <string>
#include <vector>
#include "core/counter_dimensions.hpp"
#include "core/logger.h"
#include "core/pm4_factory.h"
#include "pm4/cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/spm_builder.h"
#include "pm4/sqtt_builder.h"
#include "core/commandbuffermgr.hpp"
#include "memorymanager.hpp"
#define PUBLIC_API __attribute__((visibility("default")))
#define CONSTRUCTOR_API __attribute__((constructor))
#define DESTRUCTOR_API __attribute__((destructor))
#define ERR_CHECK(cond, err, msg) \
{ \
if (cond) { \
ERR_LOGGING << msg; \
return err; \
} \
}
#define HSA_TRY_WRAP try {
#define HSA_CATCH_WRAP \
} \
catch (std::exception & e) { \
return HSA_STATUS_ERROR; \
}
std::vector<std::string> EventDimension::dimension_list;
std::unordered_map<std::string, size_t> EventDimension::dimension_table;
namespace aql_profile_v2 {
// Command buffer partitioning manager
// Supports Pre/Post commands partitioning
// and prefix control partition
using aql_profile::event_exception;
using aql_profile::event_t;
using ::aql_profile::Pm4Factory;
uint32_t HandleSQFlagsBlock(Pm4Factory* pm4_factory, const aqlprofile_pmc_event_t& event) {
auto visible_id = event.event_id;
if (event.flags.sq_flags.accum == AQLPROFILE_ACCUMULATION_LO_RES)
visible_id = pm4_factory->GetAccumLowID();
if (event.flags.sq_flags.accum == AQLPROFILE_ACCUMULATION_HI_RES)
visible_id = pm4_factory->GetAccumHiID();
return visible_id;
}
counter_des_t GetCounter(Pm4Factory* pm4_factory, EventRequest& event,
std::map<block_des_t, uint32_t, lt_block_des>& index_map) {
const GpuBlockInfo* block_info = pm4_factory->GetBlockInfo(event.block_name);
const block_des_t block_des = {block_info->id, event.block_index};
const auto ret = index_map.insert({block_des, 0});
auto reg_index = ret.first->second;
auto visible_id = event.event_id;
if (pm4_builder::SPISkip(block_info->attr, visible_id)) {
event.bInternal = true;
return {visible_id, reg_index, block_des, block_info};
}
if (reg_index >= block_info->counter_count)
throw std::string("Event is out of block counter registers number limit");
if (event.flags.raw) {
if (event.block_name == HSA_VEN_AMD_AQLPROFILE_BLOCK_NAME_SQ) {
visible_id = HandleSQFlagsBlock(pm4_factory, event);
} else {
throw HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
}
ret.first->second++;
return {visible_id, reg_index, block_des, block_info};
}
pm4_builder::counters_vector CountersVec(std::vector<EventRequest>& events,
Pm4Factory* pm4_factory) {
pm4_builder::counters_vector vec;
std::map<block_des_t, uint32_t, lt_block_des> index_map;
for (auto& event : events) vec.push_back(GetCounter(pm4_factory, event, index_map));
if (pm4_factory->IsGFX10() && (vec.get_attr() & CounterBlockGRBMAttr) == 0) {
EventRequest grbm_event{0};
grbm_event.block_name = HSA_VEN_AMD_AQLPROFILE_BLOCK_NAME_GRBM;
vec.push_back(GetCounter(pm4_factory, grbm_event, index_map));
}
return vec;
}
// Method for iterating the events output data
hsa_status_t _internal_aqlprofile_pmc_iterate_data(aqlprofile_handle_t handle,
aqlprofile_pmc_data_callback_t callback,
void* userdata) {
auto counter_memorymgr = MemoryManager::GetManager(handle.handle);
CounterMemoryManager* memorymgr = dynamic_cast<CounterMemoryManager*>(counter_memorymgr.get());
if (!memorymgr) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(memorymgr->AgentHandle());
const uint32_t xcc_num = pm4_factory->GetXccNumber();
uint64_t* samples = reinterpret_cast<uint64_t*>(memorymgr->GetOutputBuf());
uint64_t* buffer_end_location = samples + memorymgr->GetOutputBufSize() / sizeof(uint64_t);
auto& events = memorymgr->GetEvents();
size_t umc_sample_id = 0;
if (xcc_num > 1)
for (auto& event : events) {
if (samples >= buffer_end_location) return HSA_STATUS_ERROR;
if (!(pm4_factory->GetBlockInfo(event.block_name)->attr & CounterBlockUmcAttr)) continue;
#if DEBUG_TRACE == 2
printf("DATA: sample index(%u) id(%u) bloc id(%u) index(%u) counter id(%u) res(%lu)\n",
sample_index, sample_id, p->block_name, p->block_index, p->counter_id, *samples);
#endif
hsa_status_t status = callback(event, event.block_index, *samples, userdata);
samples++;
umc_sample_id++;
if (status == HSA_STATUS_INFO_BREAK) return HSA_STATUS_SUCCESS;
if (status != HSA_STATUS_SUCCESS) return status;
}
size_t xcc_sample_count = 0;
for (uint32_t xcc_index = 0; xcc_index < xcc_num; xcc_index++)
for (auto& event : events) {
if (samples >= buffer_end_location) return HSA_STATUS_ERROR;
if (pm4_factory->GetBlockInfo(event.block_name)->attr & CounterBlockUmcAttr) continue;
// non-MI300A-AID counter event.
uint32_t block_samples_count = pm4_factory->GetNumEvents(event.block_name);
for (uint32_t blk = 0; blk < block_samples_count; ++blk) {
#if DEBUG_TRACE == 2
printf("DATA: xcc(%u) blk(%u) bloc id(%u) index(%u) counter id(%u) res(%lu)\n", xcc_index,
blk, event.block_name, event.block_index, event.event_id, *samples);
#endif
xcc_sample_count += xcc_index == 0;
size_t xcc_sample_id = xcc_sample_count * xcc_index +
static_cast<size_t>(event.block_index) * block_samples_count + blk;
if (!event.bInternal) {
hsa_status_t status = callback(event, xcc_sample_id, *samples, userdata);
if (status == HSA_STATUS_INFO_BREAK)
return HSA_STATUS_SUCCESS;
else if (status != HSA_STATUS_SUCCESS)
return status;
}
samples++;
}
}
return HSA_STATUS_SUCCESS;
}
hsa_status_t _internal_aqlprofile_pmc_create_packets(
aqlprofile_handle_t* handle, aqlprofile_pmc_aql_packets_t* packets,
aqlprofile_pmc_profile_t profile, aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb, aqlprofile_memory_copy_t memcpy_cb,
void* userdata) {
pm4_builder::CmdBuffer commands;
auto memorymgr =
std::make_shared<CounterMemoryManager>(profile.agent, alloc_cb, dealloc_cb, userdata);
MemoryManager::RegisterManager(memorymgr);
memorymgr->CopyEvents(profile.events, profile.event_count);
pm4_builder::CmdBuffer read_cmd;
pm4_builder::CmdBuffer start_cmd;
pm4_builder::CmdBuffer stop_cmd;
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile.agent);
const pm4_builder::counters_vector countersVec = CountersVec(memorymgr->GetEvents(), pm4_factory);
pm4_builder::PmcBuilder* pmc_builder = pm4_factory->GetPmcBuilder();
// Start outputbuf ptr
size_t output_bytes = 8; // Extra space for GRBM block on gfx10
for (auto& event : memorymgr->GetEvents())
output_bytes += pm4_factory->GetBytesNeeded(event.block_name);
memorymgr->CreateOutputBuf(output_bytes);
// Generate read commands
size_t data_size = pmc_builder->Read(&read_cmd, countersVec, memorymgr->GetOutputBuf());
// Generate start commands
pmc_builder->Start(&start_cmd, countersVec);
// Generate stop commands
pmc_builder->Stop(&stop_cmd, countersVec);
ERR_CHECK(data_size == 0, HSA_STATUS_ERROR, "PMC Builder Stop(): data size set to zero");
if (memorymgr->GetOutputBufSize() < data_size) return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
// Copy generated commands
size_t start_size = aql_profile::CommandBufferMgr::Align(start_cmd.Size());
size_t stop_size = aql_profile::CommandBufferMgr::Align(stop_cmd.Size());
size_t read_size = aql_profile::CommandBufferMgr::Align(read_cmd.Size());
memorymgr->CreateCmdBuf(start_size + stop_size + read_size);
handle->handle = memorymgr->GetHandler();
pm4_builder::CmdBuilder* cmd_writer = pm4_factory->GetCmdBuilder();
uint8_t* cmdbuf = reinterpret_cast<uint8_t*>(memorymgr->GetCmdBuf());
memcpy_cb(cmdbuf, read_cmd.Data(), read_cmd.Size(), userdata);
aql_profile::PopulateAql(cmdbuf, read_cmd.Size(), cmd_writer, &packets->read_packet);
cmdbuf += read_size;
memcpy_cb(cmdbuf, start_cmd.Data(), start_cmd.Size(), userdata);
aql_profile::PopulateAql(cmdbuf, start_cmd.Size(), cmd_writer, &packets->start_packet);
cmdbuf += start_size;
memcpy_cb(cmdbuf, stop_cmd.Data(), stop_cmd.Size(), userdata);
aql_profile::PopulateAql(cmdbuf, stop_cmd.Size(), cmd_writer, &packets->stop_packet);
return HSA_STATUS_SUCCESS;
}
} // namespace aql_profile_v2
extern "C" {
PUBLIC_API hsa_status_t aqlprofile_pmc_create_packets(
aqlprofile_handle_t* handle, aqlprofile_pmc_aql_packets_t* packets,
aqlprofile_pmc_profile_t profile, aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb, aqlprofile_memory_copy_t memcpy_cb,
void* userdata) {
try {
return aql_profile_v2::_internal_aqlprofile_pmc_create_packets(
handle, packets, profile, alloc_cb, dealloc_cb, memcpy_cb, userdata);
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
} catch (...) {
return HSA_STATUS_ERROR;
}
}
PUBLIC_API void aqlprofile_pmc_delete_packets(aqlprofile_handle_t handle) {
try {
MemoryManager::DeleteManager(handle.handle);
} catch (std::exception& e) {
return;
} catch (...) {
return;
}
}
PUBLIC_API hsa_status_t aqlprofile_pmc_iterate_data(aqlprofile_handle_t handle,
aqlprofile_pmc_data_callback_t callback,
void* userdata) {
try {
return aql_profile_v2::_internal_aqlprofile_pmc_iterate_data(handle, callback, userdata);
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
} catch (...) {
return HSA_STATUS_ERROR;
}
}
PUBLIC_API hsa_status_t aqlprofile_iterate_event_ids(aqlprofile_eventname_callback_t callback,
void* user_data) {
try {
EventDimension::init();
for (auto& [name, id] : EventDimension::dimension_table) {
if (auto ret = callback(id, name.c_str(), user_data); ret != HSA_STATUS_SUCCESS) {
return ret;
}
}
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (...) {
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
PUBLIC_API hsa_status_t aqlprofile_iterate_event_coord(aqlprofile_agent_handle_t agent,
aqlprofile_pmc_event_t event,
uint64_t counter_id,
aqlprofile_coordinate_callback_t callback,
void* userdata) {
try {
const EventAttribDimension& attrib = EventAttribDimension::get(agent, event.block_name);
if (!attrib.get_num()) return HSA_STATUS_ERROR;
std::array<uint8_t, 32> coord;
assert(attrib.get_num() < coord.size());
attrib.get_coordinates(coord.data(), counter_id);
for (size_t i = 0; i < attrib.get_num(); i++) {
EventDimension dim = attrib.get_dim(i);
callback(i, dim.id, dim.extent, coord.at(i), dim.name.data(), userdata);
}
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (...) {
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
PUBLIC_API hsa_status_t aqlprofile_register_agent(aqlprofile_agent_handle_t* agent_id,
const aqlprofile_agent_info_t* agent_info) {
return aqlprofile_register_agent_info(agent_id, agent_info, AQLPROFILE_AGENT_VERSION_V0);
}
PUBLIC_API hsa_status_t aqlprofile_register_agent_info(aqlprofile_agent_handle_t* agent_id,
const void* agent_info,
aqlprofile_agent_version_t version) {
try {
if (agent_info == NULL) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
switch (version) {
case AQLPROFILE_AGENT_VERSION_V0: {
const auto* info = static_cast<const aqlprofile_agent_info_t*>(agent_info);
aqlprofile_agent_info_v1_t info_v1 = {
.agent_gfxip = info->agent_gfxip,
.xcc_num = info->xcc_num,
.se_num = info->se_num,
.cu_num = info->cu_num,
.shader_arrays_per_se = info->shader_arrays_per_se,
.domain = 0,
.location_id = 0,
};
*agent_id = aql_profile::RegisterAgent(&info_v1);
} break;
case AQLPROFILE_AGENT_VERSION_V1: {
*agent_id =
aql_profile::RegisterAgent(static_cast<const aqlprofile_agent_info_v1_t*>(agent_info));
} break;
case AQLPROFILE_AGENT_VERSION_NONE:
case AQLPROFILE_AGENT_VERSION_LAST:
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (...) {
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
// Check if event is valid for the specific GPU
PUBLIC_API hsa_status_t aqlprofile_validate_pmc_event(aqlprofile_agent_handle_t agent,
const aqlprofile_pmc_event_t* event,
bool* result) {
hsa_status_t status = HSA_STATUS_SUCCESS;
*result = false;
try {
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(agent);
if (pm4_factory->GetBlockInfo(event) != NULL) *result = true;
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (...) {
return HSA_STATUS_ERROR;
}
return status;
}
PUBLIC_API hsa_status_t aqlprofile_get_pmc_info(const aqlprofile_pmc_profile_t* profile,
aqlprofile_pmc_info_type_t attribute, void* value) {
if (!profile) return HSA_STATUS_ERROR;
try {
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile->agent);
switch (attribute) {
case AQLPROFILE_INFO_BLOCK_ID: {
hsa_ven_amd_aqlprofile_id_query_t* query =
reinterpret_cast<hsa_ven_amd_aqlprofile_id_query_t*>(value);
const uint32_t block = pm4_factory->FindBlock(query->name);
const GpuBlockInfo* info = pm4_factory->GetBlockInfo(block);
if (!info) return HSA_STATUS_ERROR;
const auto& attrib =
EventAttribDimension::get(profile->agent, (hsa_ven_amd_aqlprofile_block_name_t)block);
if (!attrib.get_num()) return HSA_STATUS_ERROR;
query->id = block;
query->instance_count = attrib.get_num_instances();
} break;
case AQLPROFILE_INFO_BLOCK_COUNTERS: {
*reinterpret_cast<uint32_t*>(value) =
pm4_factory->GetBlockInfo(&profile->events[0])->counter_count;
} break;
default:
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (...) {
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
} // extern "C"
+107
View File
@@ -0,0 +1,107 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/pm4_factory.h"
#include "def/gfx10_def.h"
#include "pm4/gfx10_cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/sqtt_builder.h"
namespace aql_profile {
// Gfx10 factory class
class Gfx10Factory : public Pm4Factory {
public:
explicit Gfx10Factory(const AgentInfo* agent_info)
: Pm4Factory(BlockInfoMap(block_table_, sizeof(block_table_))) {
Init(agent_info);
}
Gfx10Factory(const GpuBlockInfo** table, const uint32_t& size, const AgentInfo* agent_info)
: Pm4Factory(BlockInfoMap(table, size)) {
Init(agent_info);
}
bool IsGFX10() const override { return true; }
virtual int GetAccumLowID() const override { return 1; };
virtual int GetAccumHiID() const override { return 1; };
protected:
// void ConstructTable(const AgentInfo* agent_info);
void Init(const AgentInfo* agent_info);
// void ConstructBuilders(const AgentInfo* agent_info);
static const GpuBlockInfo* block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER];
};
// Gfx builders init
// void Gfx10Factory::ConstructBuilders(const AgentInfo* agent_info) {
void Gfx10Factory::Init(const AgentInfo* agent_info) {
Pm4Factory::cmd_builder_ = new pm4_builder::Gfx10CmdBuilder(nullptr);
if (Pm4Factory::cmd_builder_ == NULL) throw aql_profile_exc_msg("CmdBuilder allocation failed");
// Mark and set the mode
if (Pm4Factory::IsConcurrent()) {
Pm4Factory::pmc_builder_ =
new pm4_builder::GpuPmcBuilder<pm4_builder::Gfx10CmdBuilder, gfx10_cntx_prim, true>(
agent_info);
} else {
Pm4Factory::pmc_builder_ =
new pm4_builder::GpuPmcBuilder<pm4_builder::Gfx10CmdBuilder, gfx10_cntx_prim, false>(
agent_info);
}
if (Pm4Factory::pmc_builder_ == NULL) throw aql_profile_exc_msg("PmcBuilder allocation failed");
Pm4Factory::spm_builder_ =
new pm4_builder::GpuSpmBuilder<pm4_builder::Gfx10CmdBuilder, gfx10_cntx_prim>(agent_info);
if (Pm4Factory::spm_builder_ == NULL) throw aql_profile_exc_msg("SpmBuilder allocation failed");
Pm4Factory::sqtt_builder_ =
new pm4_builder::GpuSqttBuilder<pm4_builder::Gfx10CmdBuilder, gfx10_cntx_prim>(agent_info);
if (Pm4Factory::sqtt_builder_ == NULL) throw aql_profile_exc_msg("SqttBuilder allocation failed");
agent_info_ = agent_info;
}
// GFX10 block table
const GpuBlockInfo* Gfx10Factory::block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER] = {
&CpcCounterBlockInfo, &CpfCounterBlockInfo, &GdsCounterBlockInfo, &GrbmCounterBlockInfo,
NULL /*&GrbmSeCounterBlockInfo*/, &SpiCounterBlockInfo, &SqCounterBlockInfo,
NULL /*&SqCsCounterBlockInfo*/, NULL /*GFX8 SRBM*/, &SxCounterBlockInfo, &TaCounterBlockInfo,
NULL /*&TcaCounterBlockInfo*/, NULL /*&TccCounterBlockInfo*/, NULL /*&TcpCounterBlockInfo*/,
NULL /*&TdCounterBlockInfo*/,
// MC blocks
NULL /*MC_ARB*/, NULL /*MC_HUB*/, NULL /*MC_MCBVM*/, NULL /*MC_SEQ*/,
NULL /*&McVmL2CounterBlockInfo*/, NULL /*MC_XBAR*/, NULL /*&AtcCounterBlockInfo*/,
NULL /*&AtcL2CounterBlockInfo*/, &GceaCounterBlockInfo, NULL /*&RpbCounterBlockInfo*/,
// System blocks
NULL /*&SdmaCounterBlockInfo*/,
// new navi blocks
&Gl1aCounterBlockInfo, &Gl1cCounterBlockInfo, &Gl2aCounterBlockInfo, &Gl2cCounterBlockInfo,
&GcrCounterBlockInfo, &GusCounterBlockInfo};
// Pm4Factory create mathods
Pm4Factory* Pm4Factory::Gfx10Create(const AgentInfo* agent_info) {
auto p = new Gfx10Factory(agent_info);
if (p == NULL) throw aql_profile_exc_msg("Gfx10Factory allocation failed");
return p;
}
} // namespace aql_profile
+107
View File
@@ -0,0 +1,107 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/pm4_factory.h"
#include "def/gfx11_def.h"
#include "pm4/gfx11_cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/sqtt_builder.h"
namespace aql_profile {
// Gfx11 factory class
class Gfx11Factory : public Pm4Factory {
public:
explicit Gfx11Factory(const AgentInfo* agent_info)
: Pm4Factory(BlockInfoMap(block_table_, sizeof(block_table_))) {
Init(agent_info);
}
Gfx11Factory(const GpuBlockInfo** table, const uint32_t& size, const AgentInfo* agent_info)
: Pm4Factory(BlockInfoMap(table, size)) {
Init(agent_info);
}
bool IsGFX11() const override { return true; }
virtual int GetAccumLowID() const override { return 1; };
virtual int GetAccumHiID() const override { return 1; };
protected:
// void ConstructTable(const AgentInfo* agent_info);
void Init(const AgentInfo* agent_info);
// void ConstructBuilders(const AgentInfo* agent_info);
static const GpuBlockInfo* block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER];
};
// Gfx builders init
// void Gfx11Factory::ConstructBuilders(const AgentInfo* agent_info) {
void Gfx11Factory::Init(const AgentInfo* agent_info) {
Pm4Factory::cmd_builder_ = new pm4_builder::Gfx11CmdBuilder(nullptr);
if (Pm4Factory::cmd_builder_ == NULL) throw aql_profile_exc_msg("CmdBuilder allocation failed");
// Mark and set the mode
if (Pm4Factory::IsConcurrent()) {
Pm4Factory::pmc_builder_ =
new pm4_builder::GpuPmcBuilder<pm4_builder::Gfx11CmdBuilder, gfx11_cntx_prim, true>(
agent_info);
} else {
Pm4Factory::pmc_builder_ =
new pm4_builder::GpuPmcBuilder<pm4_builder::Gfx11CmdBuilder, gfx11_cntx_prim, false>(
agent_info);
}
if (Pm4Factory::pmc_builder_ == NULL) throw aql_profile_exc_msg("PmcBuilder allocation failed");
Pm4Factory::spm_builder_ =
new pm4_builder::GpuSpmBuilder<pm4_builder::Gfx11CmdBuilder, gfx11_cntx_prim>(agent_info);
if (Pm4Factory::spm_builder_ == NULL) throw aql_profile_exc_msg("SpmBuilder allocation failed");
Pm4Factory::sqtt_builder_ =
new pm4_builder::GpuSqttBuilder<pm4_builder::Gfx11CmdBuilder, gfx11_cntx_prim>(agent_info);
if (Pm4Factory::sqtt_builder_ == NULL) throw aql_profile_exc_msg("SqttBuilder allocation failed");
agent_info_ = agent_info;
}
// GFX11 block table
const GpuBlockInfo* Gfx11Factory::block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER] = {
&CpcCounterBlockInfo, &CpfCounterBlockInfo, &GdsCounterBlockInfo, &GrbmCounterBlockInfo,
NULL /*&GrbmSeCounterBlockInfo*/, &SpiCounterBlockInfo, &SqCounterBlockInfo,
NULL /*&SqCsCounterBlockInfo*/, NULL /*GFX8 SRBM*/, &SxCounterBlockInfo, &TaCounterBlockInfo,
NULL /*&TcaCounterBlockInfo*/, NULL /*&TccCounterBlockInfo*/, &TcpCounterBlockInfo,
NULL /*&TdCounterBlockInfo*/,
// MC blocks
NULL /*MC_ARB*/, NULL /*MC_HUB*/, NULL /*MC_MCBVM*/, NULL /*MC_SEQ*/,
NULL /*&McVmL2CounterBlockInfo*/, NULL /*MC_XBAR*/, NULL /*&AtcCounterBlockInfo*/,
NULL /*&AtcL2CounterBlockInfo*/, &GceaCounterBlockInfo, NULL /*&RpbCounterBlockInfo*/,
// System blocks
NULL /*&SdmaCounterBlockInfo*/,
// new navi blocks
&Gl1aCounterBlockInfo, &Gl1cCounterBlockInfo, &Gl2aCounterBlockInfo, &Gl2cCounterBlockInfo,
&GcrCounterBlockInfo, &GusCounterBlockInfo};
// Pm4Factory create mathods
Pm4Factory* Pm4Factory::Gfx11Create(const AgentInfo* agent_info) {
auto p = new Gfx11Factory(agent_info);
if (p == NULL) throw aql_profile_exc_msg("Gfx11Factory allocation failed");
return p;
}
} // namespace aql_profile
+116
View File
@@ -0,0 +1,116 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/pm4_factory.h"
#include "def/gfx12_def.h"
#include "pm4/gfx12_cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/sqtt_builder.h"
namespace aql_profile {
// Gfx12 factory class
class Gfx12Factory : public Pm4Factory {
public:
explicit Gfx12Factory(const AgentInfo* agent_info)
: Pm4Factory(BlockInfoMap(block_table_, sizeof(block_table_))) {
Init(agent_info);
}
Gfx12Factory(const GpuBlockInfo** table, const uint32_t& size, const AgentInfo* agent_info)
: Pm4Factory(BlockInfoMap(table, size)) {
Init(agent_info);
}
bool IsGFX12() const override { return true; }
protected:
void ConstructBuilders(const AgentInfo* agent_info);
void ConstructTable(const AgentInfo* agent_info);
void Init(const AgentInfo* agent_info) {
agent_info_ = agent_info;
ConstructBuilders(agent_info);
ConstructTable(agent_info);
}
const GpuBlockInfo* block_table_[LastCounterBlockId + 1]{};
};
void Gfx12Factory::ConstructBuilders(const AgentInfo* agent_info) {
Pm4Factory::cmd_builder_ = new pm4_builder::Gfx12CmdBuilder(nullptr);
if (Pm4Factory::cmd_builder_ == NULL) throw aql_profile_exc_msg("CmdBuilder allocation failed");
// Mark and set the mode
if (Pm4Factory::IsConcurrent()) {
Pm4Factory::pmc_builder_ =
new pm4_builder::GpuPmcBuilder<pm4_builder::Gfx12CmdBuilder, gfx12_cntx_prim, true>(
agent_info);
} else {
Pm4Factory::pmc_builder_ =
new pm4_builder::GpuPmcBuilder<pm4_builder::Gfx12CmdBuilder, gfx12_cntx_prim, false>(
agent_info);
}
if (Pm4Factory::pmc_builder_ == NULL) throw aql_profile_exc_msg("PmcBuilder allocation failed");
Pm4Factory::spm_builder_ =
new pm4_builder::GpuSpmBuilder<pm4_builder::Gfx12CmdBuilder, gfx12_cntx_prim>(agent_info);
if (Pm4Factory::spm_builder_ == NULL) throw aql_profile_exc_msg("SpmBuilder allocation failed");
Pm4Factory::sqtt_builder_ =
new pm4_builder::GpuSqttBuilder<pm4_builder::Gfx12CmdBuilder, gfx12_cntx_prim>(agent_info);
if (Pm4Factory::sqtt_builder_ == NULL) throw aql_profile_exc_msg("SqttBuilder allocation failed");
}
void Gfx12Factory::ConstructTable(const AgentInfo* agent_info) {
// Global blocks
block_table_[__BLOCK_ID(CHA)] = &ChaCounterBlockInfo;
block_table_[__BLOCK_ID(CHC)] = &ChcCounterBlockInfo;
block_table_[__BLOCK_ID(CPC)] = &CpcCounterBlockInfo;
block_table_[__BLOCK_ID(CPF)] = &CpfCounterBlockInfo;
block_table_[__BLOCK_ID(CPG)] = &CpgCounterBlockInfo;
block_table_[__BLOCK_ID(GCEA)] = &GceaCounterBlockInfo;
block_table_[__BLOCK_ID(GCR)] = &GcrCounterBlockInfo;
block_table_[__BLOCK_ID(GL2A)] = &Gl2aCounterBlockInfo;
block_table_[__BLOCK_ID(GL2C)] = &Gl2cCounterBlockInfo;
block_table_[__BLOCK_ID(GRBM)] = &GrbmCounterBlockInfo;
block_table_[__BLOCK_ID(RLC)] = &RlcCounterBlockInfo;
block_table_[__BLOCK_ID(SDMA_PM)] = &SdmaPmCounterBlockInfo;
// SE blocks
block_table_[__BLOCK_ID(GCEA_SE)] = &GceaSeCounterBlockInfo;
block_table_[__BLOCK_ID(GRBMH)] = &GrbmhCounterBlockInfo;
block_table_[__BLOCK_ID(SPI)] = &SpiCounterBlockInfo;
block_table_[__BLOCK_ID(SQ)] = &SqcCounterBlockInfo;
block_table_[__BLOCK_ID(GC_UTCL1)] = &GcUtcl1CounterBlockInfo;
// SA blocks
block_table_[__BLOCK_ID(GL1A)] = &Gl1aCounterBlockInfo;
block_table_[__BLOCK_ID(GL1C)] = &Gl1cCounterBlockInfo;
// WGP blocks
block_table_[__BLOCK_ID(TA)] = &TaCounterBlockInfo;
block_table_[__BLOCK_ID(TCP)] = &TcpCounterBlockInfo;
block_table_[__BLOCK_ID(TD)] = &TdCounterBlockInfo;
}
// Pm4Factory create mathods
Pm4Factory* Pm4Factory::Gfx12Create(const AgentInfo* agent_info) {
auto p = new Gfx12Factory(agent_info);
if (p == NULL) throw aql_profile_exc_msg("Gfx12Factory allocation failed");
return p;
}
} // namespace aql_profile
+81
View File
@@ -0,0 +1,81 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/gfx9_factory.h"
#include "def/gfx908_def.h"
#include "pm4/gfx9_cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/sqtt_builder.h"
namespace aql_profile {
const GpuBlockInfo* Mi100Factory::block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER] = {};
Mi100Factory::Mi100Factory(const AgentInfo* agent_info)
: Gfx9Factory(block_table_, sizeof(block_table_), agent_info) {
for (unsigned i = 0; i < HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER; ++i) {
const GpuBlockInfo* base_table_ptr = Gfx9Factory::block_table_[i];
if (base_table_ptr == NULL) continue;
GpuBlockInfo* block_info = nullptr;
if (i == HSA_VEN_AMD_AQLPROFILE_BLOCK_NAME_RPB)
block_info = new GpuBlockInfo(RpbCounterBlockInfo);
else
block_info = new GpuBlockInfo(*base_table_ptr);
block_table_[i] = block_info;
// overwrite block info for any update from gfx9 to mi100
switch (block_info->id) {
case SqCounterBlockId:
block_info->event_id_max = 303;
break;
case TcpCounterBlockId:
block_info->event_id_max = 87;
break;
case TccCounterBlockId:
block_info->instance_count = 32;
block_info->event_id_max = 295;
break;
case TcaCounterBlockId:
block_info->instance_count = 32;
block_info->event_id_max = 58;
break;
case GceaCounterBlockId:
block_info->instance_count = 32;
block_info->event_id_max = 83;
break;
case SdmaCounterBlockId:
block_info->instance_count = gfx9_cntx_prim::SDMA_COUNTER_BLOCK_NUM_INSTANCES;
break;
case UmcCounterBlockId:
block_info->counter_count = 6;
break;
}
}
}
Pm4Factory* Pm4Factory::Mi100Create(const AgentInfo* agent_info) {
auto p = new Mi100Factory(agent_info);
if (p == NULL) throw aql_profile_exc_msg("Mi100Factory allocation failed");
return p;
}
} // namespace aql_profile
+93
View File
@@ -0,0 +1,93 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/gfx9_factory.h"
#include "def/gfx90a_def.h"
#include "pm4/gfx9_cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/sqtt_builder.h"
namespace aql_profile {
// Mi200 factory class
class Mi200Factory : public Gfx9Factory {
public:
explicit Mi200Factory(const AgentInfo* agent_info);
virtual int GetAccumLowID() const override { return 1; };
virtual int GetAccumHiID() const override { return 185; };
protected:
static const GpuBlockInfo* block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER];
};
const GpuBlockInfo* Mi200Factory::block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER] = {};
Mi200Factory::Mi200Factory(const AgentInfo* agent_info)
: Gfx9Factory(block_table_, sizeof(block_table_), agent_info) {
for (unsigned i = 0; i < HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER; ++i) {
const GpuBlockInfo* base_table_ptr = Gfx9Factory::block_table_[i];
if (base_table_ptr == NULL) continue;
GpuBlockInfo* block_info = nullptr;
if (i == HSA_VEN_AMD_AQLPROFILE_BLOCK_NAME_RPB)
block_info = new GpuBlockInfo(RpbCounterBlockInfo);
else
block_info = new GpuBlockInfo(*base_table_ptr);
block_table_[i] = block_info;
// overwrite block info for any update from gfx9 to mi100
switch (block_info->id) {
case SqCounterBlockId:
block_info->event_id_max = 303;
break;
case TcpCounterBlockId:
block_info->event_id_max = 87;
break;
case TccCounterBlockId:
block_info->instance_count = 32;
block_info->event_id_max = 295;
break;
case TcaCounterBlockId:
block_info->instance_count = 32;
block_info->event_id_max = 58;
break;
case GceaCounterBlockId:
block_info->instance_count = 32;
block_info->event_id_max = 83;
break;
case SdmaCounterBlockId:
block_info->instance_count = 5;
// Print(block_info);
break;
case UmcCounterBlockId:
block_info->counter_count = 9;
break;
}
}
}
Pm4Factory* Pm4Factory::Mi200Create(const AgentInfo* agent_info) {
auto p = new Mi200Factory(agent_info);
if (p == NULL) throw aql_profile_exc_msg("Mi200Factory allocation failed");
return p;
}
} // namespace aql_profile
+108
View File
@@ -0,0 +1,108 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/gfx9_factory.h"
#include "def/gfx940_def.h"
#include "pm4/gfx9_cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/sqtt_builder.h"
namespace aql_profile {
class Mi300Factory : public Mi100Factory {
public:
explicit Mi300Factory(const AgentInfo* agent_info) : Mi100Factory(agent_info) {
for (unsigned blockname_id = 0; blockname_id < HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER;
++blockname_id) {
const GpuBlockInfo* base_table_ptr = Gfx9Factory::block_table_[blockname_id];
if (base_table_ptr == NULL) continue;
GpuBlockInfo* block_info = nullptr;
if (blockname_id == HSA_VEN_AMD_AQLPROFILE_BLOCK_NAME_RPB)
block_info = new GpuBlockInfo(RpbCounterBlockInfo);
else if (blockname_id == HSA_VEN_AMD_AQLPROFILE_BLOCK_NAME_ATC)
block_info = new GpuBlockInfo(AtcCounterBlockInfo);
else
block_info = new GpuBlockInfo(*base_table_ptr);
block_table_[blockname_id] = block_info;
// overwrite block info for any update from gfx9 to mi300
switch (block_info->id) {
case SqCounterBlockId:
block_info->event_id_max = 373;
break;
case TcpCounterBlockId:
block_info->event_id_max = 84;
break;
case TccCounterBlockId:
block_info->instance_count = 16;
block_info->event_id_max = 199;
break;
case TcaCounterBlockId:
block_info->instance_count = 32;
block_info->event_id_max = 34;
break;
case GceaCounterBlockId:
block_info->instance_count = 32;
block_info->event_id_max = 82;
break;
case SdmaCounterBlockId:
block_info->instance_count = 4 * pm4_builder::MAX_AID;
break;
case UmcCounterBlockId:
block_info->counter_count = 11;
block_info->instance_count = 32 * pm4_builder::MAX_AID;
break;
case RpbCounterBlockId:
block_info->instance_count = 4;
break;
case AtcCounterBlockId:
block_info->instance_count = 4;
break;
}
}
}
virtual int GetAccumLowID() const override { return 1; };
virtual int GetAccumHiID() const override { return 184; };
};
Pm4Factory* Pm4Factory::Mi300Create(const AgentInfo* agent_info) {
auto p = new Mi300Factory(agent_info);
if (p == NULL) throw aql_profile_exc_msg("Mi300Factory allocation failed");
return p;
}
class Mi350Factory : public Mi300Factory {
public:
// MI350 is a copy of Mi300
explicit Mi350Factory(const AgentInfo* agent_info) : Mi300Factory(agent_info) {}
virtual int GetAccumLowID() const override { return 1; };
virtual int GetAccumHiID() const override { return 200; };
};
Pm4Factory* Pm4Factory::Mi350Create(const AgentInfo* agent_info) {
auto p = new Mi350Factory(agent_info);
if (p == NULL) throw aql_profile_exc_msg("Mi350Factory allocation failed");
return p;
}
} // namespace aql_profile
+100
View File
@@ -0,0 +1,100 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/gfx9_factory.h"
#include "def/gfx9_def.h"
#include "pm4/gfx9_cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/sqtt_builder.h"
namespace aql_profile {
// Gfx factory init
void Gfx9Factory::Init(const AgentInfo* agent_info) {
Pm4Factory::cmd_builder_ = new pm4_builder::Gfx9CmdBuilder(nullptr);
if (Pm4Factory::cmd_builder_ == NULL) throw aql_profile_exc_msg("CmdBuilder allocation failed");
// Mark and set the mode
if (Pm4Factory::IsConcurrent()) {
Pm4Factory::pmc_builder_ =
new pm4_builder::GpuPmcBuilder<pm4_builder::Gfx9CmdBuilder, gfx9_cntx_prim, true>(
agent_info);
} else {
Pm4Factory::pmc_builder_ =
new pm4_builder::GpuPmcBuilder<pm4_builder::Gfx9CmdBuilder, gfx9_cntx_prim, false>(
agent_info);
}
if (Pm4Factory::pmc_builder_ == NULL) throw aql_profile_exc_msg("PmcBuilder allocation failed");
Pm4Factory::spm_builder_ =
new pm4_builder::GpuSpmBuilder<pm4_builder::Gfx9CmdBuilder, gfx9_cntx_prim>(agent_info);
if (Pm4Factory::spm_builder_ == NULL) throw aql_profile_exc_msg("SpmBuilder allocation failed");
Pm4Factory::sqtt_builder_ =
new pm4_builder::GpuSqttBuilder<pm4_builder::Gfx9CmdBuilder, gfx9_cntx_prim>(agent_info);
if (Pm4Factory::sqtt_builder_ == NULL) throw aql_profile_exc_msg("SqttBuilder allocation failed");
agent_info_ = agent_info;
}
void Gfx9Factory::Print(const GpuBlockInfo* block_info) {
std::cout << "Block name: " << block_info->name << std::endl;
std::cout << "\tInstances: " << block_info->instance_count << std::endl;
std::cout << "\tMax Events: " << block_info->event_id_max << std::endl;
std::cout << "\tCounters: " << block_info->counter_count << std::endl;
auto counters = block_info->instance_count * block_info->counter_count;
for (int i = 0; i < counters; ++i) {
auto reg_info = block_info->counter_reg_info[i];
std::cout << "\t " << i << ": select_addr = 0x" << std::hex << reg_info.select_addr.offset
<< "(" << reg_info.select_addr.offset * 4 << ")"
<< ", control_addr = 0x" << reg_info.control_addr.offset << "("
<< reg_info.control_addr.offset * 4 << ")"
<< ", counter_addr_lo = 0x" << reg_info.register_addr_lo.offset << "("
<< reg_info.register_addr_lo.offset * 4 << ")"
<< ", counter_addr_hi = 0x" << reg_info.register_addr_hi.offset << "("
<< reg_info.register_addr_hi.offset * 4 << ")" << std::dec << std::endl;
}
}
// GFX9 block table
const GpuBlockInfo* Gfx9Factory::block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER] = {
&CpcCounterBlockInfo, &CpfCounterBlockInfo, &GdsCounterBlockInfo, &GrbmCounterBlockInfo,
&GrbmSeCounterBlockInfo, &SpiCounterBlockInfo, &SqCounterBlockInfo, &SqCsCounterBlockInfo,
NULL /*GFX? SRBM*/, &SxCounterBlockInfo, &TaCounterBlockInfo, &TcaCounterBlockInfo,
&TccCounterBlockInfo, &TcpCounterBlockInfo, &TdCounterBlockInfo,
// MC blocks
NULL /*MC_ARB*/, NULL /*MC_HUB*/, NULL /*MC_MCBVM*/, NULL /*MC_SEQ*/, &McVmL2CounterBlockInfo,
NULL /*MC_XBAR*/, &AtcCounterBlockInfo, &AtcL2CounterBlockInfo, &GceaCounterBlockInfo,
&RpbCounterBlockInfo,
// System blocks
NULL /*&SdmaCounterBlockInfo*/, NULL /*GL1A*/, NULL /*GL1C*/, NULL /*GL2A*/, NULL /*GL2C*/,
NULL /*GCR*/, NULL /*GUS*/, NULL /*&UmcCounterBlockInfo*/
};
// Pm4Factory create mathods
Pm4Factory* Pm4Factory::Gfx9Create(const AgentInfo* agent_info) {
auto p = new Gfx9Factory(agent_info);
if (p == NULL) throw aql_profile_exc_msg("Gfx9Factory allocation failed");
return p;
}
} // namespace aql_profile
+61
View File
@@ -0,0 +1,61 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#ifndef _GFX9_FACTORY_H_
#define _GFX9_FACTORY_H_
#include "core/pm4_factory.h"
namespace aql_profile {
// Gfx9 factory class
class Gfx9Factory : public Pm4Factory {
public:
explicit Gfx9Factory(const AgentInfo* agent_info)
: Pm4Factory(BlockInfoMap(block_table_, sizeof(block_table_))) {
Init(agent_info);
}
Gfx9Factory(const GpuBlockInfo** table, const uint32_t& size, const AgentInfo* agent_info)
: Pm4Factory(BlockInfoMap(table, size)) {
Init(agent_info);
}
bool IsGFX9() const override { return true; }
protected:
void Init(const AgentInfo* agent_info);
static const GpuBlockInfo* block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER];
static void Print(const GpuBlockInfo* block_info);
};
// Mi100 factory class
class Mi100Factory : public Gfx9Factory {
public:
explicit Mi100Factory(const AgentInfo* agent_info);
protected:
static const GpuBlockInfo* block_table_[HSA_VEN_AMD_AQLPROFILE_BLOCKS_NUMBER];
};
} // namespace aql_profile
#endif // _GFX9_FACTORY_H_
+7
View File
@@ -0,0 +1,7 @@
set(AQLPROFILE_HEADER_FILES
aql_profile_v2.h
)
install(
FILES ${AQLPROFILE_HEADER_FILES}
DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}/aqlprofile-sdk)
+434
View File
@@ -0,0 +1,434 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#pragma once
#include <hsa/hsa.h>
#include <hsa/hsa_ven_amd_aqlprofile.h>
#ifdef __cplusplus
extern "C" {
#endif
typedef struct {
uint64_t handle;
} aqlprofile_handle_t;
typedef enum {
AQLPROFILE_MEMORY_HINT_NONE = 0,
AQLPROFILE_MEMORY_HINT_HOST = 1,
AQLPROFILE_MEMORY_HINT_DEVICE_UNCACHED = 2,
AQLPROFILE_MEMORY_HINT_DEVICE_COHERENT = 3,
AQLPROFILE_MEMORY_HINT_DEVICE_NONCOHERENT = 4,
AQLPROFILE_MEMORY_HINT_LAST
} aqlprofile_memory_hint_t;
typedef enum {
AQLPROFILE_AGENT_VERSION_NONE = 0,
AQLPROFILE_AGENT_VERSION_V0 = 1,
AQLPROFILE_AGENT_VERSION_V1 = 2,
AQLPROFILE_AGENT_VERSION_LAST
} aqlprofile_agent_version_t;
/**
* @brief Flags to describe which agents can access given buffer.
*/
typedef union {
uint32_t raw;
struct {
uint32_t device_access : 1;
uint32_t host_access : 1;
uint32_t memory_hint : 6; // One of aqlprofile_memory_hint_t
uint32_t _reserved : 24;
};
} aqlprofile_buffer_desc_flags_t;
/**
* @brief Callback to request a memory buffer, which will be tied to a profile.
* The user is responsible for clearing up memory after the profile is no longer needed.
* @param[out] ptr The pointer containing memory.
* @param[in] size Minimum requested buffer size.
* @param[in] flags Access flags, requesting which agents need to read/write to the buffer.
* @param[in] userdata Data to be passed back to user.
* @retval HSA_STATUS_SUCCESS if successful
* @retval HSA_STATUS_ERROR if memory could not be allocated
*/
typedef hsa_status_t (*aqlprofile_memory_alloc_callback_t)(void** ptr, uint64_t size,
aqlprofile_buffer_desc_flags_t flags,
void* userdata);
/**
* @brief Callback to dealloc memory requested via aqlprofile_memory_alloc_callback_t
* @param[in] ptr The pointer containing memory.
* @param[in] userdata Data to be passed back to user.
* @retval HSA_STATUS_SUCCESS if successful
* @retval HSA_STATUS_ERROR if memory could not be allocated
*/
typedef void (*aqlprofile_memory_dealloc_callback_t)(void* ptr, void* userdata);
typedef enum {
AQLPROFILE_ACCUMULATION_NONE = 0, /** Do not accumulate event */
AQLPROFILE_ACCUMULATION_LO_RES, /**< The event should be integrated over quad-cycles */
AQLPROFILE_ACCUMULATION_HI_RES, /**< The event should be integrated every cycle */
AQLPROFILE_ACCUMULATION_LAST,
} aqlprofile_accumulation_type_t;
/**
* @brief Special flags indicating additional properties to a counter. E.g. Accumulation metrics
*/
typedef union {
uint32_t raw;
struct {
uint32_t accum : 3; /**< One of aqlprofile_accumulation_type_t */
uint32_t _reserved : 29;
} sq_flags;
} aqlprofile_pmc_event_flags_t;
/**
* @brief Struct containing all necessary information of an event (counter).
*/
typedef struct {
uint32_t block_index; /**< Block channel. */
uint32_t event_id; /**< Event ID as fined by XML */
aqlprofile_pmc_event_flags_t flags; /**< Special event flags e.g. accumulation */
hsa_ven_amd_aqlprofile_block_name_t block_name; /**< Block name as defined by block indexes */
} aqlprofile_pmc_event_t;
/**
* @brief Struct containing information about the agent. User code sets these values
* to the describe the agent to profile. Information can be obtained either from HSA
* (if loaded) or the KFD topology.
*/
typedef struct {
const char* agent_gfxip; /**< Agent GFXIP (HSA_AGENT_INFO_NAME or KFD.product_name) */
uint32_t xcc_num; /**< XCC's on the agent (HSA_AMD_AGENT_INFO_NUM_XCC or KFD.num_xcc) */
uint32_t se_num; /**< SE's on the agent (HSA_AMD_AGENT_INFO_NUM_SHADER_ENGINES or
KFD.num_shader_banks) */
uint32_t cu_num; /**< CU's on the agent (HSA_AMD_AGENT_INFO_COMPUTE_UNIT_COUNT or KFD.cu_count) */
uint32_t shader_arrays_per_se; /**< Shader arrays per SE of agent
(HSA_AMD_AGENT_INFO_NUM_SHADER_ARRAYS_PER_SE or
KFD.simd_arrays_per_engine)*/
} aqlprofile_agent_info_t;
/**
* @brief Struct containing information about the agent. User code sets these values
* to the describe the agent to profile. Information can be obtained either from HSA
* (if loaded) or the KFD topology.
*/
typedef struct {
const char* agent_gfxip; /**< Agent GFXIP (HSA_AGENT_INFO_NAME or KFD.product_name) */
uint32_t xcc_num; /**< XCC's on the agent (HSA_AMD_AGENT_INFO_NUM_XCC or KFD.num_xcc) */
uint32_t se_num; /**< SE's on the agent (HSA_AMD_AGENT_INFO_NUM_SHADER_ENGINES or
KFD.num_shader_banks) */
uint32_t cu_num; /**< CU's on the agent (HSA_AMD_AGENT_INFO_COMPUTE_UNIT_COUNT or KFD.cu_count) */
uint32_t shader_arrays_per_se; /**< Shader arrays per SE of agent
(HSA_AMD_AGENT_INFO_NUM_SHADER_ARRAYS_PER_SE or
KFD.simd_arrays_per_engine)*/
uint32_t domain; /**< PCI domain of the GPU agent (HSA_AMD_AGENT_INFO_DOMAIN or KFD.domain) */
uint32_t location_id; /**< BDF (Bus/Device/function number) of the GPU agent
(HSA_AMD_AGENT_INFO_BDFID or KFD.location_id)*/
} aqlprofile_agent_info_v1_t;
/**
* @brief Struct containing a handle to a registered agent
*
*/
typedef struct {
uint64_t handle;
} aqlprofile_agent_handle_t;
/**
* @brief Registers an agent to be used with AQL profile.
* @param[out] agent_id Handle to newly registered agent
* @param[in] agent_info Info to register a new agent with AQL Profiler
* @retval HSA_STATUS_SUCCESS registration ok
* @retval HSA_STATUS_ERROR registration failed
*/
hsa_status_t aqlprofile_register_agent(aqlprofile_agent_handle_t* agent_id,
const aqlprofile_agent_info_t* agent_info);
/**
* @brief Registers an agent to be used with AQL profile.
* @param[out] agent_id Handle to newly registered agent
* @param[in] agent_info Info to register a new agent with AQL Profiler
* @param[in] version Version of the agent info structure
* @retval HSA_STATUS_SUCCESS registration ok
* @retval HSA_STATUS_ERROR registration failed
*/
hsa_status_t aqlprofile_register_agent_info(aqlprofile_agent_handle_t* agent_id,
const void* agent_info,
aqlprofile_agent_version_t version);
/**
* @brief AQLprofile struct containing information for perfmon events
*/
typedef struct {
aqlprofile_agent_handle_t agent;
const aqlprofile_pmc_event_t* events;
uint32_t event_count;
} aqlprofile_pmc_profile_t;
// Profile attributes
typedef enum {
AQLPROFILE_INFO_COMMAND_BUFFER_SIZE = 0, // get_info returns uint32_t value
AQLPROFILE_INFO_PMC_DATA_SIZE = 1, // get_info returns uint32_t value
AQLPROFILE_INFO_PMC_DATA = 2, // get_info returns PMC uint64_t value
// in info_data object
AQLPROFILE_INFO_BLOCK_COUNTERS = 4, // get_info returns number of block counter
AQLPROFILE_INFO_BLOCK_ID = 5, // get_info returns block id, instances
// by name string using _id_query_t
AQLPROFILE_INFO_ENABLE_CMD = 6, // get_info returns size/pointer for
// counters enable command buffer
AQLPROFILE_INFO_DISABLE_CMD = 7, // get_info returns size/pointer for
// counters disable command buffer
} aqlprofile_pmc_info_type_t;
hsa_status_t aqlprofile_get_pmc_info(const aqlprofile_pmc_profile_t* profile,
aqlprofile_pmc_info_type_t attribute, void* value);
// Profile parameter object
typedef struct {
hsa_ven_amd_aqlprofile_parameter_name_t parameter_name;
union {
uint32_t value;
struct {
uint32_t counter_id : 28;
uint32_t simd_mask : 4;
};
};
} aqlprofile_att_parameter_t;
/**
* @brief AQLprofile struct containing information for Advanced Thread Trace
*/
typedef struct {
hsa_agent_t agent;
const aqlprofile_att_parameter_t* parameters;
uint32_t parameter_count;
} aqlprofile_att_profile_t;
/**
* @brief Data callback for perfmon events. Each event will call this once per coordinate
* @param[in] event The event information passed in from aqlprofile_pmc_profile_t
* @param[in] counter_id Internal ID of the counter
* @param[in] counter_value The event value, as incremented from start() to stop()
* @param[in] userdata Data returned to user
* @retval HSA_STATUS_SUCCESS to continue iteration
* @retval HSA_STATUS_ERROR to stop callback iteration
*/
typedef hsa_status_t (*aqlprofile_pmc_data_callback_t)(aqlprofile_pmc_event_t event,
uint64_t counter_id, uint64_t counter_value,
void* userdata);
/**
* @brief Data callback for thread trace. This will be called at least once per shader engine
* @param[in] shader Shader Engine ID
* @param[in] buffer Pointer containing the data
* @param[in] size Amount of bytes used by thread trace
* @param[in] callback_data Data returned to user
* @retval HSA_STATUS_SUCCESS to continue iteration
* @retval HSA_STATUS_ERROR to stop callback iteration
*/
typedef hsa_status_t (*aqlprofile_att_data_callback_t)(uint32_t shader, void* buffer, uint64_t size,
void* callback_data);
/**
* @brief Memory copy fn for aqlprofile to copy data.
* @param[in] dst Destination pointer to copy data to.
* @param[in] src Source pointer where data is to be copied from.
* @param[in] size Amount of bytes to be copied.
* @param[in] userdata Data returned to user
* @retval HSA_STATUS_SUCCESS on success
* @retval HSA_STATUS_ERROR on failure
*/
typedef hsa_status_t (*aqlprofile_memory_copy_t)(void* dst, const void* src, size_t size,
void* userdata);
/**
* @brief Validates the event for the agent.
* @param[in] agent The agent to validate the event for.
* @param[in] event The event to validate.
* @param[out] result True if the event is valid for the agent, false otherwise.
* @retval HSA_STATUS_SUCCESS if the event was validated.
* @retval HSA_STATUS_ERROR if the event was not validated.
*/
hsa_status_t aqlprofile_validate_pmc_event(aqlprofile_agent_handle_t agent,
const aqlprofile_pmc_event_t* event, bool* result);
/**
* @brief Iterate_data() will parse the event data and call @callback with the resulting event data
* @param[in] handle The handle returned from aqlprofile_pmc_create_packets()
* @param[in] callback CB where the resulting event values are going to be returned
* @param[in] userdata Data sent back to user
* @retval HSA_STATUS_SUCCESS all operations exited succesfully
* @retval HSA_STATUS_ERROR if some callback returns an error
* @retval HSA_STATUS_ERROR_INVALID_ARGUMENT if invalid handle is given
*/
hsa_status_t aqlprofile_pmc_iterate_data(aqlprofile_handle_t handle,
aqlprofile_pmc_data_callback_t callback, void* userdata);
/**
* @brief Struct to be returned by aqlprofile_pmc_create_packets
*/
typedef struct {
hsa_ext_amd_aql_pm4_packet_t start_packet; /**< Reset counters and start incrementing */
hsa_ext_amd_aql_pm4_packet_t stop_packet; /**< Pause counters from incrementing */
hsa_ext_amd_aql_pm4_packet_t read_packet; /**< Retrieve results from device */
} aqlprofile_pmc_aql_packets_t;
/**
* @brief Function to create AQL packets to be inserted into the queue.
* @param[out] handle To be passed to iterate_data()
* @param[out] packets Pointer to where the start, stop and read packets will be written to
* @param[in] profile Agent and events information
* @param[in] alloc_cb Memory allocation, which may request cpu or gpu memory for internal use
* @param[in] dealloc_cb Function to free memory allocated by alloc_cb
* @param[in] userdata Data passed back to user via memory alloc callback
*/
hsa_status_t aqlprofile_pmc_create_packets(aqlprofile_handle_t* handle,
aqlprofile_pmc_aql_packets_t* packets,
aqlprofile_pmc_profile_t profile,
aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb,
aqlprofile_memory_copy_t memcpy_cb, void* userdata);
/**
* @brief Function to delete AQL packets after creation by aqlprofile_pmc_create_packets
* @param[in] handle Returned by aqlprofile_pmc_create_packets()
*/
void aqlprofile_pmc_delete_packets(aqlprofile_handle_t handle);
/**
* @brief Iterates over thread trace data and the data to user
* @param[in] handle The handle returned from aqlprofile_att_create_packets()
* @param[in] callback CB where the resulting data is going to be returned
* @param[in] userdata Data sent back to user
* @retval HSA_STATUS_SUCCESS all operations exited succesfully
* @retval HSA_STATUS_ERROR if some callback returns an error
* @retval HSA_STATUS_ERROR_INVALID_ARGUMENT if invalid handle is given
*/
hsa_status_t aqlprofile_att_iterate_data(aqlprofile_handle_t handle,
aqlprofile_att_data_callback_t callback, void* userdata);
/**
* @brief Struct containing AQLpackets to start and stop thread trace
*/
typedef struct {
hsa_ext_amd_aql_pm4_packet_t start_packet; /**< Packet to start thread trace */
hsa_ext_amd_aql_pm4_packet_t stop_packet; /**< Packet to stop thread trace and flush data */
} aqlprofile_att_control_aql_packets_t;
/**
* @brief Fn to create start and stop thread trace packets
* @param[out] handle To be passed to iterate_data()
* @param[out] packets Packets returned by this function to start and stop thread trace
* @param[in] profile Agent information and extra parameters for thread trace
* @param[in] callback Memory allocation fn which may request cpu or gpu memory
* @retval HSA_STATUS_SUCCESS if all packets created succesfully
* @retval HSA_STATUS_ERROR otherwise
*/
hsa_status_t aqlprofile_att_create_packets(aqlprofile_handle_t* handle,
aqlprofile_att_control_aql_packets_t* packets,
aqlprofile_att_profile_t profile,
aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb,
aqlprofile_memory_copy_t memcpy_cb, void* userdata);
void aqlprofile_att_delete_packets(aqlprofile_handle_t handle);
/**
* @brief Callback for iteration of all possible event coordinate IDs and coordinate names.
* @param [in] id Integer identifying the dimension.
* @param [in] name Name of the dimension
* @param [in] data User data supplied to @ref aqlprofile_iterate_event_ids
* @retval HSA_STATUS_SUCCESS Continues iteration
* @retval OTHERS Any other HSA return values stops iteration, passing back this value through
* @ref aqlprofile_iterate_event_ids
*/
typedef hsa_status_t (*aqlprofile_eventname_callback_t)(int id, const char* name, void* data);
/**
* @brief Iterate over all possible event coordinate IDs and their names.
* @param [in] callback Callback to use for iteration of dimensions
* @param [in] user_data Data to supply to callback @ref aqlprofile_eventname_callback_t
* @retval HSA_STATUS_SUCCESS if successful
* @retval HSA_STATUS_ERROR if error on interation
* @retval OTHERS If @ref aqlprofile_eventname_callback_t returns non-HSA_STATUS_SUCCESS,
* that value is returned.
*/
hsa_status_t aqlprofile_iterate_event_ids(aqlprofile_eventname_callback_t callback,
void* user_data);
/**
* @brief Iterate over all event coordinates for a given agent_t and event_t.
* @param position A counting sequence indicating callback number.
* @param id Coordinate ID as in _iterate_event_ids.
* @param extent Coordinate extent indicating maximum allowed instances.
* @param coordinate The coordinate, in the range [0,extent-1].
* @param name Coordinate name as in _iterate_event_ids.
* @param userdata Userdata returned from _iterate_event_coord function.
*/
typedef hsa_status_t (*aqlprofile_coordinate_callback_t)(int position, int id, int extent,
int coordinate, const char* name,
void* userdata);
/**
* @brief Iterate over all event coordinates for a given agent_t and event_t.
* @param[in] agent HSA agent.
* @param[in] event The event ID and block ID to iterate for.
* @param[in] sample_id aqlprofile_info_data_t.sample_id returned from _aqlprofile_iterate_data.
* @param[in] callback Callback function to return the coordinates.
* @param[in] userdata Arbitrary data pointer to be sent back to the user via callback.
*/
hsa_status_t aqlprofile_iterate_event_coord(aqlprofile_agent_handle_t agent,
aqlprofile_pmc_event_t event, uint64_t sample_id,
aqlprofile_coordinate_callback_t callback,
void* userdata);
typedef struct {
uint64_t id;
uint64_t addr;
uint64_t size;
hsa_agent_t agent;
uint32_t isUnload : 1;
uint32_t fromStart : 1;
} aqlprofile_att_codeobj_data_t;
/**
* @brief Creates an AQL packet for marking code objects
* @param[out] packet Returned packet
* @param[out] handle The handle created for these packets
* @param[in] data Code object information
* @param[in] alloc_cb Callback to return both CPU and GPU accessible memory on demand
* @param[in] dealloc_cb Callback to free data allocated by alloc_cb()
* @param[in] userdata Userdata to be passed back to memory callbacks
*/
hsa_status_t aqlprofile_att_codeobj_marker(hsa_ext_amd_aql_pm4_packet_t* packet,
aqlprofile_handle_t* handle,
aqlprofile_att_codeobj_data_t data,
aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb,
void* userdata);
#ifdef __cplusplus
}
#endif
+51
View File
@@ -0,0 +1,51 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#ifndef SRC_CORE_IP_DISCOVERY_H_
#define SRC_CORE_IP_DISCOVERY_H_
#include <array>
#include <vector>
#include <string>
#include <unordered_map>
#include <optional>
#include "util/reg_offsets.h"
using base_addr_segments_t = std::array<uint32_t, HWIP_MAX_SEGMENT>;
// Represents a single entry in the discovery table, containing information about a specific IP
// block.
struct discovery_table_entry_t {
int die{0}; // Die index
base_addr_segments_t segments{}; // Base address segments
int major{0}; // Major version of the IP
int minor{0}; // Minor version of the IP
int revision{0}; // Revision number of the IP
int instance{0}; // Instance ID of the IP
std::string ipname{}; // Name of the IP block
};
using discovery_table_t = std::vector<discovery_table_entry_t>;
discovery_table_t parse_ip_discovery(uint32_t domain, uint32_t bdf);
#endif
+98
View File
@@ -0,0 +1,98 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include <iostream>
#include <mutex>
#include <stdexcept>
#include <shared_mutex>
#include <array>
#include <unordered_map>
#include "util/hsa_rsrc_factory.h"
#include "util/reg_offsets.h"
#include "ip_offset_table_init.h"
// Pair of pcie domain, bdf
using domain_bdf_t = std::pair<uint32_t, uint32_t>;
// Hash function for domain_bdf_t
template <>
struct std::hash<domain_bdf_t> {
std::size_t operator()(const domain_bdf_t& key) const {
return std::hash<uint32_t>()(key.first) ^ (std::hash<uint32_t>()(key.second) << 1);
}
};
// Map from (Domain, BDF) to reg_base_offset_table*
using reg_base_offset_table_cache = std::unordered_map<domain_bdf_t, const reg_base_offset_table*>;
class locked_ip_offset_table_cache {
public:
const reg_base_offset_table* get(const AgentInfo* agent_info) {
{
std::shared_lock lock{mutex};
auto it = cache.find(std::make_pair(agent_info->domain, agent_info->bdf_id));
if (it != cache.end()) return it->second;
}
{
std::string_view gfxip(agent_info->gfxip);
std::unique_lock lock{mutex};
const reg_base_offset_table* table = nullptr;
if (auto gfxip_prefix = gfxip.substr(0, 4); gfxip_prefix == "gfx9")
table = vega20_reg_base_init();
else {
if (auto gfxip_prefix = gfxip.substr(0, 5);
gfxip_prefix == "gfx10" || gfxip_prefix == "gfx11" || gfxip_prefix == "gfx12") {
table = navi_ip_offset_table_discovery_sysfs(agent_info->domain, agent_info->bdf_id);
if (!table) table = sienna_cichlid_reg_base_init();
}
}
if (table) cache.emplace(std::make_pair(agent_info->domain, agent_info->bdf_id), table);
return table;
}
}
static locked_ip_offset_table_cache& get_instance() {
// Note: never cleanup, keep in memory to prevent issue with global destructor
static auto* cache = new locked_ip_offset_table_cache{};
return *cache;
}
private:
std::shared_mutex mutex;
reg_base_offset_table_cache cache;
};
// acquire the IP offset table for the device using the domain and bdf_id
const reg_base_offset_table* acquire_ip_offset_table(const AgentInfo* agent_info) {
auto ip_offset_table = locked_ip_offset_table_cache::get_instance().get(agent_info);
if (ip_offset_table == nullptr) {
throw std::runtime_error(
"Failed to acquire the IP offset table for the device. Possible reasons include:\n"
" 1. Incorrect or incomplete ROCm setup. Please verify your installation.\n"
" 2. The device is not supported.\n"
" 3. An internal error or bug.\n");
}
return ip_offset_table;
}
+33
View File
@@ -0,0 +1,33 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#ifndef SRC_CORE_IP_OFFSET_TABLE_INIT_H_
#define SRC_CORE_IP_OFFSET_TABLE_INIT_H_
// static IP offset table init functions
const reg_base_offset_table* vega20_reg_base_init();
const reg_base_offset_table* sienna_cichlid_reg_base_init();
// dynamic IP offset table functions
const reg_base_offset_table* navi_ip_offset_table_discovery_sysfs(uint32_t domain, uint32_t bdf);
#endif // SRC_CORE_IP_OFFSET_TABLE_INIT_H_
+177
View File
@@ -0,0 +1,177 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#ifndef SRC_CORE_LOGGER_H_
#define SRC_CORE_LOGGER_H_
#include <stdarg.h>
#include <stdio.h>
#include <stdlib.h>
#include <sys/file.h>
#include <sys/syscall.h>
#include <sys/types.h>
#include <time.h>
#include <unistd.h>
#include <exception>
#include <fstream>
#include <iostream>
#include <map>
#include <mutex>
#include <sstream>
#include <string>
namespace aql_profile {
class Logger {
public:
typedef std::recursive_mutex mutex_t;
template <typename T>
Logger& operator<<(const T& m) {
std::ostringstream oss;
oss << m;
if (!streaming_)
Log(oss.str());
else
Put(oss.str());
streaming_ = true;
return *this;
}
typedef void (*manip_t)();
Logger& operator<<(manip_t f) {
f();
return *this;
}
static void begm() { Instance().messaging_ = true; }
static void endl() { Instance().ResetStreaming(); }
static const std::string& LastMessage() {
Logger& logger = Instance();
std::lock_guard<mutex_t> lck(mutex_);
return logger.message_[GetTid()];
}
static Logger& Instance() {
std::lock_guard<mutex_t> lck(mutex_);
if (instance_ == NULL) instance_ = new Logger();
return *instance_;
}
static void Destroy() {
std::lock_guard<mutex_t> lck(mutex_);
if (instance_ != NULL) delete instance_;
instance_ = NULL;
}
private:
static uint32_t GetPid() { return syscall(__NR_getpid); }
static uint32_t GetTid() { return syscall(__NR_gettid); }
Logger() : file_(NULL), dirty_(false), streaming_(false), messaging_(false) {
const char* path = getenv("HSA_VEN_AMD_AQLPROFILE_LOG");
if (path != NULL) {
file_ = fopen("/tmp/aql_profile_log.txt", "a");
}
ResetStreaming();
}
~Logger() {
if (file_ != NULL) {
if (dirty_) Put("\n");
fclose(file_);
}
}
void ResetStreaming() {
std::lock_guard<mutex_t> lck(mutex_);
if (messaging_) {
message_[GetTid()] = "";
}
messaging_ = false;
streaming_ = false;
}
void Put(const std::string& m) {
std::lock_guard<mutex_t> lck(mutex_);
if (messaging_) {
message_[GetTid()] += m;
}
if (file_ != NULL) {
dirty_ = true;
flock(fileno(file_), LOCK_EX);
fprintf(file_, "%s", m.c_str());
fflush(file_);
flock(fileno(file_), LOCK_UN);
}
}
void Log(const std::string& m) {
const time_t rawtime = time(NULL);
tm tm_info;
localtime_r(&rawtime, &tm_info);
char tm_str[26];
strftime(tm_str, 26, "%Y-%m-%d %H:%M:%S", &tm_info);
std::ostringstream oss;
oss << "\n<" << tm_str << std::dec << " pid" << GetPid() << " tid" << GetTid() << "> " << m;
Put(oss.str());
}
FILE* file_;
bool dirty_;
bool streaming_;
bool messaging_;
static mutex_t mutex_;
static Logger* instance_;
std::map<uint32_t, std::string> message_;
};
} // namespace aql_profile
#define ERR_LOGGING \
(aql_profile::Logger::Instance() \
<< aql_profile::Logger::endl \
<< "Error: " << __FUNCTION__ << "(): " << aql_profile::Logger::begm)
#define ERR2_LOGGING \
(aql_profile::Logger::Instance() << aql_profile::Logger::endl \
<< "Error: " << __FUNCTION__ << "(): ")
#define INFO_LOGGING \
(aql_profile::Logger::Instance() \
<< aql_profile::Logger::endl \
<< "Info: " << __FUNCTION__ << "(): " << aql_profile::Logger::begm)
#define WARN_LOGGING \
(aql_profile::Logger::Instance() \
<< aql_profile::Logger::endl \
<< "Warning: " << __FUNCTION__ << "(): " << aql_profile::Logger::begm)
#ifdef DEBUG
#define DBG_LOGGING \
(aql_profile::Logger::Instance() << aql_profile::Logger::endl \
<< "Debug: in " << __FUNCTION__ << " at " << __FILE__ \
<< " line " << __LINE__ << aql_profile::Logger::begm)
#endif
#endif // SRC_CORE_LOGGER_H_
+61
View File
@@ -0,0 +1,61 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "memorymanager.hpp"
#include <algorithm>
std::atomic<size_t> MemoryManager::HANDLE_COUNTER{1};
std::unordered_map<size_t, std::shared_ptr<MemoryManager>> MemoryManager::managers;
std::mutex MemoryManager::managers_map_mutex;
void CounterMemoryManager::CopyEvents(const aqlprofile_pmc_event_t* _events, size_t count) {
events.reserve(count + 4);
int num_flag_metrics = 0;
for (size_t i = 0; i < count; i++) {
events.push_back(EventRequest{_events[i], false});
num_flag_metrics += _events[i].flags.raw != 0;
}
if (!num_flag_metrics) return;
std::sort(events.begin(), events.end());
std::vector<EventRequest> acc_requests;
for (auto it = events.begin(); it != events.end(); it++) {
if (!it->flags.raw) continue;
if (it != events.begin()) {
auto prev = std::prev(it);
if (it->IsSameNoFlags(*prev) && (!prev->flags.raw || prev->bInternal)) continue;
}
EventRequest req = *it;
req.bInternal = true;
req.flags.raw = 0;
acc_requests.push_back(req);
}
if (!acc_requests.size()) return;
events.insert(events.end(), acc_requests.begin(), acc_requests.end());
std::sort(events.begin(), events.end());
}
+258
View File
@@ -0,0 +1,258 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#pragma once
#include <vector>
#include <atomic>
#include <mutex>
#include <unordered_map>
#include <memory>
#include "include/aql_profile_v2.h"
#include <stdexcept>
#include "pm4/trace_config.h"
struct EventRequest : public aqlprofile_pmc_event_t {
bool bInternal;
auto GetOrder() const -> auto{
uint64_t idx = bInternal ? 0 : 1;
idx |= uint64_t(flags.raw) << 1;
idx |= uint64_t(event_id) << 33;
uint64_t blk = block_index;
blk |= uint64_t(block_name) << 32;
return std::pair<uint64_t, uint64_t>{blk, idx};
}
bool operator<(const EventRequest& other) const {
auto idx1 = this->GetOrder();
auto idx2 = other.GetOrder();
if (idx1.first == idx2.first)
return idx1.second < idx2.second;
else
return idx1.first < idx2.first;
}
bool operator==(const EventRequest& other) const {
auto idx1 = this->GetOrder();
auto idx2 = other.GetOrder();
return idx1.second == idx2.second && idx1.first == idx2.first;
}
bool IsSameNoFlags(const EventRequest& other) const {
auto idx1 = this->GetOrder();
auto idx2 = other.GetOrder();
return idx1.first == idx2.first && event_id == other.event_id;
}
};
class MemoryManager {
public:
MemoryManager(hsa_agent_t agent, aqlprofile_memory_alloc_callback_t alloc,
aqlprofile_memory_dealloc_callback_t dealloc, void* data)
: agent(agent),
alloc_cb(alloc),
dealloc_cb(dealloc),
userdata(data),
handle(HANDLE_COUNTER.fetch_add(1)) {}
MemoryManager(aqlprofile_agent_handle_t agent, aqlprofile_memory_alloc_callback_t alloc,
aqlprofile_memory_dealloc_callback_t dealloc, void* data)
: agent_handle(agent),
alloc_cb(alloc),
dealloc_cb(dealloc),
userdata(data),
handle(HANDLE_COUNTER.fetch_add(1)) {}
virtual ~MemoryManager() {}
void CheckStatus(hsa_status_t status) const {
if (status != HSA_STATUS_SUCCESS) throw status;
}
void* GetCmdBuf() const { return cmdbuf.get(); }
void* GetOutputBuf() const { return outputbuf.get(); }
size_t GetOutputBufSize() const { return outputbuf_size; }
size_t GetHandler() const { return handle; }
hsa_agent_t GetAgent() const { return agent; }
aqlprofile_agent_handle_t AgentHandle() const { return agent_handle; }
void CreateCmdBuf(size_t size) {
aqlprofile_buffer_desc_flags_t flags{};
flags.host_access = true;
flags.device_access = true;
flags.memory_hint = AQLPROFILE_MEMORY_HINT_DEVICE_NONCOHERENT;
cmdbuf = AllocMemory(size, flags);
}
virtual void CreateOutputBuf(size_t size) = 0;
static void RegisterManager(const std::shared_ptr<MemoryManager>& shared) {
std::lock_guard<std::mutex> lk(managers_map_mutex);
managers[shared->handle] = shared;
}
static void DeleteManager(size_t handle) {
std::lock_guard<std::mutex> lk(managers_map_mutex);
managers.erase(handle);
}
static std::shared_ptr<MemoryManager> GetManager(size_t handle) {
std::lock_guard<std::mutex> lk(managers_map_mutex);
try {
return managers.at(handle);
} catch (std::exception& e) {
return nullptr;
}
}
protected:
struct MemoryDeleter {
aqlprofile_memory_dealloc_callback_t free_fn;
void* userdata;
void operator()(void* ptr) const {
if (ptr && free_fn) free_fn(ptr, userdata);
};
};
std::unique_ptr<void, MemoryDeleter> AllocMemory(size_t size,
aqlprofile_buffer_desc_flags_t flags) const {
void* ptr;
CheckStatus(alloc_cb(&ptr, size, flags, userdata));
return std::unique_ptr<void, MemoryDeleter>{ptr, MemoryDeleter{dealloc_cb, userdata}};
}
aqlprofile_agent_handle_t agent_handle = {.handle = 0};
hsa_agent_t agent = {.handle = 0};
std::unique_ptr<void, MemoryDeleter> cmdbuf = nullptr;
std::unique_ptr<void, MemoryDeleter> outputbuf = nullptr;
size_t outputbuf_size = 0;
void* const userdata;
aqlprofile_memory_alloc_callback_t const alloc_cb;
aqlprofile_memory_dealloc_callback_t const dealloc_cb;
size_t handle;
static std::atomic<size_t> HANDLE_COUNTER;
static std::unordered_map<size_t, std::shared_ptr<MemoryManager>> managers;
static std::mutex managers_map_mutex;
};
class CounterMemoryManager : public MemoryManager {
public:
CounterMemoryManager(hsa_agent_t agent, aqlprofile_memory_alloc_callback_t alloc,
aqlprofile_memory_dealloc_callback_t dealloc, void* data)
: MemoryManager(agent, alloc, dealloc, data) {}
CounterMemoryManager(aqlprofile_agent_handle_t agent, aqlprofile_memory_alloc_callback_t alloc,
aqlprofile_memory_dealloc_callback_t dealloc, void* data)
: MemoryManager(agent, alloc, dealloc, data) {}
void CreateOutputBuf(size_t size) override {
aqlprofile_buffer_desc_flags_t flags{};
flags.host_access = flags.device_access = true;
flags.memory_hint = AQLPROFILE_MEMORY_HINT_DEVICE_UNCACHED;
outputbuf = AllocMemory(size, flags);
outputbuf_size = size;
}
std::vector<EventRequest>& GetEvents() { return events; }
void CopyEvents(const aqlprofile_pmc_event_t* events, size_t count);
protected:
std::vector<EventRequest> events;
};
class TraceMemoryManager : public MemoryManager {
public:
TraceMemoryManager(hsa_agent_t agent, aqlprofile_memory_alloc_callback_t alloc,
aqlprofile_memory_dealloc_callback_t dealloc,
aqlprofile_memory_copy_t _copy_fn, void* data)
: MemoryManager(agent, alloc, dealloc, data), copy_fn(_copy_fn) {}
TraceMemoryManager(aqlprofile_agent_handle_t agent, aqlprofile_memory_alloc_callback_t alloc,
aqlprofile_memory_dealloc_callback_t dealloc, void* data)
: MemoryManager(agent, alloc, dealloc, data) {}
void CreateOutputBuf(size_t size) override {
aqlprofile_buffer_desc_flags_t flags{};
flags.device_access = true;
flags.memory_hint = AQLPROFILE_MEMORY_HINT_DEVICE_NONCOHERENT;
outputbuf = AllocMemory(size, flags);
outputbuf_size = size;
}
void CreateTraceControlBuf(size_t size) {
aqlprofile_buffer_desc_flags_t flags{};
flags.host_access = flags.device_access = true;
flags.memory_hint = AQLPROFILE_MEMORY_HINT_HOST;
trace_control_buf = AllocMemory(size, flags);
}
const std::vector<hsa_ven_amd_aqlprofile_parameter_t>& GetATTParams() const { return att_params; }
void CopyATTParams(hsa_ven_amd_aqlprofile_parameter_t* params, size_t count) {
for (size_t i = 0; i < count; i++) this->att_params.push_back(params[i]);
for (auto& param : att_params) {
if (param.parameter_name == HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_COMPUTE_UNIT_TARGET)
target_cu = param.value;
else if (param.parameter_name == HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_SIMD_SELECTION)
simd_mask = param.value;
}
}
template <typename Type>
Type* GetTraceControlBuf() const {
return reinterpret_cast<Type*>(trace_control_buf.get());
}
void CopyMemory(void* dst, const void* src, size_t size) {
this->copy_fn(dst, src, size, this->userdata);
}
int GetSimdMask() const { return simd_mask; }
pm4_builder::TraceConfig config{};
protected:
int target_cu = -1;
int simd_mask = 0xF;
aqlprofile_memory_copy_t copy_fn;
std::vector<hsa_ven_amd_aqlprofile_parameter_t> att_params;
std::unique_ptr<void, MemoryDeleter> trace_control_buf = nullptr;
};
class CodeobjMemoryManager : public MemoryManager {
public:
CodeobjMemoryManager(hsa_agent_t agent, aqlprofile_memory_alloc_callback_t alloc,
aqlprofile_memory_dealloc_callback_t dealloc, size_t size, void* data)
: MemoryManager(agent, alloc, dealloc, data) {
aqlprofile_buffer_desc_flags_t flags{};
flags.host_access = flags.device_access = true;
this->cmd_buffer = AllocMemory(size, flags);
}
void CreateOutputBuf(size_t size) override{};
std::unique_ptr<void, MemoryDeleter> cmd_buffer;
};
+103
View File
@@ -0,0 +1,103 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include <iostream>
#include <cstdint>
#include <unordered_map>
#include <memory>
#include <mutex>
#include <stdexcept>
#include <shared_mutex>
#include "ip_discovery.h"
#define __maybe_unused __attribute__((__unused__))
#include "linux/registers/sienna_cichlid_ip_offset.h"
#include "util/reg_offsets.h"
#define LOG_VERBOSE 0
namespace {
void LogErrors(std::string msg) {
#if LOG_VERBOSE
std::cerr << msg << std::endl;
#endif /* LOG_VERBOSE */
}
} // namespace
const reg_base_offset_table* sienna_cichlid_reg_base_init() {
static_assert(HWIP_MAX_INSTANCE >= MAX_INSTANCE,
"HWIP_MAX_INSTANCE must be greater than MAX_INSTANCE");
static_assert(HWIP_MAX_SEGMENT >= MAX_SEGMENT,
"HWIP_MAX_SEGMENT must be greater than MAX_SEGMENT");
static const auto* sienna_cichlid_reg_table = []() {
auto* reg_table = new reg_base_offset_table();
// helper lambda to initialize blocks
auto init_hwip = [&](amd_hw_ip_block_type hwip, const auto& base) {
for (uint32_t i = 0; i < MAX_INSTANCE; ++i) {
std::copy(std::begin(base.instance[i].segment), std::end(base.instance[i].segment),
std::begin(reg_table->reg_offset[hwip][i]));
}
};
// HW has more IP blocks, only initialize the blocks needed
init_hwip(GC_HWIP, GC_BASE);
init_hwip(ATHUB_HWIP, ATHUB_BASE);
return reg_table;
}();
return sienna_cichlid_reg_table;
}
const reg_base_offset_table* navi_ip_offset_table_discovery_sysfs(uint32_t domain, uint32_t bdf) {
// Read the drm device properties, which includes all the IP base offsets for a GPU card on the
// system.
discovery_table_t table;
try {
table = parse_ip_discovery(domain, bdf);
} catch (const std::exception& e) {
LogErrors("Error in IP discovery for domain=" + std::to_string(domain) +
" bdf=" + std::to_string(bdf) + ": \n" + e.what());
return nullptr;
}
// Note: never cleanup, keep in memory to prevent issue with global destructor
struct reg_base_offset_table* reg_table = new reg_base_offset_table();
// helper lambda to initialize blocks
auto init_hwip = [&](amd_hw_ip_block_type hwip, const auto& entry) {
std::copy(std::begin(entry.segments), std::end(entry.segments),
std::begin(reg_table->reg_offset[hwip][entry.instance]));
};
for (auto& entry : table) {
if (entry.ipname == "gc") {
init_hwip(GC_HWIP, entry);
} else if (entry.ipname == "athub") {
init_hwip(ATHUB_HWIP, entry);
}
}
return reg_table;
}
+269
View File
@@ -0,0 +1,269 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include <iostream>
#include <string>
#include <vector>
#include <memory>
#include <sstream>
#include <fstream>
#include <algorithm>
#include <filesystem>
#include <unordered_map>
#include <regex>
#include <iomanip>
#include <cassert>
#include "ip_discovery.h"
#define PCI_BUS_NUM(x) (((x) >> 8) & 0xff)
#define PCI_SLOT(devfn) (((devfn) >> 3) & 0x1f)
#define PCI_FUNC(devfn) ((devfn)&0x07)
namespace fs = std::filesystem;
namespace {
/**
* @brief Reads a single integer (decimal or hexadecimal) from a sysfs file.
*
* This helper function reads a file containing a single numeric value and parses it
* as either a decimal or hexadecimal integer, based on the provided flag.
*
* @param fname The path to the sysfs file containing the numeric value.
*
* @return An `std::optional<int>` containing the parsed integer if successful, or `std::nullopt`
* if the file does not exist, cannot be opened, or contains invalid data.
*/
std::optional<int> read_sysfs_single_int(const fs::path& path) {
std::ifstream file(path);
if (!file.is_open()) return std::nullopt; // Failed to open file
int value;
file >> value;
if (file.fail()) return std::nullopt; // Failed to parse data
file.close();
return value;
}
/**
* @brief Reads base address segments from a sysfs file.
*
* This helper function reads a file containing hexadecimal values representing
* base address segments and parses them into a `base_addr_segments_t` structure.
*
* @param fname The path to the sysfs file containing base address segments.
*
* @return An `std::optional<base_addr_segments_t>` containing the parsed base address segments
* if successful, or `std::nullopt` if the file does not exist, cannot be opened,
* or contains invalid data.
*/
std::optional<base_addr_segments_t> read_sysfs_base_addr_segments(const fs::path& path) {
std::ifstream file(path);
if (!file.is_open()) return std::nullopt; // Failed to open file
base_addr_segments_t segments{0};
std::string databuf;
size_t x = 0;
while (std::getline(file, databuf) && x < segments.size()) {
std::stringstream ss(databuf);
ss >> std::hex >> segments[x++];
if (ss.fail()) return std::nullopt; // Failed to parse data
}
return segments;
}
/**
* @brief Parses IP instances for a given die and IP name from the sysfs directory structure.
*
* This function reads attributes such as base address segments, version information,
* and instance number for each IP instance and stores them in the discovery table.
*
* @param die_num The die number associated with the IP instances.
* @param diepath The sysfs path to the die directory.
* @param ipname The name of the IP to be parsed.
*
* @return The discovery table where parsed IP instance data will be stored.
*/
discovery_table_t parse_ip_instances(int die_num, const fs::path& diepath,
const std::string& ipname) {
// /sys/bus/pci/devices/{domain_bdf_str}/ip_discovery/die{die_num}/{ipname}
const fs::path dir_path = fs::path(diepath) / ipname;
if (!fs::exists(dir_path) || !fs::is_directory(dir_path)) {
throw std::runtime_error("sysfs path does not exist or is not a directory: " +
dir_path.string());
}
discovery_table_t instances{};
// sub-folders in "/sys/bus/pci/devices/{domain_bdf_str}/ip_discovery/die{die_num}/{ipname}"
for (const auto& dir_entry : fs::directory_iterator(dir_path)) {
if (!std::isdigit(dir_entry.path().filename().string()[0])) continue;
discovery_table_entry_t table_entry{};
table_entry.die = die_num;
// "/sys/bus/pci/devices/{domain_bdf_str}/ip_discovery/die{die_num}/{ipname}/{instance_num}"
fs::path instance_path = dir_path / dir_entry.path().filename();
// base_addr list
if (auto segments = read_sysfs_base_addr_segments(instance_path / "base_addr"))
table_entry.segments = *segments;
else
throw std::runtime_error("Failed to read IP base_addr segments for ipname=" + ipname +
" die=" + std::to_string(die_num));
// major
if (auto major = read_sysfs_single_int(instance_path / "major"))
table_entry.major = *major;
else
throw std::runtime_error("Failed to read IP major version for ipname=" + ipname +
" die=" + std::to_string(die_num));
// minor
if (auto minor = read_sysfs_single_int(instance_path / "minor"))
table_entry.minor = *minor;
else
throw std::runtime_error("Failed to read IP minor version for ipname=" + ipname +
" die=" + std::to_string(die_num));
// revision
if (auto revision = read_sysfs_single_int(instance_path / "revision"))
table_entry.revision = *revision;
else
throw std::runtime_error("Failed to read IP revision for ipname=" + ipname +
" die=" + std::to_string(die_num));
// instance
if (auto instance = read_sysfs_single_int(instance_path / "num_instance"))
table_entry.instance = *instance;
else
throw std::runtime_error("Failed to read IP instance for ipname=" + ipname +
" die=" + std::to_string(die_num));
// convert name to lowercase
table_entry.ipname = ipname;
std::transform(table_entry.ipname.begin(), table_entry.ipname.end(), table_entry.ipname.begin(),
[](unsigned char c) { return std::tolower(c); });
instances.emplace_back(table_entry);
}
return instances;
}
/**
* @brief Generates a PCI domain BDF (Bus:Device.Function) string.
*
* This function converts the given PCI domain and BDF (Bus:Device.Function) values
* into a standardized string format: "Domain:Bus:Device.Function".
*
* @param domain The PCI domain number (32-bit unsigned integer).
* @param bdf The PCI Bus/Device/Function (BDF) value (32-bit unsigned integer).
*
* @return A string representing the PCI domain and BDF in the format "Domain:Bus:Device.Function".
* Example: "0000:47:00.0".
*
* @details
* - The domain is represented as a 4-digit hexadecimal value.
* - The bus is represented as a 2-digit hexadecimal value.
* - The device is represented as a 2-digit hexadecimal value.
* - The function is represented as a single decimal digit.
*/
std::string get_domain_bdf_str(uint32_t domain, uint32_t bdf) {
uint8_t pci_bus = PCI_BUS_NUM(bdf);
uint8_t pci_devfn = bdf & 0xFF;
uint8_t pci_dev = PCI_SLOT(pci_devfn);
uint8_t pci_func = 0; // PCI_FUNC(pci_devfn); // Future ToDo: Use the macro PCI_FUNC() to support
// multiple functions. For now, it's always zero.
std::stringstream ss;
ss << std::hex << std::setfill('0') << std::setw(4) << domain << ":" << std::setw(2)
<< static_cast<int>(pci_bus) << ":" << std::setw(2) << static_cast<int>(pci_dev) << "."
<< static_cast<int>(pci_func);
return ss.str();
}
} // namespace
/**
* @brief Parses IP discovery information for a given PCI domain and BDF (Bus:Device.Function).
*
* This function discovers IP instances for all dies associated with a given PCI device.
* It reads the sysfs directory structure to extract information about IP instances
* and populates the provided discovery table.
*
* @param domain The PCI domain number (32-bit unsigned integer).
* @param bdf The PCI Bus/Device/Function (BDF) value (32-bit unsigned integer).
* @return table The discovery table where parsed IP instance data will be stored.
*
* @throws std::runtime_error If the sysfs directory does not exist, is not a directory,
* or if no IP instances are found.
*
* @details
* - Constructs the sysfs path for the PCI device using the domain and BDF values.
* - Iterates over the dies in the `/sys/bus/pci/devices/{domain_bdf_str}/ip_discovery/die`
* directory.
* - For each die, iterates over the IP directories and calls `parse_ip_instances` to parse
* individual IP instance data.
* - If no IP instances are found, throws an exception.
*/
discovery_table_t parse_ip_discovery(uint32_t domain, uint32_t bdf) {
// /sys/bus/pci/devices/{domain_bdf_str}/ip_discovery/die
const fs::path die_path =
fs::path("/sys/bus/pci/devices") / get_domain_bdf_str(domain, bdf) / "ip_discovery/die";
if (!fs::exists(die_path) || !fs::is_directory(die_path)) {
throw std::runtime_error("sysfs path does not exist or is not a directory: " +
die_path.string());
}
discovery_table_t table{};
// iterate over every die
// subfolders in "/sys/bus/pci/devices/{domain_bdf_str}/ip_discovery/die"
for (const auto& die_entry : fs::directory_iterator(die_path)) {
if (!die_entry.is_directory()) continue;
// "/sys/bus/pci/devices/{domain_bdf_str}/ip_discovery/die/{die_num}"
const fs::path die_entry_path = die_entry.path();
int die_num = std::stoi(die_entry_path.filename());
// subfolders in "/sys/bus/pci/devices/{domain_bdf_str}/ip_discovery/die/{die_num}"
for (const auto& ip_entry : fs::directory_iterator(die_entry_path)) {
if (!ip_entry.is_directory()) continue;
const std::string filename = ip_entry.path().filename();
if (std::isalpha(filename[0])) {
const auto instances = parse_ip_instances(die_num, die_entry.path(), filename);
table.insert(table.end(), instances.begin(), instances.end());
}
}
}
if (table.empty()) {
throw std::runtime_error("No IP instances found");
}
return table;
}
+76
View File
@@ -0,0 +1,76 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "pm4_factory.h"
#include <mutex>
#include <shared_mutex>
namespace aql_profile {
namespace {
struct locked_agent_cache {
std::shared_mutex mutex;
std::unordered_map<uint64_t, AgentInfo> cache;
void add(uint64_t& agent_id, const AgentInfo& agent_info) {
auto lock = std::unique_lock{mutex};
agent_id = cache.size();
cache[agent_id] = agent_info;
}
const AgentInfo* get(uint64_t agent_id) {
auto lock = std::shared_lock{mutex};
auto it = cache.find(agent_id);
if (it == cache.end()) return nullptr;
return &it->second;
}
};
locked_agent_cache& get_cache() {
static auto* cache = new locked_agent_cache{};
return *cache;
}
} // namespace
aqlprofile_agent_handle_t RegisterAgent(const aqlprofile_agent_info_v1_t* agent_info) {
aqlprofile_agent_handle_t agent_id;
AgentInfo int_agent_info;
int_agent_info.cu_num = agent_info->cu_num;
int_agent_info.se_num = agent_info->se_num;
int_agent_info.xcc_num = agent_info->xcc_num;
int_agent_info.shader_arrays_per_se = agent_info->shader_arrays_per_se;
int_agent_info.domain = agent_info->domain;
int_agent_info.bdf_id = agent_info->location_id;
auto len = strlen(agent_info->agent_gfxip);
memset(int_agent_info.gfxip, 0, sizeof(int_agent_info.gfxip));
memcpy(int_agent_info.gfxip, agent_info->agent_gfxip,
(len >= sizeof(int_agent_info.gfxip) ? sizeof(int_agent_info.gfxip) - 1 : len));
get_cache().add(agent_id.handle, int_agent_info);
return agent_id;
}
const AgentInfo* GetAgentInfo(aqlprofile_agent_handle_t agent_id) {
return get_cache().get(agent_id.handle);
}
} // namespace aql_profile
+417
View File
@@ -0,0 +1,417 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#ifndef SRC_CORE_PM4_FACTORY_H_
#define SRC_CORE_PM4_FACTORY_H_
#include <assert.h>
#include <hsa/hsa_ext_amd.h>
#include <stdint.h>
#include <string.h>
#include <climits>
#include <map>
#include <mutex>
#include <sstream>
#include <string>
#include "core/include/aql_profile_v2.h"
#include "core/aql_profile.hpp"
#include "core/aql_profile_exception.h"
#include "def/gpu_block_info.h"
#include "pm4/cmd_builder.h"
#include "pm4/pmc_builder.h"
#include "pm4/spm_builder.h"
#include "pm4/sqtt_builder.h"
#include "util/hsa_rsrc_factory.h"
namespace aql_profile {
struct pm4_agent_info {
std::string agent_gfxip;
uint32_t cu_num;
uint32_t se_num;
uint32_t shader_arrays_per_se;
uint32_t xcc_num;
};
const AgentInfo* GetAgentInfo(aqlprofile_agent_handle_t agent_id);
aqlprofile_agent_handle_t RegisterAgent(const aqlprofile_agent_info_v1_t* agent_info);
// GPU enumeration
enum gpu_id_t {
INVAL_GPU_ID, // invalid GPU id
GFX9_GPU_ID, // generic Gfx9 id
MI100_GPU_ID, // Mi100 GPU id
MI200_GPU_ID, // Mi200 GPU id
MI300_GPU_ID, // Mi300 GPU id
MI350_GPU_ID, // Mi350 GPU id
GFX10_GPU_ID, // generic Gfx10 id
GFX11_GPU_ID, // generic Gfx11 id
GFX12_GPU_ID, // generic Gfx12 id
};
// Block info map class
class BlockInfoMap {
public:
BlockInfoMap(const GpuBlockInfo** table, const uint32_t& size)
: block_table_(table), block_count_(size / sizeof(uintptr_t)) {}
BlockInfoMap(const BlockInfoMap& map)
: block_table_(map.block_table_), block_count_(map.block_count_) {}
// Get block info for a given block id
const GpuBlockInfo* Get(const uint32_t& block_id) const {
return (block_id < block_count_) ? block_table_[block_id] : NULL;
}
// Find block by name
// Return block id or UINT32_MAX if not found
uint32_t Find(const char* name) const {
uint32_t index = 0;
while (index < block_count_) {
const GpuBlockInfo* entry = block_table_[index];
if (entry) {
if (strcmp(name, entry->name) == 0) break;
}
++index;
}
return (index == block_count_) ? UINT32_MAX : index;
}
private:
// Block info table
const GpuBlockInfo** const block_table_;
// Number of elements in the block info table
const uint32_t block_count_;
};
// Factory of PM4 builders
class Pm4Factory {
public:
typedef std::mutex mutex_t;
static Pm4Factory* Create(aqlprofile_agent_handle_t agent_info, bool concurrent = false);
static Pm4Factory* Create(const AgentInfo* agent_info, gpu_id_t gpu_id, bool concurrent);
// Create factory for a given agent
static Pm4Factory* Create(const hsa_agent_t agent, const bool concurrent = false);
// Create factory for a given profile
static Pm4Factory* Create(const profile_t* profile) {
// First check and save the mode
return Create(profile->agent, CheckConcurrent(profile));
}
// Destroy factory
static void Destroy();
// Return gpu id
gpu_id_t GetGpuId() const { return gpu_id_; }
// Is pmc to be profiled concurrently?
bool IsConcurrent() const { return concurrent_mode_; }
// Is getting SPM data using driver public API?
bool SpmKfdMode() const { return spm_kfd_mode_; }
// Return PM4 command builder
pm4_builder::CmdBuilder* GetCmdBuilder() const { return cmd_builder_; }
// Return PMC PM4 packets builder
pm4_builder::PmcBuilder* GetPmcBuilder() const { return pmc_builder_; }
// Return SPM PM4 packets builder
pm4_builder::SpmBuilder* GetSpmBuilder() const { return spm_builder_; }
// Return SQTT PM4 packets builder
pm4_builder::SqttBuilder* GetSqttBuilder() const { return sqtt_builder_; }
// Return Shader Engines number
uint32_t GetShaderEnginesNumber() const { return agent_info_->se_num; }
uint32_t GetShaderArraysNumber() const { return agent_info_->shader_arrays_per_se; }
uint32_t GetComputeUnitNumber() const { return agent_info_->cu_num; }
// Return SQTT buffer alignment
uint32_t GetSQTTBufferAlignment() const { return 0x1000; }
const char* GetGFX() const { return agent_info_->name; }
virtual bool IsGFX9() const { return false; }
virtual bool IsGFX10() const { return false; }
virtual bool IsGFX11() const { return false; }
virtual bool IsGFX12() const { return false; }
// Return number of XCC on the GPU
uint32_t GetXccNumber() const { return agent_info_->xcc_num; }
const GpuBlockInfo* GetBlockInfo(const aqlprofile_pmc_event_t* event) const {
const GpuBlockInfo* info = block_map_.Get(event->block_name);
if (info == NULL) throw std::runtime_error("Bad Block");
// Checking that the block index is in proper range
if (event->block_index >= info->instance_count) throw std::runtime_error("Bad Index");
// Checking that the counter event index is in proper range
#if 0
if (event->counter_id > info->event_id_max)
throw event_exception(std::string("Bad event ID, "), *event);
#endif
return info;
}
// Return block info foor a given event
const GpuBlockInfo* GetBlockInfo(const event_t* event) const {
const GpuBlockInfo* info = block_map_.Get(event->block_name);
if (info == NULL) throw event_exception(std::string("Bad block, "), *event);
// Checking that the block index is in proper range
if (event->block_index >= info->instance_count)
throw event_exception(std::string("Bad block index, "), *event);
// Checking that the counter event index is in proper range
#if 0
if (event->counter_id > info->event_id_max)
throw event_exception(std::string("Bad event ID, "), *event);
#endif
return info;
}
// Return block info for a given block id
const GpuBlockInfo* GetBlockInfo(const uint32_t& block_id) const {
return block_map_.Get(block_id);
}
virtual size_t GetNumEvents(uint32_t block_name) const {
size_t se_number = GetShaderEnginesNumber() / GetXccNumber();
size_t block_samples_count = 1;
auto* block_info = GetBlockInfo(block_name);
if (block_info->attr & CounterBlockSeAttr)
block_samples_count *= se_number;
if (block_info->attr & CounterBlockSaAttr)
block_samples_count *= 2;
if (block_info->attr & CounterBlockWgpAttr)
block_samples_count *= GetNumWGPs();
if ((block_info->attr & CounterBlockSqAttr) && IsGFX11()) // TODO: Move to CounterBlockWgpAttr
block_samples_count *= GetNumWGPs();
return block_samples_count;
}
virtual size_t GetBytesNeeded(uint32_t block_name) const {
return GetNumEvents(block_name) * GetXccNumber() * sizeof(uint64_t);
}
// Return block id for a given block name string
uint32_t FindBlock(const char* name) const { return block_map_.Find(name); }
/// Workaround for GFX11. PMC Builder overrides this.
virtual int GetNumWGPs() const {
if (pmc_builder_) return pmc_builder_->GetNumWGPs();
return 1;
};
virtual int GetAccumLowID() const { throw HSA_STATUS_ERROR_INVALID_ARGUMENT; };
virtual int GetAccumHiID() const { throw HSA_STATUS_ERROR_INVALID_ARGUMENT; };
protected:
explicit Pm4Factory(const BlockInfoMap& map)
: cmd_builder_(NULL),
pmc_builder_(NULL),
spm_builder_(NULL),
sqtt_builder_(NULL),
agent_info_(NULL),
concurrent_mode_(concurrent_create_mode_),
block_map_(map) {}
virtual ~Pm4Factory() {
delete cmd_builder_;
delete pmc_builder_;
delete spm_builder_;
delete sqtt_builder_;
}
// PM4 command builder
pm4_builder::CmdBuilder* cmd_builder_;
// PMC PM4 packets builder
pm4_builder::PmcBuilder* pmc_builder_;
// SPM PM4 packets builder
pm4_builder::SpmBuilder* spm_builder_;
// SQTT PM4 packets builder
pm4_builder::SqttBuilder* sqtt_builder_;
// agent info
const AgentInfo* agent_info_;
gpu_id_t gpu_id_;
// Concurrent mode
static bool concurrent_create_mode_;
static bool spm_kfd_mode_;
bool concurrent_mode_;
private:
// PM4 factory instance map type
struct instances_fncomp_t {
bool operator()(const hsa_agent_t& a, const hsa_agent_t& b) const {
return a.handle < b.handle;
}
};
typedef std::map<hsa_agent_t, Pm4Factory*, instances_fncomp_t> instances_t;
// Create GFX9 generic factory
static Pm4Factory* Gfx9Create(const AgentInfo* agent_info);
// Create GFX10 generic factory
static Pm4Factory* Gfx10Create(const AgentInfo* agent_info);
// Create GFX11 generic factory
static Pm4Factory* Gfx11Create(const AgentInfo* agent_info);
// Create GFX12 generic factory
static Pm4Factory* Gfx12Create(const AgentInfo* agent_info);
// Create MI100 factory
static Pm4Factory* Mi100Create(const AgentInfo* agent_info);
// Create MI200 factory
static Pm4Factory* Mi200Create(const AgentInfo* agent_info);
// Create MI300 factory
static Pm4Factory* Mi300Create(const AgentInfo* agent_info);
// Create MI350 factory
static Pm4Factory* Mi350Create(const AgentInfo* agent_info);
// Return GPU id for a given agent
static gpu_id_t GetGpuId(std::string_view);
static bool CheckConcurrent(const profile_t* profile);
// Mutex for inter thread synchronization for the instances create/destroy
static mutex_t mutex_;
// Factory instances container
static instances_t* instances_;
// Block info container
const BlockInfoMap block_map_;
};
inline Pm4Factory* Pm4Factory::Create(const AgentInfo* agent_info, gpu_id_t gpu_id,
bool concurrent) {
// Check if we have the instance already created
if (instances_ == NULL) instances_ = new instances_t;
const auto ret = instances_->insert({agent_info->dev_id, NULL});
instances_t::iterator it = ret.first;
concurrent_create_mode_ = concurrent;
static bool spm_kfd = getenv("ROCP_SPM_KFD_MODE") != NULL;
spm_kfd_mode_ = spm_kfd;
// Create a factory implementation for the GPU id
if (ret.second) {
switch (gpu_id) {
// Create Gfx9 generic factory
case GFX9_GPU_ID:
it->second = Gfx9Create(agent_info);
break;
// Create Gfx10 generic factory
case GFX10_GPU_ID:
it->second = Gfx10Create(agent_info);
break;
// Create Gfx11 generic factory
case GFX11_GPU_ID:
it->second = Gfx11Create(agent_info);
break;
case GFX12_GPU_ID:
it->second = Gfx12Create(agent_info);
break;
// Create MI100 generic factory
case MI100_GPU_ID:
it->second = Mi100Create(agent_info);
break;
case MI200_GPU_ID:
it->second = Mi200Create(agent_info);
break;
case MI300_GPU_ID:
it->second = Mi300Create(agent_info);
break;
case MI350_GPU_ID:
it->second = Mi350Create(agent_info);
break;
default:
throw aql_profile_exc_val<gpu_id_t>("GPU id error", gpu_id);
}
}
if (it->second == NULL) throw aql_profile_exc_msg("Pm4Factory::Create() failed");
it->second->gpu_id_ = gpu_id;
return it->second;
}
// Create PM4 factory
inline Pm4Factory* Pm4Factory::Create(const hsa_agent_t agent, bool concurrent) {
std::lock_guard<mutex_t> lck(mutex_);
const AgentInfo* agent_info = HsaRsrcFactory::Instance().GetAgentInfo(agent);
// Get GPU id for a given agent
hsa_status_t status = HSA_STATUS_ERROR;
std::vector<char> agent_name{};
agent_name.resize(64);
uint32_t device_id = 0;
// Getting GfxIP name
status = hsa_agent_get_info(agent, HSA_AGENT_INFO_NAME, agent_name.data());
if (status == HSA_STATUS_SUCCESS) {
// Getting DeviceId
hsa_agent_info_t attribute = static_cast<hsa_agent_info_t>(HSA_AMD_AGENT_INFO_CHIP_ID);
status = hsa_agent_get_info(agent, attribute, &device_id);
}
if (status != HSA_STATUS_SUCCESS) {
throw aql_profile_exc_msg("Pm4Factory::Create() bad agent");
}
const gpu_id_t gpu_id = GetGpuId(agent_name.data());
return Pm4Factory::Create(agent_info, gpu_id, concurrent);
}
inline Pm4Factory* Pm4Factory::Create(aqlprofile_agent_handle_t agent_info, bool concurrent) {
const auto* info = GetAgentInfo(agent_info);
if (info == NULL) throw aql_profile_exc_val<uint64_t>("Bad agent handle", agent_info.handle);
const gpu_id_t gpu_id = GetGpuId(info->gfxip);
return Pm4Factory::Create(info, gpu_id, concurrent);
}
// Destroy PM4 factory
inline void Pm4Factory::Destroy() {
std::lock_guard<mutex_t> lck(mutex_);
if (instances_ != NULL) {
for (auto& item : *instances_) delete item.second;
delete instances_;
instances_ = NULL;
}
}
// Check the setting of pmc profiling mode
inline bool Pm4Factory::CheckConcurrent(const profile_t* profile) {
for (const hsa_ven_amd_aqlprofile_parameter_t* p = profile->parameters;
p < (profile->parameters + profile->parameter_count); ++p) {
if (p->parameter_name == HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_K_CONCURRENT) return true;
}
return false;
}
// Return GPU id for a given agent
inline gpu_id_t Pm4Factory::GetGpuId(std::string_view gfx_ip) {
std::vector<std::pair<std::string, gpu_id_t>> gfxip_map = {
{"gfx908", MI100_GPU_ID}, {"gfx90a", MI200_GPU_ID}, {"gfx900", GFX9_GPU_ID},
{"gfx902", GFX9_GPU_ID}, {"gfx906", GFX9_GPU_ID}, {"gfx94", MI300_GPU_ID},
{"gfx95", MI350_GPU_ID}, {"gfx10", GFX10_GPU_ID}, {"gfx11", GFX11_GPU_ID},
{"gfx12", GFX12_GPU_ID},
};
for (const auto& [name, id] : gfxip_map) {
if (gfx_ip.rfind(name, 0) == 0) {
return id;
}
}
return INVAL_GPU_ID;
}
} // namespace aql_profile
#endif // SRC_CORE_PM4_FACTORY_H_
+71
View File
@@ -0,0 +1,71 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include <assert.h>
#include <iomanip>
#include <iostream>
#include <sstream>
#include "core/amd_aql_pm4_ib_packet.h"
#include "core/aql_profile.hpp"
#include "pm4/cmd_builder.h"
namespace aql_profile {
void PopulateAql(const uint32_t* ib_packet, packet_t* aql_packet) {
// Populate relevant fields of Aql pkt
// Size of IB pkt is four DWords
// Header and completion sinal are not set
amd_aql_pm4_ib_packet_t* aql_pm4_ib = reinterpret_cast<amd_aql_pm4_ib_packet_t*>(aql_packet);
aql_pm4_ib->pm4_ib_format = AMD_AQL_PM4_IB_FORMAT;
aql_pm4_ib->pm4_ib_command[0] = ib_packet[0];
aql_pm4_ib->pm4_ib_command[1] = ib_packet[1];
aql_pm4_ib->pm4_ib_command[2] = ib_packet[2];
aql_pm4_ib->pm4_ib_command[3] = ib_packet[3];
aql_pm4_ib->dw_count_remain = AMD_AQL_PM4_IB_DW_COUNT_REMAIN;
for (unsigned i = 0; i < AMD_AQL_PM4_IB_RESERVED_COUNT; ++i) {
aql_pm4_ib->reserved[i] = 0;
}
#if defined(DEBUG_TRACE)
const uint32_t* dwords = (uint32_t*)aql_packet;
const uint32_t dword_count = sizeof(*aql_packet) / sizeof(uint32_t);
std::ostringstream oss;
oss << "AQL 'IB' size(" << dword_count << ")";
std::clog << std::setw(40) << std::left << "AQL 'IB' size(16)"
<< ":";
for (unsigned idx = 0; idx < dword_count; idx++) {
std::clog << " " << std::hex << std::setw(8) << std::setfill('0') << dwords[idx];
}
std::clog << std::setfill(' ') << std::endl;
#endif
}
void PopulateAql(const void* cmd_buffer, uint32_t cmd_size, pm4_builder::CmdBuilder* cmd_writer,
packet_t* aql_packet) {
pm4_builder::CmdBuffer ib_buffer;
cmd_writer->BuildIndirectBufferCmd(&ib_buffer, cmd_buffer, (size_t)cmd_size);
PopulateAql((const uint32_t*)ib_buffer.Data(), aql_packet);
}
} // namespace aql_profile
+239
View File
@@ -0,0 +1,239 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include <dirent.h>
#include "hsa/hsa_ext_amd.h"
#include <pthread.h>
#include <sys/types.h>
#include <unistd.h>
#include <atomic>
#include <iostream>
#include <mutex>
#include <string>
#include <thread>
#include "core/aql_profile.hpp"
#include "core/logger.h"
#include "core/pm4_factory.h"
#define PTHREAD_CALL(call) \
do { \
int err = call; \
if (err != 0) { \
errno = err; \
perror(#call); \
abort(); \
} \
} while (0)
namespace spm_kfd_namespace {
int get_gpu_node_id(uint32_t gpu_ind) {
int gpu_node = -1;
uint32_t index = 0;
// find a valid gpu node from /sys/class/kfd/kfd/topology/nodes
std::string path = "/sys/class/kfd/kfd/topology/nodes";
DIR* dir;
struct dirent* ent;
if ((dir = opendir(path.c_str())) != NULL) {
while ((ent = readdir(dir)) != NULL) {
std::string dir = ent->d_name;
if (dir.find_first_not_of("0123456789") == std::string::npos) {
std::string file = path + "/" + ent->d_name + "/gpu_id";
std::ifstream infile(file);
int id;
infile >> id;
if ((id != 0) && (index == gpu_ind)) {
++index;
gpu_node = atoi(ent->d_name);
break;
}
}
}
closedir(dir);
}
if (gpu_node == -1) {
printf("get_gpu_node_id`error: GPU[%d] not found\n", gpu_ind);
fflush(stdout);
abort();
}
return gpu_node;
}
int get_gpu_node_id(hsa_agent_t agent) {
const uint32_t gpu_ind = HsaRsrcFactory::Instance().GetAgentInfo(agent)->dev_index;
return get_gpu_node_id(gpu_ind);
}
struct state_t {
bool thread_stop;
int node_id;
uint32_t buf_size;
uint32_t timeout;
uint32_t data_size;
void* kfd_buf;
void* prod_buf;
void* cons_buf;
bool data_loss;
bool ready;
pthread_mutex_t work_mutex;
pthread_cond_t work_cond;
hsa_agent_t agent;
};
void producer_fun(state_t* state) {
uint32_t timeout = 0;
hsa_status_t status = HSA_STATUS_SUCCESS;
// hsa_amd_spm_set_dest_buffer(state->agent, state->buf_size, &timeout, &(state->data_size),
// state->kfd_buf, &(state->data_loss));
if (status != HSA_STATUS_SUCCESS) {
printf("hsa SPM Set DestBuffer init error\n");
fflush(stdout);
abort();
}
do {
timeout = state->timeout;
status = HSA_STATUS_SUCCESS;
// hsa_amd_spm_set_dest_buffer(state->agent, state->buf_size, &timeout, &(state->data_size),
// state->prod_buf, &(state->data_loss));
if (status != HSA_STATUS_SUCCESS) {
printf("hsa SPM Set DestBuffer error\n");
fflush(stdout);
abort();
}
PTHREAD_CALL(pthread_mutex_lock(&(state->work_mutex)));
void* tmp = state->prod_buf;
state->prod_buf = state->cons_buf;
state->cons_buf = state->kfd_buf;
state->kfd_buf = tmp;
state->ready = true;
PTHREAD_CALL(pthread_cond_signal(&(state->work_cond)));
PTHREAD_CALL(pthread_mutex_unlock(&(state->work_mutex)));
} while (!state->thread_stop);
status = HSA_STATUS_SUCCESS;
// hsa_amd_spm_set_dest_buffer(state->agent, 0, &timeout, &(state->data_size), NULL,
// &(state->data_loss));
if (status != HSA_STATUS_SUCCESS) {
printf("hsa SPM Set DestBuffer stop error\n");
fflush(stdout);
abort();
}
}
void consumer_fun(state_t* state, hsa_ven_amd_aqlprofile_data_callback_t callback, void* data) {
const uint32_t sample_id = 0;
PTHREAD_CALL(pthread_mutex_lock(&(state->work_mutex)));
do {
while (state->ready == false) {
PTHREAD_CALL(pthread_cond_wait(&(state->work_cond), &(state->work_mutex)));
}
state->ready = false;
hsa_ven_amd_aqlprofile_info_data_t sample_info;
sample_info.sample_id = sample_id;
sample_info.trace_data.ptr = state->cons_buf;
sample_info.trace_data.size = state->data_size;
hsa_status_t status = callback(HSA_VEN_AMD_AQLPROFILE_INFO_TRACE_DATA, &sample_info, data);
if (status == HSA_STATUS_INFO_BREAK) {
status = HSA_STATUS_SUCCESS;
state->thread_stop = true;
break;
} else if (status != HSA_STATUS_SUCCESS) {
printf("SPM consumer callback failed\n");
abort();
}
} while (1);
PTHREAD_CALL(pthread_mutex_unlock(&(state->work_mutex)));
}
void mananger_fun(const hsa_ven_amd_aqlprofile_profile_t* profile,
hsa_ven_amd_aqlprofile_data_callback_t callback, void* data) {
state_t obj{};
const int gpu_node_id = get_gpu_node_id(profile->agent);
char* buf_ptr = (char*)(profile->output_buffer.ptr);
// SPM data buffer size 256 byte aligned
const uint32_t buf_size = (profile->output_buffer.size / 3) & ~(uint32_t(256) - 1);
obj.timeout = 1000000; // 1sec
obj.node_id = gpu_node_id;
obj.buf_size = buf_size;
obj.kfd_buf = buf_ptr;
obj.prod_buf = buf_ptr + buf_size;
obj.cons_buf = buf_ptr + 2 * buf_size;
obj.agent = profile->agent;
PTHREAD_CALL(pthread_mutex_init(&(obj.work_mutex), NULL));
PTHREAD_CALL(pthread_cond_init(&(obj.work_cond), NULL));
hsa_status_t status = HSA_STATUS_SUCCESS; // hsa_amd_spm_acquire(profile->agent);
if (status != HSA_STATUS_SUCCESS) {
printf("hsa SPM Acquire error\n");
fflush(stdout);
abort();
}
// spm threads
std::thread producer(producer_fun, &obj);
std::thread consumer(consumer_fun, &obj, callback, data);
producer.join();
consumer.join();
status = HSA_STATUS_SUCCESS; // hsa_amd_spm_release(profile->agent);
if (status != HSA_STATUS_SUCCESS) {
printf("hsa SPM Release error\n");
fflush(stdout);
abort();
}
}
typedef std::mutex spm_mutex_t;
spm_mutex_t spm_mutex;
// Getting SPM data using driver API
hsa_status_t spm_iterate_data(const hsa_ven_amd_aqlprofile_profile_t* profile,
hsa_ven_amd_aqlprofile_data_callback_t callback, void* data) {
std::lock_guard<spm_mutex_t> lck(spm_mutex);
static std::thread* t = NULL;
if (t == NULL) {
// spm manager thread
t = new std::thread(mananger_fun, profile, callback, data);
} else {
t->join();
}
return HSA_STATUS_SUCCESS;
}
} // namespace spm_kfd_namespace
+420
View File
@@ -0,0 +1,420 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include "core/aql_profile.hpp"
#include "core/include/aql_profile_v2.h"
#include <cstdint>
#include <future>
#include <map>
#include <string>
#include <vector>
#include <shared_mutex>
#include "core/logger.h"
#include "core/pm4_factory.h"
#include "pm4/cmd_builder.h"
#include "pm4/sqtt_builder.h"
#include "core/commandbuffermgr.hpp"
#include "memorymanager.hpp"
#define THREAD_TRACE_PREFIX_SIZE 0x100
#define DEFAULT_TRACE_BUFFER_SIZE (3 << 26)
typedef union {
struct {
uint64_t legacy_version : 13;
uint64_t gfx9_version2 : 3;
uint64_t DSIMDM : 4;
uint64_t DCU : 5;
uint64_t DSA : 1;
uint64_t SEID : 6;
uint64_t reserved2 : 32;
};
uint64_t raw;
} att_header_packet_t;
typedef enum {
ATT_MARKER_HEADER_CHANNEL = 0,
ATT_MARKER_SIZE_LO_CHANNEL,
ATT_MARKER_ADDR_LO_CHANNEL,
ATT_MARKER_ADDR_HI_CHANNEL,
ATT_MARKER_SIZE_HI_CHANNEL,
ATT_MARKER_ID_LO_CHANNEL,
ATT_MARKER_ID_HI_CHANNEL,
ATT_MARKER_WAIT_FOR_HEADER = 32
} att_marker_state;
typedef union {
struct {
uint32_t isUnload : 1; // 0 if code object is being loaded, 1 for unload
uint32_t bFromStart : 1; // Has this code object been loaded before thread trace started?
uint32_t legacy_id : 30; // Legacy code object ID, if it fits in 30 bits.
};
uint32_t raw;
} aqlprofile_att_header_marker_t;
inline att_header_packet_t getHeaderPacket(int SE, int CU, int SIMD) {
att_header_packet_t header{.raw = 0};
header.legacy_version = 0x11; // The thread trace viewer only sees gfx9 for 0x11
header.gfx9_version2 = 4;
header.SEID = SE;
header.DCU = CU;
header.DSIMDM = SIMD;
header.DSA = 0;
return header;
}
namespace aql_profile_v2 {
hsa_status_t _internal_aqlprofile_att_iterate_data(aqlprofile_handle_t handle,
aqlprofile_att_data_callback_t callback,
void* userdata) {
hsa_status_t status = HSA_STATUS_SUCCESS;
auto shared_memorymgr = MemoryManager::GetManager(handle.handle);
TraceMemoryManager* memorymgr = dynamic_cast<TraceMemoryManager*>(shared_memorymgr.get());
if (!memorymgr) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(memorymgr->GetAgent());
pm4_builder::SqttBuilder* sqttbuilder = pm4_factory->GetSqttBuilder();
const size_t se_number_total = pm4_factory->GetShaderEnginesNumber();
auto* control_ptr = memorymgr->GetTraceControlBuf<pm4_builder::TraceControl>();
// Check if SQTT buffer was wrapped
for (size_t se = 0; se < se_number_total; se++) {
if (control_ptr[se].status & sqttbuilder->GetUTCErrorMask()) {
ERR_LOGGING << "SQTT memory error received, SE(" << se << ")";
status = HSA_STATUS_ERROR_EXCEPTION;
} else if (control_ptr[se].status & sqttbuilder->GetBufferFullMask()) {
ERR2_LOGGING << "SQTT data buffer full, SE(" << se << ")";
if (status == HSA_STATUS_SUCCESS) status = HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
}
std::vector<size_t> sample_sizes(se_number_total, 0);
size_t max_sample_size = 0;
// The samples sizes are returned in the control buffer
for (uint64_t se_index = 0; se_index < se_number_total; se_index++) {
bool bMaskedIn = memorymgr->config.GetTargetCU(se_index) >= 0;
uint64_t sample_capacity = memorymgr->config.GetCapacity(se_index);
void* sample_ptr = reinterpret_cast<void*>(memorymgr->config.GetSEBaseAddr(se_index));
// WPTR specifies the index in thread trace buffer where next token will be
// written by hardware. The index is incremented by size of 32 bytes.
size_t wptr_mask = sqttbuilder->GetWritePtrMask();
size_t sample_size = (control_ptr[se_index].wptr & wptr_mask) * sqttbuilder->GetWritePtrBlk();
// GFX11 hardware bug workaround
if (pm4_factory->GetGpuId() == aql_profile::GFX11_GPU_ID) {
sample_size = sample_size - reinterpret_cast<uint64_t>(sample_ptr);
sample_size &= (1ull << 29) - 1;
}
if (sample_size >= sample_capacity) {
ERR_LOGGING << "SQTT data out of bounds, sample_id(" << se_index << ") size(" << sample_size
<< "/" << sample_capacity << ")";
sample_size = sample_capacity;
if (status == HSA_STATUS_SUCCESS) status = HSA_STATUS_ERROR_OUT_OF_RESOURCES;
}
sample_sizes.at(se_index) = sample_size;
max_sample_size = std::max(sample_size, max_sample_size);
}
std::vector<size_t> cpu_sample(max_sample_size / sizeof(size_t) + sizeof(att_header_packet_t), 0);
// The samples sizes are returned in the control buffer
for (uint64_t se_index = 0; se_index < se_number_total; se_index++) {
int target_cu = memorymgr->config.GetTargetCU(se_index);
if (target_cu < 0) continue;
void* sample_ptr = reinterpret_cast<void*>(memorymgr->config.GetSEBaseAddr(se_index));
size_t sample_size = sample_sizes.at(se_index);
size_t sample_size_plus_header = sample_size;
char* sample_data_ptr = (char*)cpu_sample.data();
if (pm4_factory->GetGpuId() < aql_profile::GFX10_GPU_ID) {
auto* header = reinterpret_cast<att_header_packet_t*>(cpu_sample.data());
*header = getHeaderPacket(se_index, target_cu, memorymgr->GetSimdMask());
sample_data_ptr += sizeof(att_header_packet_t);
sample_size_plus_header = sample_size + sizeof(att_header_packet_t);
}
memorymgr->CopyMemory((void*)sample_data_ptr, sample_ptr, sample_size);
callback(se_index, (void*)cpu_sample.data(), sample_size_plus_header, userdata);
}
return status;
}
hsa_status_t _internal_aqlprofile_att_create_packets(
aqlprofile_handle_t* handle, aqlprofile_att_control_aql_packets_t* packets,
aqlprofile_att_profile_t profile, aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb, aqlprofile_memory_copy_t copy_fn,
void* userdata) {
pm4_builder::CmdBuffer start_cmd;
pm4_builder::CmdBuffer stop_cmd;
aql_profile::Pm4Factory* pm4_factory = aql_profile::Pm4Factory::Create(profile.agent);
auto memorymgr =
std::make_shared<TraceMemoryManager>(profile.agent, alloc_cb, dealloc_cb, copy_fn, userdata);
auto& trace_config = memorymgr->config;
trace_config.vmIdMask = 0;
trace_config.simd_sel = 0xF;
trace_config.perfMASK = ~0u;
trace_config.se_mask = 0x11111111;
const size_t se_number_total = pm4_factory->GetShaderEnginesNumber();
size_t buffer_size = DEFAULT_TRACE_BUFFER_SIZE;
if (profile.parameters)
for (const auto* p = profile.parameters; p < profile.parameters + profile.parameter_count; p++)
switch (p->parameter_name) {
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_SE_MASK:
trace_config.se_mask = p->value & ((1ull << se_number_total) - 1);
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_COMPUTE_UNIT_TARGET:
if (p->value > 15)
throw aql_profile::aql_profile_exc_val<uint32_t>(
"ThreadTraceConfig: CuId must be between 0 and 15, TargetCu", p->value);
trace_config.targetCu = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_VM_ID_MASK:
trace_config.vmIdMask = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_MASK:
if ((p->value & 0x50) != 0)
throw aql_profile::aql_profile_exc_val<uint32_t>(
"ThreadTraceConfig: Mask should have bits [4,6] set to Zero, Mask", p->value);
trace_config.deprecated_mask = p->value;
trace_config.targetCu = p->value & 0xF;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_TOKEN_MASK:
if ((p->value & 0xFF000000) != 0)
throw aql_profile::aql_profile_exc_val<uint32_t>(
"ThreadTraceConfig: TokenMask should have bits [31:25] set to Zero, TokenMask",
p->value);
trace_config.deprecated_tokenMask = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_TOKEN_MASK2:
trace_config.deprecated_tokenMask2 = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_SAMPLE_RATE:
trace_config.sampleRate = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_K_CONCURRENT:
trace_config.concurrent = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_SIMD_SELECTION:
trace_config.simd_sel = p->value & 0xF;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_OCCUPANCY_MODE:
trace_config.occupancy_mode = p->value ? 1 : 0;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_ATT_BUFFER_SIZE:
buffer_size = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_PERFCOUNTER_MASK:
trace_config.perfMASK = p->value;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_PERFCOUNTER_CTRL:
trace_config.perfCTRL = ((p->value & 0x1F) << 8) | 0xFFFF007F;
break;
case HSA_VEN_AMD_AQLPROFILE_PARAMETER_NAME_PERFCOUNTER_NAME:
if (trace_config.perfcounters.size() >= 8) return HSA_STATUS_ERROR_INVALID_ARGUMENT;
trace_config.perfcounters.push_back({p->counter_id, p->simd_mask});
break;
default:
ERR_LOGGING << "Bad trace parameter name (" << p->parameter_name << ")";
return HSA_STATUS_ERROR_INVALID_ARGUMENT;
}
const size_t control_size = sizeof(pm4_builder::TraceControl) * se_number_total;
memorymgr->CreateTraceControlBuf(control_size + THREAD_TRACE_PREFIX_SIZE);
memorymgr->CreateOutputBuf(buffer_size);
MemoryManager::RegisterManager(memorymgr);
auto* control_ptr = memorymgr->GetTraceControlBuf<pm4_builder::TraceControl>();
trace_config.control_buffer_ptr = control_ptr;
trace_config.control_buffer_size = control_size;
trace_config.data_buffer_ptr = memorymgr->GetOutputBuf();
trace_config.data_buffer_size = memorymgr->GetOutputBufSize();
uint32_t se_per_xcc = pm4_factory->GetShaderEnginesNumber() / pm4_factory->GetXccNumber();
pm4_builder::SqttBuilder* sqtt_builder = pm4_factory->GetSqttBuilder();
// Generate start commands
sqtt_builder->Begin(&start_cmd, &trace_config);
// Generate stop commands
sqtt_builder->End(&stop_cmd, &trace_config);
// Copy generated commands
const size_t start_size = aql_profile::CommandBufferMgr::Align(start_cmd.Size());
const size_t stop_size = aql_profile::CommandBufferMgr::Align(stop_cmd.Size());
memorymgr->CreateCmdBuf(start_size + stop_size);
handle->handle = memorymgr->GetHandler();
pm4_builder::CmdBuilder* cmd_writer = pm4_factory->GetCmdBuilder();
uint8_t* cmdbuf = reinterpret_cast<uint8_t*>(memorymgr->GetCmdBuf());
copy_fn(cmdbuf, start_cmd.Data(), start_cmd.Size(), userdata);
aql_profile::PopulateAql(cmdbuf, start_cmd.Size(), cmd_writer, &packets->start_packet);
cmdbuf += start_size;
copy_fn(cmdbuf, stop_cmd.Data(), stop_cmd.Size(), userdata);
aql_profile::PopulateAql(cmdbuf, stop_cmd.Size(), cmd_writer, &packets->stop_packet);
return HSA_STATUS_SUCCESS;
}
// Method to populate the provided AQL packet with ATT Markers
hsa_status_t _internal_aqlprofile_att_codeobj_marker(
hsa_ext_amd_aql_pm4_packet_t* packet, aqlprofile_handle_t* handle,
aqlprofile_att_codeobj_data_t data, aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb, void* userdata) {
static auto* mut = new std::shared_mutex{};
static auto* factory_cache = new std::map<uint64_t, aql_profile::Pm4Factory*>{};
auto _slk = std::shared_lock{*mut};
if (factory_cache->find(data.agent.handle) == factory_cache->end()) {
_slk.unlock();
{
auto _unique = std::unique_lock{*mut};
factory_cache->emplace(data.agent.handle, aql_profile::Pm4Factory::Create(data.agent));
}
_slk.lock();
}
aql_profile::Pm4Factory* pm4_factory = factory_cache->at(data.agent.handle);
pm4_builder::SqttBuilder* sqttbuilder = pm4_factory->GetSqttBuilder();
pm4_builder::CmdBuilder* cmd_writer = pm4_factory->GetCmdBuilder();
pm4_builder::CmdBuffer commands;
if (!data.isUnload) {
sqttbuilder->InsertMarker(&commands, uint32_t(data.addr), ATT_MARKER_ADDR_LO_CHANNEL);
sqttbuilder->InsertMarker(&commands, data.addr >> 32, ATT_MARKER_ADDR_HI_CHANNEL);
sqttbuilder->InsertMarker(&commands, uint32_t(data.size), ATT_MARKER_SIZE_LO_CHANNEL);
sqttbuilder->InsertMarker(&commands, data.size >> 32, ATT_MARKER_SIZE_HI_CHANNEL);
}
aqlprofile_att_header_marker_t header{};
header.bFromStart = data.fromStart;
header.isUnload = data.isUnload;
if (data.id >= (1 << 30)) {
sqttbuilder->InsertMarker(&commands, uint32_t(data.id), ATT_MARKER_ID_LO_CHANNEL);
sqttbuilder->InsertMarker(&commands, data.id >> 32, ATT_MARKER_ID_HI_CHANNEL);
} else
header.legacy_id = data.id;
sqttbuilder->InsertMarker(&commands, header.raw, ATT_MARKER_HEADER_CHANNEL);
auto memorymgr = std::make_shared<CodeobjMemoryManager>(data.agent, alloc_cb, dealloc_cb,
commands.Size(), userdata);
MemoryManager::RegisterManager(memorymgr);
handle->handle = memorymgr->GetHandler();
void* cmdbuffer = memorymgr->cmd_buffer.get();
memcpy(cmdbuffer, commands.Data(), commands.Size());
aql_profile::PopulateAql(cmdbuffer, commands.Size(), cmd_writer, packet);
return HSA_STATUS_SUCCESS;
}
} // namespace aql_profile_v2
extern "C" {
// Method to populate the provided AQL packet with ATT Markers
PUBLIC_API hsa_status_t aqlprofile_att_codeobj_marker(
hsa_ext_amd_aql_pm4_packet_t* packet, aqlprofile_handle_t* handle,
aqlprofile_att_codeobj_data_t data, aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb, void* userdata) {
try {
return aql_profile_v2::_internal_aqlprofile_att_codeobj_marker(packet, handle, data, alloc_cb,
dealloc_cb, userdata);
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
} catch (...) {
return HSA_STATUS_ERROR;
}
return HSA_STATUS_SUCCESS;
}
PUBLIC_API hsa_status_t aqlprofile_att_iterate_data(aqlprofile_handle_t handle,
aqlprofile_att_data_callback_t callback,
void* userdata) {
try {
return aql_profile_v2::_internal_aqlprofile_att_iterate_data(handle, callback, userdata);
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
} catch (...) {
return HSA_STATUS_ERROR;
}
}
PUBLIC_API hsa_status_t aqlprofile_att_create_packets(
aqlprofile_handle_t* handle, aqlprofile_att_control_aql_packets_t* packets,
aqlprofile_att_profile_t profile, aqlprofile_memory_alloc_callback_t alloc_cb,
aqlprofile_memory_dealloc_callback_t dealloc_cb, aqlprofile_memory_copy_t copy_fn,
void* userdata) {
try {
return aql_profile_v2::_internal_aqlprofile_att_create_packets(
handle, packets, profile, alloc_cb, dealloc_cb, copy_fn, userdata);
} catch (hsa_status_t err) {
ERR_LOGGING << err;
return err;
} catch (std::exception& e) {
ERR_LOGGING << e.what();
return HSA_STATUS_ERROR;
} catch (...) {
return HSA_STATUS_ERROR;
}
};
PUBLIC_API void aqlprofile_att_delete_packets(aqlprofile_handle_t handle) {
try {
MemoryManager::DeleteManager(handle.handle);
} catch (std::exception& e) {
return;
} catch (...) {
return;
}
}
} // extern "C"
+73
View File
@@ -0,0 +1,73 @@
// MIT License
//
// Copyright (c) 2017-2025 Advanced Micro Devices, Inc.
//
// Permission is hereby granted, free of charge, to any person obtaining a copy
// of this software and associated documentation files (the "Software"), to deal
// in the Software without restriction, including without limitation the rights
// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
// copies of the Software, and to permit persons to whom the Software is
// furnished to do so, subject to the following conditions:
//
// The above copyright notice and this permission notice shall be included in
// all copies or substantial portions of the Software.
//
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
// THE SOFTWARE.
#include <cstdint>
#include <cstring>
#include <mutex>
#include "linux/registers/vega20_ip_offset.h"
#include "util/reg_offsets.h"
#include "util/soc15_common.h"
const reg_base_offset_table* vega20_reg_base_init() {
static_assert(HWIP_MAX_INSTANCE >= MAX_INSTANCE,
"HWIP_MAX_INSTANCE must be greater than MAX_INSTANCE");
static_assert(HWIP_MAX_SEGMENT >= MAX_SEGMENT,
"HWIP_MAX_SEGMENT must be greater than MAX_SEGMENT");
static const auto* vega20_reg_table = []() {
auto* reg_table = new reg_base_offset_table();
// helper lambda to initialize blocks
auto init_hwip = [&](amd_hw_ip_block_type hwip, const auto& base) {
for (uint32_t i = 0; i < MAX_INSTANCE; i++) {
std::copy(std::begin(base.instance[i].segment), std::end(base.instance[i].segment),
std::begin(reg_table->reg_offset[hwip][i]));
}
};
// Initialize all HWIP blocks
init_hwip(GC_HWIP, GC_BASE);
init_hwip(HDP_HWIP, HDP_BASE);
init_hwip(MMHUB_HWIP, MMHUB_BASE);
init_hwip(ATHUB_HWIP, ATHUB_BASE);
init_hwip(NBIO_HWIP, NBIO_BASE);
init_hwip(MP0_HWIP, MP0_BASE);
init_hwip(MP1_HWIP, MP1_BASE);
init_hwip(UVD_HWIP, UVD_BASE);
init_hwip(VCE_HWIP, VCE_BASE);
init_hwip(DF_HWIP, DF_BASE);
init_hwip(DCE_HWIP, DCE_BASE);
init_hwip(OSSSYS_HWIP, OSSSYS_BASE);
init_hwip(SDMA0_HWIP, SDMA0_BASE);
init_hwip(SDMA1_HWIP, SDMA1_BASE);
init_hwip(SMUIO_HWIP, SMUIO_BASE);
init_hwip(NBIF_HWIP, NBIO_BASE);
init_hwip(THM_HWIP, THM_BASE);
init_hwip(CLK_HWIP, CLK_BASE);
init_hwip(UMC_HWIP, UMC_BASE);
init_hwip(RSMU_HWIP, RSMU_BASE);
return reg_table;
}();
return vega20_reg_table;
}