[SWDEV-518325/SWDEV-518320/SWDEV-443309] Fix Partition Enumeration

* Changes:
  - Updates to DRM renderD* / card* pathing for partition
  - Now use KFD to discover AMD devices and populate accordingly
    Device MUST have an accessible KFD node (via cgroups)
  - Updated serveral AMD SMI CLI outputs to handle SYSFS files
    which are not accessible on partition nodes
  - Tests are updated to handle not supported features
  - Added new method to help get card/drm info
    (rsmi_dev_device_identifiers_get) from ROCm SMI
  - Renamed device->get_card_id() & device->get_drm_render_minor()
    These can now be used on internal AMD SMI calls.
  - Removed warnings shown in build

Change-Id: Ice882fd9b97fb625a5bd4ef327f3ceaf247dc570
Signed-off-by: Charis Poag <Charis.Poag@amd.com>


[ROCm/amdsmi commit: 4782528770]
This commit is contained in:
Charis Poag
2025-04-02 14:08:48 -05:00
committato da Arif, Maisam
parent 484614fe9b
commit 8d4a4d7b14
15 ha cambiato i file con 764 aggiunte e 580 eliminazioni
+14 -5
Vedi File
@@ -619,10 +619,10 @@ amdsmi_get_gpu_enumeration_info(amdsmi_processor_handle processor_handle,
}
// Retrieve DRM Card ID
info->drm_card = gpu_device->get_card_from_bdf();
info->drm_card = gpu_device->get_card_id();
// Retrieve DRM Render ID
info->drm_render = gpu_device->get_render_id();
info->drm_render = gpu_device->get_drm_render_minor();
// Retrieve HIP ID (difference from the smallest node ID) and HSA ID
std::map<uint64_t, std::shared_ptr<amd::smi::KFDNode>> nodes;
@@ -2267,6 +2267,7 @@ amdsmi_get_gpu_accelerator_partition_profile_config(amdsmi_processor_handle proc
<< "\n profile_config->profiles[i].num_resources: "
<< profile_config->profiles[i].num_resources
<< std::endl;
// std::cout << ss.str() << std::endl;
LOG_DEBUG(ss);
}
@@ -2425,6 +2426,7 @@ amdsmi_get_gpu_accelerator_partition_profile_config(amdsmi_processor_handle proc
}
ss << __PRETTY_FUNCTION__
<< " | END returning " << smi_amdgpu_get_status_string(return_status, false);
// std::cout << ss.str() << std::endl;
LOG_INFO(ss);
return return_status;
@@ -2791,6 +2793,9 @@ amdsmi_get_gpu_metrics_header_info(amdsmi_processor_handle processor_handle,
{
AMDSMI_CHECK_INIT();
// nullptr api supported
if (header_value != nullptr) {
*header_value = amd_metrics_table_header_t{}; // Use a default initializer for the struct
}
return rsmi_wrapper(rsmi_dev_metrics_header_info_get, processor_handle, 0,
reinterpret_cast<metrics_table_header_t*>(header_value));
@@ -2802,7 +2807,7 @@ amdsmi_status_t amdsmi_get_gpu_metrics_info(
AMDSMI_CHECK_INIT();
// nullptr api supported
if (pgpu_metrics != nullptr) {
*pgpu_metrics = {};
*pgpu_metrics = amdsmi_gpu_metrics_t{}; // Use a default initializer for the struct
}
return rsmi_wrapper(rsmi_dev_gpu_metrics_info_get, processor_handle, 0,
reinterpret_cast<rsmi_gpu_metrics_t*>(pgpu_metrics));
@@ -3805,7 +3810,7 @@ amdsmi_get_gpu_cper_entries(
return status;
}
std::string path = std::string("/sys/kernel/debug/dri/") +
std::to_string(gpu_device->get_card_from_bdf()) +
std::to_string(gpu_device->get_card_id()) +
"/amdgpu_ring_cper";
@@ -3957,6 +3962,7 @@ amdsmi_status_t amdsmi_get_gpu_driver_info(amdsmi_processor_handle processor_han
amdsmi_status_t amdsmi_get_pcie_info(amdsmi_processor_handle processor_handle, amdsmi_pcie_info_t *info) {
AMDSMI_CHECK_INIT();
std::ostringstream ss;
if (info == nullptr) {
return AMDSMI_STATUS_INVAL;
@@ -3984,7 +3990,10 @@ amdsmi_status_t amdsmi_get_pcie_info(amdsmi_processor_handle processor_handle, a
fscanf(fp, "%d", &pcie_width);
fclose(fp);
} else {
printf("Failed to open file: %s \n", path_max_link_width.c_str());
ss << __PRETTY_FUNCTION__
<< " | Failed to open file: " << path_max_link_width
<< " | returning AMDSMI_STATUS_API_FAILED";
LOG_ERROR(ss);
return AMDSMI_STATUS_API_FAILED;
}
info->pcie_static.max_pcie_width = (uint16_t)pcie_width;
@@ -42,6 +42,64 @@ uint32_t AMDSmiGPUDevice::get_gpu_id() const {
return gpu_id_;
}
uint32_t AMDSmiGPUDevice::get_card_id() {
std::ostringstream ss;
// Should never return not_supported, but just in case
rsmi_status_t ret = rsmi_status_t::RSMI_STATUS_NOT_SUPPORTED;
uint32_t gpu_index = this->get_gpu_id();
rsmi_device_identifiers_t identifiers = rsmi_device_identifiers_t{};
ret = rsmi_dev_device_identifiers_get(gpu_index, &identifiers);
if (ret != rsmi_status_t::RSMI_STATUS_SUCCESS) {
this->card_index_ = std::numeric_limits<uint32_t>::max();
} else {
this->card_index_ = identifiers.card_index;
}
ss << __PRETTY_FUNCTION__
<< " | rsmi_dev_identifiers_get status: " << getRSMIStatusString(ret, false) << "\n"
<< " | gpu_id_: " << gpu_id_ << "\n"
<< " | identifiers.card_index: " << identifiers.card_index << "\n"
<< " | identifiers.drm_render_minor: " << identifiers.drm_render_minor << "\n"
<< " | identifiers.bdfid: " << std::hex << "0x" << identifiers.bdfid << "\n"
<< " | identifiers.kfd_gpu_id: " << std::dec << identifiers.kfd_gpu_id << "\n"
<< " | identifiers.partition_id: " << identifiers.partition_id << "\n"
<< " | identifiers.smi_device_id: " << identifiers.smi_device_id << "\n"
<< " | returning card_index_: "
<< this->card_index_ << std::endl;
// std::cout << ss.str();
LOG_DEBUG(ss);
return this->card_index_;
}
uint32_t AMDSmiGPUDevice::get_drm_render_minor() {
std::ostringstream ss;
// Should never return not_supported, but just in case
rsmi_status_t ret = rsmi_status_t::RSMI_STATUS_NOT_SUPPORTED;
uint32_t gpu_index = this->get_gpu_id();
rsmi_device_identifiers_t identifiers = rsmi_device_identifiers_t{};
ret = rsmi_dev_device_identifiers_get(gpu_index, &identifiers);
if (ret != rsmi_status_t::RSMI_STATUS_SUCCESS) {
this->drm_render_minor_ = std::numeric_limits<uint32_t>::max();
} else {
this->drm_render_minor_ = identifiers.drm_render_minor;
}
ss << __PRETTY_FUNCTION__
<< " | rsmi_dev_identifiers_get status: " << getRSMIStatusString(ret, false) << "\n"
<< " | gpu_id_: " << gpu_id_ << "\n"
<< " | identifiers.card_index: " << identifiers.card_index << "\n"
<< " | identifiers.drm_render_minor: " << identifiers.drm_render_minor << "\n"
<< " | identifiers.bdfid: " << std::hex << "0x" << identifiers.bdfid << "\n"
<< " | identifiers.kfd_gpu_id: " << std::dec << identifiers.kfd_gpu_id << "\n"
<< " | identifiers.partition_id: " << identifiers.partition_id << "\n"
<< " | identifiers.smi_device_id: " << identifiers.smi_device_id << "\n"
<< " | returning drm_render_minor_: "
<< this->drm_render_minor_ << std::endl;
// std::cout << ss.str();
LOG_DEBUG(ss);
return this->drm_render_minor_;
}
uint32_t AMDSmiGPUDevice::get_gpu_fd() const {
return fd_;
}
@@ -323,81 +381,6 @@ std::string AMDSmiGPUDevice::bdf_to_string() const {
}
uint32_t AMDSmiGPUDevice::get_card_from_bdf() const {
const std::string drm_path = "/sys/class/drm/";
DIR* dir = opendir(drm_path.c_str());
if (!dir) {
return std::numeric_limits<uint32_t>::max();
}
struct dirent* entry;
while ((entry = readdir(dir)) != nullptr) {
std::string device_name = entry->d_name;
// Check if the entry starts with "card"
if (device_name.find("card") == 0) {
const std::string card_path = drm_path + device_name + "/device";
// Open the uevent file for the device
std::ifstream uevent_file(card_path + "/uevent");
if (!uevent_file) {
continue; // Skip if the file is not found
}
std::string line;
while (std::getline(uevent_file, line)) {
// Check for the PCI_SLOT_NAME and if it contains the BDF
if (line.rfind("PCI_SLOT_NAME", 0) == 0 && line.find(bdf_to_string()) != std::string::npos) {
closedir(dir);
return std::stoi(device_name.substr(4)); // Convert extracted number to int
}
}
}
}
closedir(dir);
return std::numeric_limits<uint32_t>::max(); // Return -1 if no matching card is found
}
uint32_t AMDSmiGPUDevice::get_render_id() const {
const std::string drm_path = "/sys/class/drm/";
DIR* dir = opendir(drm_path.c_str());
if (!dir) {
return std::numeric_limits<uint32_t>::max();
}
struct dirent* entry;
while ((entry = readdir(dir)) != nullptr) {
std::string device_name = entry->d_name;
// Check if the entry starts with "renderD"
if (device_name.find("renderD") == 0) {
const std::string render_path = drm_path + device_name + "/device";
// Open the uevent file for the device
std::ifstream uevent_file(render_path + "/uevent");
if (!uevent_file) {
continue; // Skip if the file is not found
}
std::string line;
while (std::getline(uevent_file, line)) {
// Check for the PCI_SLOT_NAME and if it contains the BDF
if (line.rfind("PCI_SLOT_NAME", 0) == 0 && line.find(bdf_to_string()) != std::string::npos) {
closedir(dir);
return std::stoi(device_name.substr(7)); // Extract only the number after "renderD"
}
}
}
}
closedir(dir);
return std::numeric_limits<uint32_t>::max(); // Return -1 if no matching render ID is found
}
} // namespace smi
} // namespace amd
@@ -115,7 +115,7 @@ int openFileAndModifyBuffer(std::string path, char *buff, size_t sizeOfBuff,
bool errorDiscovered = false;
std::ifstream file(path, std::ifstream::in);
std::string contents = {std::istreambuf_iterator<char>{file}, std::istreambuf_iterator<char>{}};
clearCharBufferAndReinitialize(buff, sizeOfBuff, contents);
clearCharBufferAndReinitialize(buff, static_cast<uint32_t>(sizeOfBuff), contents);
if (!file.is_open()) {
errorDiscovered = true;
} else {
@@ -453,21 +453,12 @@ amdsmi_status_t smi_amdgpu_get_bad_page_info(amd::smi::AMDSmiGPUDevice* device,
return AMDSMI_STATUS_SUCCESS;
}
static uint32_t GetDeviceIndex(const std::string s) {
std::string t = s;
size_t tmp = t.find_last_not_of("0123456789");
t.erase(0, tmp+1);
assert(stoi(t) >= 0);
return static_cast<uint32_t>(stoi(t));
}
amdsmi_status_t smi_amdgpu_get_bad_page_threshold(amd::smi::AMDSmiGPUDevice* device,
uint32_t *threshold) {
SMIGPUDEVICE_MUTEX(device->get_mutex())
//TODO: Accessing the node requires root privileges, and its interface may need to be exposed in another path
uint32_t index = GetDeviceIndex(device->get_gpu_path());
uint32_t index = device->get_card_id();
std::string fullpath = "/sys/kernel/debug/dri/" + std::to_string(index) + std::string("/ras/bad_page_cnt_threshold");
std::ifstream fs(fullpath.c_str());
@@ -489,7 +480,6 @@ amdsmi_status_t smi_amdgpu_get_bad_page_threshold(amd::smi::AMDSmiGPUDevice* dev
amdsmi_status_t smi_amdgpu_validate_ras_eeprom(amd::smi::AMDSmiGPUDevice* device) {
SMIGPUDEVICE_MUTEX(device->get_mutex())
//uint32_t index = GetDeviceIndex(device->get_gpu_path());
//TODO: need to expose the corresponding interface to validate the checksum of ras eeprom table.
//verify fail: return AMDSMI_STATUS_CORRUPTED_EEPROM
return AMDSMI_STATUS_NOT_SUPPORTED;