Implement memory fault analysis through context save area
When a fatal memory fault occurs the scheduler context-saves all queues
in the process and notifies the runtime through the memory event. The
saved state contains all GPR/LDS data at the moment of the fault.
Retrieve this state and present it to the user if HSA_DEBUG_FAULT is set
to "analyze" and the wavefront caused the fault. If amdgcn-capable objdump
is in the PATH invoke this to disassemble code around the PC.
Queue lifetime is now managed by the runtime to allow querying the
context save state for all active queues.
Change-Id: I6fee662fad1c4f9aa125bf5c53d7d0ea1ab32f95
[ROCm/ROCR-Runtime commit: 75c9506f9d]
This commit is contained in:
@@ -84,10 +84,9 @@ void* AqlQueue::operator new(size_t size) {
|
||||
|
||||
void AqlQueue::operator delete(void* ptr) { _aligned_free(ptr); }
|
||||
|
||||
AqlQueue::AqlQueue(GpuAgent* agent, size_t req_size_pkts, HSAuint32 node_id,
|
||||
ScratchInfo& scratch, core::HsaEventCallback callback,
|
||||
void* err_data, bool is_kv)
|
||||
: Queue(),
|
||||
AqlQueue::AqlQueue(GpuAgent* agent, size_t req_size_pkts, HSAuint32 node_id, ScratchInfo& scratch,
|
||||
core::HsaEventCallback callback, void* err_data, bool is_kv)
|
||||
: Queue(*agent),
|
||||
Signal(0),
|
||||
ring_buf_(NULL),
|
||||
ring_buf_alloc_bytes_(0),
|
||||
@@ -962,4 +961,113 @@ void AqlQueue::InitScratchSRD() {
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
WaveStates AqlQueue::GetWaveStates() {
|
||||
WaveStates wave_states;
|
||||
|
||||
// Retrieve the control stack and context save area for the queue.
|
||||
HsaQueueInfo queue_info;
|
||||
HSAKMT_STATUS status = hsaKmtGetQueueInfo(queue_id_, &queue_info);
|
||||
|
||||
if (status != HSAKMT_STATUS_SUCCESS) {
|
||||
return wave_states;
|
||||
}
|
||||
|
||||
// The control stack is processed from start to end.
|
||||
// The save area is processed from end to start.
|
||||
uint32_t* ctl_stack = reinterpret_cast<uint32_t*>(queue_info.ControlStackTop);
|
||||
uint32_t* wave_area = reinterpret_cast<uint32_t*>(uintptr_t(queue_info.UserContextSaveArea) +
|
||||
queue_info.SaveAreaSizeInBytes);
|
||||
uint32_t ctl_stack_ndw = uint32_t(queue_info.ControlStackUsedInBytes / sizeof(uint32_t));
|
||||
|
||||
// Control stack persists resource allocation until changed by a command.
|
||||
uint32_t n_vgprs = 0;
|
||||
uint32_t n_sgprs = 0;
|
||||
uint32_t lds_size_dw = 0;
|
||||
|
||||
// LDS is saved per-workgroup but the stack is parsed per-wavefront.
|
||||
// Track the LDS save area for the current workgroup.
|
||||
uint32_t* lds = nullptr;
|
||||
|
||||
// Parse each write to COMPUTE_RELAUNCH in sequence.
|
||||
// First two dwords are SET_SH_REG leader.
|
||||
for (uint32_t idx = 2; idx < ctl_stack_ndw; ++idx) {
|
||||
uint32_t relaunch = ctl_stack[idx];
|
||||
|
||||
#define COMPUTE_RELAUNCH_PAYLOAD_VGPRS(x) (((x) >> 0x0) & 0x3F)
|
||||
#define COMPUTE_RELAUNCH_PAYLOAD_SGPRS(x) (((x) >> 0x6) & 0x7)
|
||||
#define COMPUTE_RELAUNCH_PAYLOAD_LDS_SIZE(x) (((x) >> 0x9) & 0x1FF)
|
||||
#define COMPUTE_RELAUNCH_PAYLOAD_FIRST_WAVE(x) (((x) >> 0x11) & 0x1)
|
||||
#define COMPUTE_RELAUNCH_IS_EVENT(x) (((x) >> 0x1E) & 0x1)
|
||||
#define COMPUTE_RELAUNCH_IS_STATE(x) (((x) >> 0x1F) & 0x1)
|
||||
|
||||
bool is_event = COMPUTE_RELAUNCH_IS_EVENT(relaunch);
|
||||
bool is_state = COMPUTE_RELAUNCH_IS_STATE(relaunch);
|
||||
|
||||
if (is_state && !is_event) {
|
||||
// Resource allocation state change, update tracked state.
|
||||
n_vgprs = (0x1 + COMPUTE_RELAUNCH_PAYLOAD_VGPRS(relaunch)) * 0x4;
|
||||
n_sgprs = ((0x1 + COMPUTE_RELAUNCH_PAYLOAD_SGPRS(relaunch)) - 0x1 /* no trap SGPRs */) * 0x10;
|
||||
lds_size_dw = COMPUTE_RELAUNCH_PAYLOAD_LDS_SIZE(relaunch) * 0x80;
|
||||
} else if (!is_state && !is_event) {
|
||||
// Reference to one wavefront in the save area.
|
||||
bool first_wave_in_group = COMPUTE_RELAUNCH_PAYLOAD_FIRST_WAVE(relaunch);
|
||||
|
||||
// Save area layout is fixed by context save trap handler and SPI.
|
||||
uint32_t vgprs_offset = 0x0;
|
||||
uint32_t sgprs_offset = vgprs_offset + n_vgprs * 0x40;
|
||||
uint32_t hwregs_offset = sgprs_offset + n_sgprs;
|
||||
uint32_t lds_offset = hwregs_offset + 0x20;
|
||||
uint32_t unused_offset = lds_offset + (first_wave_in_group ? lds_size_dw : 0x0);
|
||||
uint32_t wave_area_size = unused_offset + 0x10; // trap SGPRs were allocated but not saved
|
||||
uint32_t hwreg_m0_offset = hwregs_offset + 0x0;
|
||||
uint32_t hwreg_pc_lo_offset = hwregs_offset + 0x1;
|
||||
uint32_t hwreg_pc_hi_offset = hwregs_offset + 0x2;
|
||||
uint32_t hwreg_exec_lo_offset = hwregs_offset + 0x3;
|
||||
uint32_t hwreg_exec_hi_offset = hwregs_offset + 0x4;
|
||||
uint32_t hwreg_status_offset = hwregs_offset + 0x5;
|
||||
uint32_t hwreg_trapsts_offset = hwregs_offset + 0x6;
|
||||
|
||||
// Find beginning of wavefront state in the save area.
|
||||
wave_area -= wave_area_size;
|
||||
|
||||
if (first_wave_in_group) {
|
||||
// Track the LDS save area for this workgroup.
|
||||
if (lds_size_dw > 0) {
|
||||
lds = wave_area + lds_offset;
|
||||
} else {
|
||||
lds = nullptr;
|
||||
}
|
||||
}
|
||||
|
||||
WaveState wave_state;
|
||||
|
||||
wave_state.num_sgprs = n_sgprs;
|
||||
wave_state.sgprs = wave_area + sgprs_offset;
|
||||
wave_state.num_vgprs = n_vgprs;
|
||||
wave_state.num_vgpr_lanes = 0x40;
|
||||
wave_state.vgprs = wave_area + vgprs_offset;
|
||||
wave_state.regs.pc = (uint64_t(wave_area[hwreg_pc_lo_offset]) |
|
||||
(uint64_t(wave_area[hwreg_pc_hi_offset]) << 0x20));
|
||||
wave_state.regs.exec = uint64_t(wave_area[hwreg_exec_lo_offset]) |
|
||||
(uint64_t(wave_area[hwreg_exec_hi_offset]) << 0x20);
|
||||
wave_state.regs.status = wave_area[hwreg_status_offset];
|
||||
wave_state.regs.trapsts = wave_area[hwreg_trapsts_offset];
|
||||
wave_state.regs.m0 = wave_area[hwreg_m0_offset];
|
||||
wave_state.lds_size_dw = lds_size_dw;
|
||||
wave_state.lds = lds;
|
||||
|
||||
#define SQ_WAVE_TRAPSTS_XNACK_ERROR(x) (((x) >> 0x1C) & 0x1)
|
||||
|
||||
if (SQ_WAVE_TRAPSTS_XNACK_ERROR(wave_state.regs.trapsts)) {
|
||||
// Correct the PC: context save handler subtracted 0x8.
|
||||
wave_state.regs.pc += 0x8;
|
||||
}
|
||||
|
||||
wave_states.push_back(wave_state);
|
||||
}
|
||||
}
|
||||
|
||||
return wave_states;
|
||||
}
|
||||
} // namespace amd
|
||||
|
||||
@@ -369,4 +369,32 @@ hsa_status_t CpuAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type,
|
||||
return HSA_STATUS_ERROR;
|
||||
}
|
||||
|
||||
hsa_status_t CpuAgent::HostQueueCreate(hsa_region_t region, uint32_t ring_size,
|
||||
hsa_queue_type32_t type, uint32_t features,
|
||||
hsa_signal_t doorbell_signal, core::Queue** queue) {
|
||||
core::HostQueue* host_queue =
|
||||
new core::HostQueue(*this, region, ring_size, type, features, doorbell_signal);
|
||||
|
||||
if (!host_queue->IsValid()) {
|
||||
delete host_queue;
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
}
|
||||
|
||||
queues_.emplace_back(host_queue);
|
||||
*queue = host_queue;
|
||||
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
hsa_status_t CpuAgent::QueueDestroy(core::Queue* queue) {
|
||||
auto it = std::find_if(
|
||||
queues_.begin(), queues_.end(),
|
||||
[&](std::unique_ptr<core::Queue>& queue_ptr) { return queue_ptr.get() == queue; });
|
||||
|
||||
assert(it != queues_.end() && "attempt to destroy an untracked queue");
|
||||
queues_.erase(it);
|
||||
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
} // namespace amd
|
||||
|
||||
@@ -0,0 +1,307 @@
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
//
|
||||
// The University of Illinois/NCSA
|
||||
// Open Source License (NCSA)
|
||||
//
|
||||
// Copyright (c) 2014-2015, Advanced Micro Devices, Inc. All rights reserved.
|
||||
//
|
||||
// Developed by:
|
||||
//
|
||||
// AMD Research and AMD HSA Software Development
|
||||
//
|
||||
// Advanced Micro Devices, Inc.
|
||||
//
|
||||
// www.amd.com
|
||||
//
|
||||
// Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
// of this software and associated documentation files (the "Software"), to
|
||||
// deal with the Software without restriction, including without limitation
|
||||
// the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
||||
// and/or sell copies of the Software, and to permit persons to whom the
|
||||
// Software is furnished to do so, subject to the following conditions:
|
||||
//
|
||||
// - Redistributions of source code must retain the above copyright notice,
|
||||
// this list of conditions and the following disclaimers.
|
||||
// - Redistributions in binary form must reproduce the above copyright
|
||||
// notice, this list of conditions and the following disclaimers in
|
||||
// the documentation and/or other materials provided with the distribution.
|
||||
// - Neither the names of Advanced Micro Devices, Inc,
|
||||
// nor the names of its contributors may be used to endorse or promote
|
||||
// products derived from this Software without specific prior written
|
||||
// permission.
|
||||
//
|
||||
// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
||||
// THE CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR
|
||||
// OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
|
||||
// ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
|
||||
// DEALINGS WITH THE SOFTWARE.
|
||||
//
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include "core/inc/amd_debugger.h"
|
||||
#include "core/inc/amd_loader_context.hpp"
|
||||
#include "core/inc/amd_aql_queue.h"
|
||||
|
||||
#include <cstdlib>
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <sys/wait.h>
|
||||
#include <unistd.h>
|
||||
|
||||
namespace amd {
|
||||
|
||||
void Debugger::HandleFault(const HsaMemoryAccessFault& fault, GpuAgentInt* agent) {
|
||||
std::stringstream err;
|
||||
|
||||
uint64_t fault_page_idx = fault.VirtualAddress >> 0xC;
|
||||
err << "\nMemory access fault by GPU node " << agent->node_id();
|
||||
err << " for address 0x" << std::hex << std::uppercase << fault_page_idx << "xxx (";
|
||||
|
||||
if (fault.Failure.NotPresent == 1) {
|
||||
err << "page not present";
|
||||
} else if (fault.Failure.ReadOnly == 1) {
|
||||
err << "write access to a read-only page";
|
||||
} else if (fault.Failure.NoExecute == 1) {
|
||||
err << "execute access to a non-executable page";
|
||||
} else if (fault.Failure.ECC == 1) {
|
||||
err << "uncorrectable ECC failure";
|
||||
} else {
|
||||
err << "unknown reason";
|
||||
}
|
||||
|
||||
err << ")\n\n";
|
||||
|
||||
if (core::Runtime::runtime_singleton_->flag().debug_fault() != Flag::DEBUG_FAULT_ANALYZE) {
|
||||
if (agent->isa()->GetMajorVersion() >= 9) {
|
||||
err << "For more detail set: HSA_DEBUG_FAULT=\"analyze\"\n\n";
|
||||
}
|
||||
|
||||
std::cerr << err.str();
|
||||
std::abort();
|
||||
}
|
||||
|
||||
WaveStates wave_states = agent->GetWaveStates();
|
||||
|
||||
for (WaveState& wave_state : wave_states) {
|
||||
#define SQ_WAVE_TRAPSTS_XNACK_ERROR(x) (((x) >> 0x1C) & 0x1)
|
||||
|
||||
if (SQ_WAVE_TRAPSTS_XNACK_ERROR(wave_state.regs.trapsts)) {
|
||||
err << "Wavefront found in XNACK error state:\n\n";
|
||||
err << " PC: 0x" << std::setw(0x10) << std::setfill('0') << wave_state.regs.pc << "\n";
|
||||
err << " EXEC: 0x" << std::setw(0x10) << std::setfill('0') << wave_state.regs.exec << "\n";
|
||||
err << " STATUS: 0x" << std::setw(0x8) << std::setfill('0') << wave_state.regs.status << "\n";
|
||||
err << "TRAPSTS: 0x" << std::setw(0x8) << std::setfill('0') << wave_state.regs.trapsts
|
||||
<< "\n";
|
||||
err << " M0: 0x" << std::setw(0x8) << std::setfill('0') << wave_state.regs.m0 << "\n\n";
|
||||
|
||||
uint32_t n_sgpr_cols = 4;
|
||||
uint32_t n_sgpr_rows = wave_state.num_sgprs / n_sgpr_cols;
|
||||
|
||||
for (uint32_t sgpr_row = 0; sgpr_row < n_sgpr_rows; ++sgpr_row) {
|
||||
err << " ";
|
||||
|
||||
for (uint32_t sgpr_col = 0; sgpr_col < n_sgpr_cols; ++sgpr_col) {
|
||||
uint32_t sgpr_idx = (sgpr_row * n_sgpr_cols) + sgpr_col;
|
||||
uint32_t sgpr_val = wave_state.sgprs[sgpr_idx];
|
||||
|
||||
std::stringstream sgpr_str;
|
||||
sgpr_str << "s" << sgpr_idx;
|
||||
|
||||
err << std::setw(6) << std::setfill(' ') << sgpr_str.str();
|
||||
err << ": 0x" << std::setw(8) << std::setfill('0') << sgpr_val;
|
||||
}
|
||||
|
||||
err << "\n";
|
||||
}
|
||||
|
||||
err << "\n";
|
||||
|
||||
uint32_t n_vgpr_cols = 4;
|
||||
uint32_t n_vgpr_rows = wave_state.num_vgprs / n_vgpr_cols;
|
||||
|
||||
for (uint32_t lane_idx = 0; lane_idx < wave_state.num_vgpr_lanes; ++lane_idx) {
|
||||
err << "Lane 0x" << lane_idx << "\n";
|
||||
|
||||
for (uint32_t vgpr_row = 0; vgpr_row < n_vgpr_rows; ++vgpr_row) {
|
||||
err << " ";
|
||||
|
||||
for (uint32_t vgpr_col = 0; vgpr_col < n_vgpr_cols; ++vgpr_col) {
|
||||
uint32_t vgpr_idx = (vgpr_row * n_vgpr_cols) + vgpr_col;
|
||||
uint32_t vgpr_val = wave_state.vgprs[(vgpr_idx * wave_state.num_vgpr_lanes) + lane_idx];
|
||||
|
||||
std::stringstream vgpr_str;
|
||||
vgpr_str << "v" << vgpr_idx;
|
||||
|
||||
err << std::setw(6) << std::setfill(' ') << vgpr_str.str();
|
||||
err << ": 0x" << std::setw(8) << std::setfill('0') << vgpr_val;
|
||||
}
|
||||
|
||||
err << "\n";
|
||||
}
|
||||
}
|
||||
|
||||
err << "\n";
|
||||
|
||||
if (wave_state.lds) {
|
||||
err << "LDS:\n\n";
|
||||
|
||||
uint32_t n_lds_cols = 4;
|
||||
uint32_t n_lds_rows = wave_state.lds_size_dw / n_lds_cols;
|
||||
|
||||
for (uint32_t lds_row = 0; lds_row < n_lds_rows; ++lds_row) {
|
||||
uint32_t lds_addr = lds_row * n_lds_cols * 4;
|
||||
|
||||
err << "0x" << std::setw(4) << std::setfill('0') << lds_addr << ":";
|
||||
|
||||
for (uint32_t lds_col = 0; lds_col < n_lds_cols; ++lds_col) {
|
||||
uint32_t lds_idx = (lds_row * n_lds_cols) + lds_col;
|
||||
uint32_t lds_val = wave_state.lds[lds_idx];
|
||||
|
||||
err << " 0x" << std::setw(8) << std::setfill('0') << lds_val;
|
||||
}
|
||||
|
||||
err << "\n";
|
||||
}
|
||||
|
||||
err << "\n";
|
||||
}
|
||||
|
||||
// Attempt to match the PC to a loaded code object.
|
||||
amd::hsa::loader::LoadedCodeObject* pc_code_obj = nullptr;
|
||||
uint64_t pc_code_obj_offset = 0;
|
||||
|
||||
auto iter_execs = [&](hsa_executable_t exec) {
|
||||
auto iter_code_objs = [&](hsa_loaded_code_object_t code_obj) {
|
||||
auto iter_segments = [&](amd_loaded_segment_t segment) {
|
||||
auto segment_int = amd::hsa::loader::LoadedSegment::Object(segment);
|
||||
|
||||
uint64_t load_base, load_size;
|
||||
segment_int->GetInfo(AMD_LOADED_SEGMENT_INFO_LOAD_BASE_ADDRESS, &load_base);
|
||||
segment_int->GetInfo(AMD_LOADED_SEGMENT_INFO_SIZE, &load_size);
|
||||
|
||||
if ((wave_state.regs.pc >= load_base) &&
|
||||
(wave_state.regs.pc < (load_base + load_size))) {
|
||||
pc_code_obj = amd::hsa::loader::LoadedCodeObject::Object(code_obj);
|
||||
pc_code_obj_offset = wave_state.regs.pc - load_base;
|
||||
}
|
||||
|
||||
return HSA_STATUS_SUCCESS;
|
||||
};
|
||||
|
||||
amd::hsa::loader::LoadedCodeObject::Object(code_obj)->IterateLoadedSegments(
|
||||
[](amd_loaded_segment_t segment, void* data) {
|
||||
return (*reinterpret_cast<decltype(iter_segments)*>(data))(segment);
|
||||
},
|
||||
&iter_segments);
|
||||
|
||||
return HSA_STATUS_SUCCESS;
|
||||
};
|
||||
|
||||
amd::hsa::loader::Executable::Object(exec)->IterateLoadedCodeObjects(
|
||||
[](hsa_loaded_code_object_t code_obj, void* data) {
|
||||
return (*reinterpret_cast<decltype(iter_code_objs)*>(data))(code_obj);
|
||||
},
|
||||
&iter_code_objs);
|
||||
|
||||
return HSA_STATUS_SUCCESS;
|
||||
};
|
||||
|
||||
core::Runtime::runtime_singleton_->loader()->IterateExecutables(
|
||||
[](hsa_executable_t exec, void* data) {
|
||||
return (*reinterpret_cast<decltype(iter_execs)*>(data))(exec);
|
||||
},
|
||||
&iter_execs);
|
||||
|
||||
if (pc_code_obj) {
|
||||
// Write the code object to a temporary file.
|
||||
uint64_t elf_addr;
|
||||
size_t elf_size;
|
||||
pc_code_obj->GetInfo(AMD_LOADED_CODE_OBJECT_INFO_ELF_IMAGE, &elf_addr);
|
||||
pc_code_obj->GetInfo(AMD_LOADED_CODE_OBJECT_INFO_ELF_IMAGE_SIZE, &elf_size);
|
||||
|
||||
char code_obj_path[] = "/tmp/hsartXXXXXX";
|
||||
int code_obj_fd = ::mkstemp(code_obj_path);
|
||||
::write(code_obj_fd, (const void*)uintptr_t(elf_addr), elf_size);
|
||||
::close(code_obj_fd);
|
||||
|
||||
// Invoke binutils objdump on the code object.
|
||||
int pipe_fd[2];
|
||||
::pipe(pipe_fd);
|
||||
|
||||
pid_t pid = ::fork();
|
||||
|
||||
if (pid == 0) {
|
||||
::dup2(pipe_fd[1], STDOUT_FILENO);
|
||||
::dup2(pipe_fd[1], STDERR_FILENO);
|
||||
::close(pipe_fd[0]);
|
||||
::close(pipe_fd[1]);
|
||||
|
||||
// Disassemble X bytes before/after the PC.
|
||||
uint32_t disasm_context = 0x20;
|
||||
|
||||
std::stringstream arg_start_addr, arg_stop_addr;
|
||||
arg_start_addr << "--start-addr=0x" << std::hex << (pc_code_obj_offset - disasm_context);
|
||||
arg_stop_addr << "--stop-addr=0x" << std::hex << (pc_code_obj_offset + disasm_context);
|
||||
|
||||
std::exit(execlp("objdump", "-d", "-S", "-l", arg_start_addr.str().c_str(),
|
||||
arg_stop_addr.str().c_str(), code_obj_path, nullptr));
|
||||
}
|
||||
|
||||
// Collect the output of objdump.
|
||||
::close(pipe_fd[1]);
|
||||
|
||||
std::vector<char> objdump_out_buf;
|
||||
std::vector<char> buf(0x1000);
|
||||
ssize_t n_read_b;
|
||||
|
||||
while ((n_read_b = read(pipe_fd[0], buf.data(), buf.size())) > 0) {
|
||||
objdump_out_buf.insert(objdump_out_buf.end(), &buf[0], &buf[n_read_b]);
|
||||
}
|
||||
|
||||
::close(pipe_fd[0]);
|
||||
|
||||
int child_status = 0;
|
||||
int ret = ::waitpid(pid, &child_status, 0);
|
||||
|
||||
if (ret != -1 && child_status == 0) {
|
||||
// Attempt to trim the leading output from objdump.
|
||||
std::string objdump_out(objdump_out_buf.begin(), objdump_out_buf.end());
|
||||
size_t trim_start = objdump_out.find(":\n\n") + 3;
|
||||
|
||||
if (trim_start != objdump_out.npos) {
|
||||
objdump_out = objdump_out.substr(trim_start);
|
||||
}
|
||||
|
||||
// Attempt to add a PC indicator inside the disassembly text.
|
||||
std::stringstream pc_offset_find;
|
||||
pc_offset_find << std::hex << pc_code_obj_offset << ":\t";
|
||||
size_t replace_idx = objdump_out.find(pc_offset_find.str());
|
||||
|
||||
if (replace_idx != objdump_out.npos) {
|
||||
std::stringstream pc_offset_replace;
|
||||
pc_offset_replace << std::hex << pc_code_obj_offset << ": >>>>>\t";
|
||||
objdump_out.replace(replace_idx, pc_offset_find.str().size(), pc_offset_replace.str());
|
||||
err << objdump_out << "\n";
|
||||
} else {
|
||||
err << objdump_out;
|
||||
err << "\nPC offset: " << std::hex << pc_code_obj_offset << "\n\n";
|
||||
}
|
||||
} else {
|
||||
err << "(Disassembly unavailable - is amdgcn-capable objdump in PATH?)\n\n";
|
||||
}
|
||||
|
||||
::unlink(code_obj_path);
|
||||
} else {
|
||||
err << "(Cannot match PC to a loaded code object)\n\n";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::cerr << err.str();
|
||||
std::abort();
|
||||
}
|
||||
}
|
||||
@@ -73,7 +73,8 @@ GpuAgent::GpuAgent(HSAuint32 node, const HsaNodeProperties& node_props)
|
||||
properties_(node_props),
|
||||
current_coherency_type_(HSA_AMD_COHERENCY_TYPE_COHERENT),
|
||||
blits_(),
|
||||
queues_(),
|
||||
queue_util_(nullptr),
|
||||
queue_blit_(nullptr),
|
||||
local_region_(NULL),
|
||||
is_kv_device_(false),
|
||||
trap_code_buf_(NULL),
|
||||
@@ -137,9 +138,7 @@ GpuAgent::~GpuAgent() {
|
||||
}
|
||||
}
|
||||
|
||||
for (int i = 0; i < QueueCount; ++i) {
|
||||
delete queues_[i];
|
||||
}
|
||||
queues_.clear();
|
||||
|
||||
if (end_ts_base_addr_ != NULL) {
|
||||
core::Runtime::runtime_singleton_->FreeMemory(end_ts_base_addr_);
|
||||
@@ -581,16 +580,16 @@ void GpuAgent::InitDma() {
|
||||
// Fall back to blit kernel if SDMA is unavailable.
|
||||
if (blits_[BlitHostToDev] == NULL) {
|
||||
// Create a dedicated compute queue for host-to-device blits.
|
||||
queues_[QueueBlitOnly] = CreateInterceptibleQueue();
|
||||
assert(queues_[QueueBlitOnly] != NULL && "Queue creation failed");
|
||||
queue_blit_ = CreateInterceptibleQueue();
|
||||
assert(queue_blit_ != NULL && "Queue creation failed");
|
||||
|
||||
blits_[BlitHostToDev] = CreateBlitKernel(queues_[QueueBlitOnly]);
|
||||
blits_[BlitHostToDev] = CreateBlitKernel(queue_blit_);
|
||||
assert(blits_[BlitHostToDev] != NULL && "Blit creation failed");
|
||||
}
|
||||
|
||||
if (blits_[BlitDevToHost] == NULL) {
|
||||
// Share utility queue with device-to-host blits.
|
||||
blits_[BlitDevToHost] = CreateBlitKernel(queues_[QueueUtility]);
|
||||
blits_[BlitDevToHost] = CreateBlitKernel(queue_util_);
|
||||
assert(blits_[BlitDevToHost] != NULL && "Blit creation failed");
|
||||
}
|
||||
|
||||
@@ -605,14 +604,14 @@ hsa_status_t GpuAgent::PostToolsInit() {
|
||||
BindTrapHandler();
|
||||
|
||||
// Defer utility queue creation to allow tools to intercept.
|
||||
queues_[QueueUtility] = CreateInterceptibleQueue();
|
||||
queue_util_ = CreateInterceptibleQueue();
|
||||
|
||||
if (queues_[QueueUtility] == NULL) {
|
||||
if (queue_util_ == NULL) {
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
}
|
||||
|
||||
// Share utility queue with device-to-device blits.
|
||||
blits_[BlitDevToDev] = CreateBlitKernel(queues_[QueueUtility]);
|
||||
blits_[BlitDevToDev] = CreateBlitKernel(queue_util_);
|
||||
|
||||
if (blits_[BlitDevToDev] == NULL) {
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
@@ -926,6 +925,7 @@ hsa_status_t GpuAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type,
|
||||
event_callback, data, is_kv_device_);
|
||||
if (hw_queue && hw_queue->IsValid()) {
|
||||
// return queue
|
||||
queues_.emplace_back(hw_queue);
|
||||
*queue = hw_queue;
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
@@ -935,6 +935,28 @@ hsa_status_t GpuAgent::QueueCreate(size_t size, hsa_queue_type32_t queue_type,
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
}
|
||||
|
||||
WaveStates GpuAgent::GetWaveStates() {
|
||||
WaveStates wave_states;
|
||||
|
||||
for (auto& queue : queues_) {
|
||||
WaveStates queue_wave_states = queue->GetWaveStates();
|
||||
wave_states.insert(wave_states.end(), queue_wave_states.begin(), queue_wave_states.end());
|
||||
}
|
||||
|
||||
return wave_states;
|
||||
}
|
||||
|
||||
hsa_status_t GpuAgent::QueueDestroy(core::Queue* queue) {
|
||||
auto it = std::find_if(queues_.begin(), queues_.end(), [&](std::unique_ptr<AqlQueue>& queue_ptr) {
|
||||
return static_cast<core::Queue*>(queue_ptr.get()) == queue;
|
||||
});
|
||||
|
||||
assert(it != queues_.end() && "attempt to destroy an untracked queue");
|
||||
queues_.erase(it);
|
||||
|
||||
return HSA_STATUS_SUCCESS;
|
||||
}
|
||||
|
||||
void GpuAgent::AcquireQueueScratch(ScratchInfo& scratch) {
|
||||
bool need_queue_scratch_base = (isa_->GetMajorVersion() > 8);
|
||||
|
||||
@@ -1219,7 +1241,7 @@ void GpuAgent::InvalidateCodeCaches() {
|
||||
cache_inv[6] = 0;
|
||||
|
||||
// Submit the command to the utility queue and wait for it to complete.
|
||||
queues_[QueueUtility]->ExecutePM4(cache_inv, sizeof(cache_inv));
|
||||
queue_util_->ExecutePM4(cache_inv, sizeof(cache_inv));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
@@ -46,12 +46,9 @@
|
||||
#include "core/util/utils.h"
|
||||
|
||||
namespace core {
|
||||
HostQueue::HostQueue(hsa_region_t region, uint32_t ring_size,
|
||||
hsa_queue_type32_t type, uint32_t features,
|
||||
hsa_signal_t doorbell_signal)
|
||||
: Queue(),
|
||||
size_(ring_size),
|
||||
active_(false) {
|
||||
HostQueue::HostQueue(Agent& agent, hsa_region_t region, uint32_t ring_size, hsa_queue_type32_t type,
|
||||
uint32_t features, hsa_signal_t doorbell_signal)
|
||||
: Queue(agent), size_(ring_size), active_(false) {
|
||||
if (!Shared::IsSharedObjectAllocationValid()) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -608,17 +608,14 @@ hsa_status_t hsa_soft_queue_create(hsa_region_t region, uint32_t size,
|
||||
const core::Signal* signal = core::Signal::Convert(doorbell_signal);
|
||||
IS_VALID(signal);
|
||||
|
||||
core::HostQueue* host_queue =
|
||||
new core::HostQueue(region, size, type, features, doorbell_signal);
|
||||
core::Agent* agent = core::Runtime::runtime_singleton_->cpu_agents().front();
|
||||
core::Queue* host_queue = nullptr;
|
||||
hsa_status_t status =
|
||||
agent->HostQueueCreate(region, size, type, features, doorbell_signal, &host_queue);
|
||||
|
||||
if (!host_queue->active()) {
|
||||
delete host_queue;
|
||||
return HSA_STATUS_ERROR_OUT_OF_RESOURCES;
|
||||
}
|
||||
*queue = (host_queue ? core::Queue::Convert(host_queue) : nullptr);
|
||||
|
||||
*queue = core::Queue::Convert(host_queue);
|
||||
|
||||
return HSA_STATUS_SUCCESS;
|
||||
return status;
|
||||
}
|
||||
|
||||
/// @brief Api to destroy a user mode queue
|
||||
@@ -631,8 +628,7 @@ hsa_status_t hsa_queue_destroy(hsa_queue_t* queue) {
|
||||
IS_BAD_PTR(queue);
|
||||
core::Queue* cmd_queue = core::Queue::Convert(queue);
|
||||
IS_VALID(cmd_queue);
|
||||
delete cmd_queue;
|
||||
return HSA_STATUS_SUCCESS;
|
||||
return cmd_queue->agent().QueueDestroy(cmd_queue);
|
||||
}
|
||||
|
||||
/// @brief Api to inactivate a user mode queue
|
||||
|
||||
@@ -53,6 +53,7 @@
|
||||
|
||||
#include "core/inc/hsa_ext_interface.h"
|
||||
#include "core/inc/amd_cpu_agent.h"
|
||||
#include "core/inc/amd_debugger.h"
|
||||
#include "core/inc/amd_gpu_agent.h"
|
||||
#include "core/inc/amd_memory_region.h"
|
||||
#include "core/inc/amd_topology.h"
|
||||
@@ -916,55 +917,24 @@ void Runtime::BindVmFaultHandler() {
|
||||
return;
|
||||
}
|
||||
|
||||
SetAsyncSignalHandler(core::Signal::Convert(vm_fault_signal_),
|
||||
HSA_SIGNAL_CONDITION_NE, 0, VMFaultHandler,
|
||||
reinterpret_cast<void*>(vm_fault_signal_));
|
||||
SetAsyncSignalHandler(core::Signal::Convert(vm_fault_signal_), HSA_SIGNAL_CONDITION_NE, 0,
|
||||
VMFaultHandler, this);
|
||||
}
|
||||
}
|
||||
|
||||
bool Runtime::VMFaultHandler(hsa_signal_value_t val, void* arg) {
|
||||
core::InterruptSignal* vm_fault_signal =
|
||||
reinterpret_cast<core::InterruptSignal*>(arg);
|
||||
Runtime* runtime = reinterpret_cast<Runtime*>(arg);
|
||||
assert(runtime->vm_fault_signal_ != NULL);
|
||||
|
||||
assert(vm_fault_signal != NULL);
|
||||
HsaEvent* vm_fault_event = runtime->vm_fault_signal_->EopEvent();
|
||||
const HsaMemoryAccessFault& fault = vm_fault_event->EventData.EventData.MemoryAccessFault;
|
||||
|
||||
if (vm_fault_signal == NULL) {
|
||||
return false;
|
||||
}
|
||||
auto agent_it = std::find_if(runtime->gpu_agents_.begin(), runtime->gpu_agents_.end(),
|
||||
[&](Agent* agent) { return agent->node_id() == fault.NodeId; });
|
||||
assert(agent_it != runtime->gpu_agents_.end());
|
||||
|
||||
if (runtime_singleton_->flag().enable_vm_fault_message()) {
|
||||
HsaEvent* vm_fault_event = vm_fault_signal->EopEvent();
|
||||
amd::Debugger::HandleFault(fault, static_cast<amd::GpuAgentInt*>(*agent_it));
|
||||
|
||||
const HsaMemoryAccessFault& fault =
|
||||
vm_fault_event->EventData.EventData.MemoryAccessFault;
|
||||
|
||||
std::string reason = "";
|
||||
if (fault.Failure.NotPresent == 1) {
|
||||
reason += "Page not present or supervisor privilege";
|
||||
} else if (fault.Failure.ReadOnly == 1) {
|
||||
reason += "Write access to a read-only page";
|
||||
} else if (fault.Failure.NoExecute == 1) {
|
||||
reason += "Execute access to a page marked NX";
|
||||
} else if (fault.Failure.GpuAccess == 1) {
|
||||
reason += "Host access only";
|
||||
} else if (fault.Failure.ECC == 1) {
|
||||
reason += "ECC failure (if supported by HW)";
|
||||
} else {
|
||||
reason += "Unknown";
|
||||
}
|
||||
|
||||
fprintf(stderr,
|
||||
"Memory access fault by GPU node-%u on address %p%s. Reason: %s.\n",
|
||||
fault.NodeId, reinterpret_cast<const void*>(fault.VirtualAddress),
|
||||
(fault.Failure.Imprecise == 1) ? "(may not be exact address)" : "",
|
||||
reason.c_str());
|
||||
} else {
|
||||
assert(false && "GPU memory access fault.");
|
||||
}
|
||||
|
||||
std::abort();
|
||||
|
||||
// No need to keep the signal because we are done.
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user