rocr: GPU core file location support (#1732)

* rocr: WIP Support dump of GPU core file

* WIP new core dump tests compile

* WIP: anony namespaces, test updates, progress

Added disabled Fault test. Other non-disabled coredump tests don't work.

* WIP: address code review feedback

* WIP: gpu core dump rocrtst works; combined

* WIP: remove rocrtst changes for this commit
This commit is contained in:
cfreeamd
2025-11-20 20:50:51 -06:00
کامیت شده توسط GitHub
والد adf6a5ec3b
کامیت 24c2a84e3f
4فایلهای تغییر یافته به همراه407 افزوده شده و 61 حذف شده
@@ -1323,8 +1323,8 @@ bool AqlQueue::ExceptionHandler(hsa_signal_value_t error_code, void* arg) {
return exceptionHandlerDone();
}
// Fallback if KFD does not support GPU core dump. In this case, there core dump is
// generated by hsa-runtime.
// Fallback if KFD does not support GPU core dump. In this case, the core
// dump is generated by hsa-runtime.
if (!core::Runtime::runtime_singleton_->KfdVersion().supports_core_dump &&
queue->agent_->supported_isas()[0]->GetMajorVersion() != 11) {
@@ -2236,7 +2236,7 @@ bool Runtime::VMFaultHandler(hsa_signal_value_t val, void* arg) {
PrintMemoryMapNear(reinterpret_cast<void*>(fault.VirtualAddress));
#endif
}
// Fallback if KFD does not support GPU core dump. In this case, there core dump is
// Fallback if KFD does not support GPU core dump. In this case, the core dump is
// generated by hsa-runtime.
if (faulty_agent &&
faulty_agent->supported_isas()[0]->GetMajorVersion() != 11 &&
@@ -298,6 +298,14 @@ class Flag {
var = os::GetEnvVar("HSA_CO_DMACOPY_SIZE");
co_dmacopy_size_ = var.empty() ? 1024*1024 : atoi(var.c_str());
var = os::GetEnvVar("HSA_COREDUMP_SHOW_PROGRESS");
enable_core_dump_progress_ = (var == "1");
var = os::GetEnvVar("HSA_DISABLE_COREDUMP_ON_EXCEPTION");
core_dump_disable_ = (var == "1");
core_dump_pattern_ = os::GetEnvVar("HSA_COREDUMP_PATTERN");
}
void parse_masks(uint32_t maxGpu, uint32_t maxCU) {
@@ -430,6 +438,17 @@ class Flag {
bool enable_dxg_detection() const { return enable_dxg_detection_; }
[[nodiscard]]
bool core_dump_disable() const { return core_dump_disable_; }
[[nodiscard]]
bool enable_core_dump_progress() const {
return enable_core_dump_progress_; }
[[nodiscard]]
const std::string& core_dump_pattern() const {
return core_dump_pattern_; }
void set_sdma(bool peer_sdma, bool sdma_gang) {
enable_peer_sdma_ = peer_sdma ? SDMA_ENABLE : SDMA_DISABLE;
enable_sdma_gang_ = sdma_gang ? SDMA_ENABLE : SDMA_DISABLE;
@@ -522,6 +541,10 @@ class Flag {
size_t co_dmacopy_size_;
bool core_dump_disable_ = false;
bool enable_core_dump_progress_ = false;
std::string core_dump_pattern_;
// Map GPU index post RVD to its default cu mask.
std::map<uint32_t, std::vector<uint32_t>> cu_mask_;