SWDEV-396239: Automatic ISA dump from ATT

Change-Id: Ia66346c0048b779487157961ead08d7567c9153a
This commit is contained in:
Giovanni LB
2023-07-13 03:46:03 -03:00
کامیت شده توسط Giovanni Baraldi
والد c94d4c3e81
کامیت a5555ac45f
25فایلهای تغییر یافته به همراه1723 افزوده شده و 182 حذف شده
+3 -3
مشاهده پرونده
@@ -26,7 +26,7 @@ set(ENV{ROCPROFV2_ATT_LIB_PATH} $ROCPROFV2_ATT)
# Building att plugin library
file(GLOB ROCPROFILER_UTIL_SRC_FILES ${PROJECT_SOURCE_DIR}/src/utils/helper.cpp)
file(GLOB FILE_SOURCES att.cpp)
file(GLOB FILE_SOURCES att.cpp disassembly.cpp code_printing.cpp)
add_library(att_plugin SHARED ${FILE_SOURCES} ${ROCPROFILER_UTIL_SRC_FILES})
set_target_properties(
@@ -44,8 +44,7 @@ target_include_directories(att_plugin PRIVATE ${PROJECT_SOURCE_DIR}
target_link_options(
att_plugin PRIVATE -Wl,--version-script=${CMAKE_CURRENT_SOURCE_DIR}/../exportmap
-Wl,--no-undefined)
target_link_libraries(att_plugin PRIVATE rocprofiler-v2 hsa-runtime64::hsa-runtime64
stdc++fs)
target_link_libraries(att_plugin PRIVATE rocprofiler-v2 hsa-runtime64::hsa-runtime64 stdc++fs dw elf amd_comgr)
install(TARGETS att_plugin LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}/${PROJECT_NAME}
COMPONENT asan)
@@ -56,6 +55,7 @@ configure_file(att.py att/att.py COPYONLY)
configure_file(trace_view.py att/trace_view.py COPYONLY)
configure_file(stitch.py att/stitch.py COPYONLY)
configure_file(drawing.py att/drawing.py COPYONLY)
configure_file(att_to_csv.py att/att_to_csv.py COPYONLY)
configure_file(ui/index.html att/ui/index.html COPYONLY)
configure_file(ui/logo.svg att/ui/logo.svg COPYONLY)
configure_file(ui/styles.css att/ui/styles.css COPYONLY)
+84 -31
مشاهده پرونده
@@ -44,8 +44,11 @@
#include "rocprofiler.h"
#include "rocprofiler_plugin.h"
#include "../utils.h"
#include "code_printing.hpp"
#define ATT_FILENAME_MAXBYTES 90
#define TEST_INVALID_KERNEL size_t(-1)
namespace {
@@ -74,24 +77,33 @@ class att_plugin_t {
bool IsValid() const { return is_valid_; }
void FlushATTRecord(const rocprofiler_record_att_tracer_t* att_tracer_record,
rocprofiler_session_id_t session_id, rocprofiler_buffer_id_t buffer_id) {
int FlushATTRecord(const rocprofiler_record_att_tracer_t* att_tracer_record,
rocprofiler_session_id_t session_id, rocprofiler_buffer_id_t buffer_id) {
std::lock_guard<std::mutex> lock(writing_lock);
if (!att_tracer_record) {
printf("No att data buffer received\n");
return;
if (!att_tracer_record) return ROCPROFILER_STATUS_ERROR;
std::string kernel_name_mangled{};
// Found problem with rocprofiler API for invalid kernel_ids;
if (att_tracer_record->kernel_id.handle != TEST_INVALID_KERNEL) {
size_t name_length;
CHECK_ROCPROFILER(rocprofiler_query_kernel_info_size(
ROCPROFILER_KERNEL_NAME, att_tracer_record->kernel_id, &name_length));
const char* kernel_name_c = nullptr;
CHECK_ROCPROFILER(rocprofiler_query_kernel_info(
ROCPROFILER_KERNEL_NAME, att_tracer_record->kernel_id, &kernel_name_c));
assert(kernel_name_c && "Rocprofv2 returned an invalid kernel name");
kernel_name_mangled = std::string(kernel_name_c);
free(const_cast<char*>(kernel_name_c));
} else { // Temporary. Adding a valid string.
kernel_name_mangled = "test_kernel";
}
size_t name_length;
CHECK_ROCPROFILER(rocprofiler_query_kernel_info_size(
ROCPROFILER_KERNEL_NAME, att_tracer_record->kernel_id, &name_length));
const char* kernel_name_c = static_cast<const char*>(malloc(name_length * sizeof(char)));
CHECK_ROCPROFILER(rocprofiler_query_kernel_info(ROCPROFILER_KERNEL_NAME,
att_tracer_record->kernel_id, &kernel_name_c));
std::string name_demangled =
rocprofiler::truncate_name(rocprofiler::cxx_demangle(kernel_name_c));
rocprofiler::truncate_name(rocprofiler::cxx_demangle(kernel_name_mangled));
if (name_demangled.size() > ATT_FILENAME_MAXBYTES) // Limit filename size
name_demangled = name_demangled.substr(0, ATT_FILENAME_MAXBYTES);
@@ -110,11 +122,11 @@ class att_plugin_t {
file_iteration += 1;
outfilepath += std::to_string(file_iteration);
auto dispatch_id = att_tracer_record->header.id.handle;
auto writer_id = att_tracer_record->writer_id;
std::string fname = outfilepath + "_kernel.txt";
std::ofstream(fname.c_str()) << name_demangled << " dispatch[" << dispatch_id << "] GPU["
<< att_tracer_record->gpu_id.handle << "]: " << kernel_name_c
std::ofstream(fname.c_str()) << name_demangled << " dispatch[" << writer_id << "] GPU["
<< att_tracer_record->gpu_id.handle << "]: " << kernel_name_mangled
<< '\n';
// iterate over each shader engine att trace
@@ -125,29 +137,69 @@ class att_plugin_t {
continue;
printf("--------------collecting data for shader_engine %d---------------\n", i);
rocprofiler_record_se_att_data_t* se_att_trace = &att_tracer_record->shader_engine_data[i];
const char* data_buffer_ptr = reinterpret_cast<char*>(se_att_trace->buffer_ptr);
char* data_buffer_ptr = reinterpret_cast<char*>(se_att_trace->buffer_ptr);
// dump data in binary format
std::ofstream out(outfilepath + "_se" + std::to_string(i) + ".att", std::ios::binary);
if (out.is_open())
out.write((char*)data_buffer_ptr, se_att_trace->buffer_size);
else
if (!out.is_open()) {
std::cerr << "ATT Failed to open file: " << outfilepath << "_se" << i << ".att\n";
return ROCPROFILER_STATUS_ERROR;
}
out.write(data_buffer_ptr, se_att_trace->buffer_size);
}
std::ofstream isafile(outfilepath + "_isa.s");
if (!isafile.is_open()) {
std::cerr << "Could not open ISA file: " << outfilepath << "_isa.s" << std::endl;
return ROCPROFILER_STATUS_ERROR;
}
uint64_t kernel_begin_addr = att_tracer_record->intercept_list.userdata;
isafile << "<Kernel> " << kernel_name_mangled << '\n';
for (size_t i = 0; i < att_tracer_record->intercept_list.count; i++) {
const rocprofiler_intercepted_codeobj_t& symbol =
att_tracer_record->intercept_list.symbols[i];
std::unique_ptr<CodeObjectBinary> binary;
std::unique_ptr<code_object_decoder_t> decoder;
if (symbol.data && symbol.size) {
decoder = std::make_unique<code_object_decoder_t>(symbol.data, symbol.size);
} else if (std::string(symbol.filepath).find("file://") != std::string::npos) {
binary = std::make_unique<CodeObjectBinary>(symbol.filepath);
decoder =
std::make_unique<code_object_decoder_t>(binary->buffer.data(), binary->buffer.size());
} else {
continue;
}
for (auto& instance : decoder->instructions) {
uint64_t addr = instance.address + symbol.base_address;
if (kernel_begin_addr == addr)
isafile << "; Begin <Kernel> " << kernel_name_mangled << '\n';
else if (decoder->m_symbol_map.find(instance.address) != decoder->m_symbol_map.end())
isafile << "; Begin " << decoder->m_symbol_map[instance.address].first << '\n';
if (instance.cpp_reference) isafile << "; " << instance.cpp_reference << '\n';
isafile << instance.instruction << " // " << std::hex << addr << '\n';
}
}
return ROCPROFILER_STATUS_SUCCESS;
}
int WriteBufferRecords(const rocprofiler_record_header_t* begin,
const rocprofiler_record_header_t* end,
rocprofiler_session_id_t session_id, rocprofiler_buffer_id_t buffer_id) {
while (begin < end) {
if (!begin) return 0;
if (!begin) return ROCPROFILER_STATUS_ERROR;
switch (begin->kind) {
case ROCPROFILER_PROFILER_RECORD:
case ROCPROFILER_TRACER_RECORD:
case ROCPROFILER_PC_SAMPLING_RECORD:
case ROCPROFILER_SPM_RECORD:
case ROCPROFILER_COUNTERS_SAMPLER_RECORD:
printf("Invalid record Kind: %d", begin->kind);
rocprofiler::warning("Invalid record Kind: %d\n", begin->kind);
break;
case ROCPROFILER_ATT_TRACER_RECORD: {
@@ -158,10 +210,11 @@ class att_plugin_t {
break;
}
}
rocprofiler_next_record(begin, &begin, session_id, buffer_id);
int status = rocprofiler_next_record(begin, &begin, session_id, buffer_id);
if (status != ROCPROFILER_STATUS_SUCCESS) return status;
}
return 0;
return ROCPROFILER_STATUS_SUCCESS;
}
private:
@@ -176,17 +229,17 @@ ROCPROFILER_EXPORT int rocprofiler_plugin_initialize(uint32_t rocprofiler_major_
void* data) {
if (rocprofiler_major_version != ROCPROFILER_VERSION_MAJOR ||
rocprofiler_minor_version < ROCPROFILER_VERSION_MINOR)
return -1;
return ROCPROFILER_STATUS_ERROR;
if (att_plugin != nullptr) return -1;
if (att_plugin != nullptr) return ROCPROFILER_STATUS_ERROR;
att_plugin = new att_plugin_t();
if (att_plugin->IsValid()) return 0;
if (att_plugin->IsValid()) return ROCPROFILER_STATUS_SUCCESS;
// The plugin failed to initialied, destroy it and return an error.
delete att_plugin;
att_plugin = nullptr;
return -1;
return ROCPROFILER_STATUS_ERROR;
}
ROCPROFILER_EXPORT void rocprofiler_plugin_finalize() {
@@ -198,12 +251,12 @@ ROCPROFILER_EXPORT void rocprofiler_plugin_finalize() {
ROCPROFILER_EXPORT int rocprofiler_plugin_write_buffer_records(
const rocprofiler_record_header_t* begin, const rocprofiler_record_header_t* end,
rocprofiler_session_id_t session_id, rocprofiler_buffer_id_t buffer_id) {
if (!att_plugin || !att_plugin->IsValid()) return -1;
if (!att_plugin || !att_plugin->IsValid()) return ROCPROFILER_STATUS_ERROR;
return att_plugin->WriteBufferRecords(begin, end, session_id, buffer_id);
}
ROCPROFILER_EXPORT int rocprofiler_plugin_write_record(rocprofiler_record_tracer_t record) {
if (!att_plugin || !att_plugin->IsValid()) return -1;
if (record.header.id.handle == 0) return 0;
return 0;
if (!att_plugin || !att_plugin->IsValid()) return ROCPROFILER_STATUS_ERROR;
if (record.header.id.handle == 0) return ROCPROFILER_STATUS_SUCCESS;
return ROCPROFILER_STATUS_SUCCESS;
}
+37 -22
مشاهده پرونده
@@ -40,25 +40,26 @@ class PerfEvent(ctypes.Structure):
class CodeWrapped(ctypes.Structure):
""" Matches CodeWrapped on the python side """
_fields_ = [('line', ctypes.c_char_p),
('loc', ctypes.c_char_p),
('value', ctypes.c_int),
('to_line', ctypes.c_int),
('index', ctypes.c_int),
('line_num', ctypes.c_int)]
('loc', ctypes.c_char_p),
('to_line', ctypes.c_int),
('value', ctypes.c_int),
('index', ctypes.c_int),
('line_num', ctypes.c_int),
('addr', ctypes.c_int64)]
class KvPair(ctypes.Structure):
""" Matches pair<int, int> = (key, value) on the python side """
_fields_ = [('key', ctypes.c_int),
('value', ctypes.c_int)]
('value', ctypes.c_int)]
class ReturnAssemblyInfo(ctypes.Structure):
""" Matches ReturnAssemblyInfo on the python side """
_fields_ = [('code', POINTER(CodeWrapped)),
('jumps', POINTER(KvPair)),
('code_len', ctypes.c_int),
('jumps_len', ctypes.c_int)]
('jumps', POINTER(KvPair)),
('code_len', ctypes.c_int),
('jumps_len', ctypes.c_int)]
class Wave(ctypes.Structure):
@@ -155,8 +156,9 @@ def parse_binary(filename, kernel=None):
to_line = int(code_entry.to_line) if (code_entry.to_line >= 0) else None
loc = loc if len(loc) > 0 else None
code.append([line, int(code_entry.value), to_line, loc,
int(code_entry.index), int(code_entry.line_num), 0, 0]) # hitcount + cycles
# asm, inst_type, addr, loc, index, line_num, hitcount, cycles
code.append([line, int(code_entry.value), to_line, loc, int(code_entry.index),
int(code_entry.line_num), int(code_entry.addr), 0, 0])
jumps = {}
for k in range(info.jumps_len):
@@ -187,9 +189,9 @@ def getWaves_binary(name, shader_engine_data_dict, target_cu, depth):
shader_engine_data_dict[name] = (waves_python, events, occupancy, flags)
def getWaves_stitch(SIMD, code, jumps, flags, latency_map, hitcount_map):
def getWaves_stitch(SIMD, code, jumps, flags, latency_map, hitcount_map, bIsAuto):
for pwave in SIMD:
pwave.instructions = stitch(pwave.instructions, code, jumps, flags)
pwave.instructions = stitch(pwave.instructions, code, jumps, flags, bIsAuto)
for inst in pwave.instructions[0]:
hitcount_map[inst[-1]] += 1
@@ -363,7 +365,10 @@ if __name__ == "__main__":
network: Open att server over the network.''', type=str, default="off")
args = parser.parse_args()
if args.mode.lower() == 'file':
CSV_MODE = False
if args.mode.lower() == 'csv':
CSV_MODE = True
elif args.mode.lower() == 'file':
args.dumpfiles = True
elif args.mode.lower() == 'network':
args.dumpfiles = False
@@ -387,12 +392,6 @@ if __name__ == "__main__":
if args.target_cu is None:
args.target_cu = 1
# Assembly parsing
path = Path(args.assembly_code)
if not path.is_file():
print("Invalid assembly_code('{0}')!".format(args.assembly_code))
sys.exit(1)
att_kernel = glob.glob(args.att_kernel)
if len(att_kernel) == 0:
@@ -418,6 +417,16 @@ if __name__ == "__main__":
else:
args.att_kernel = att_kernel[0]
# Assembly parsing
bIsAuto = False
if args.assembly_code.lower().strip() == 'auto':
args.assembly_code = args.att_kernel.split('_kernel.txt')[0]+'_isa.s'
bIsAuto = True
path = Path(args.assembly_code)
if not path.is_file():
print("Invalid assembly_code('{0}')!".format(args.assembly_code))
sys.exit(1)
# Trace Parsing
if args.trace_file is None:
filenames = glob.glob(args.att_kernel.split('_kernel.txt')[0]+'_*.att')
@@ -431,7 +440,7 @@ if __name__ == "__main__":
code = jumps = None
if mpi_root:
print('Att kernel:', args.att_kernel)
code, jumps = parse_binary(args.assembly_code, args.att_kernel)
code, jumps = parse_binary(args.assembly_code, None if bIsAuto else args.att_kernel)
DBFILES = []
TIMELINES = [np.zeros(int(1E4),dtype=np.int16) for k in range(5)]
@@ -460,7 +469,7 @@ if __name__ == "__main__":
if np.sum([0]+[len(s.instructions) for s in SIMD]) == 0:
print("No waves from", name)
continue
getWaves_stitch(SIMD, code, jumps, gfxv, latency_map, hitcount_map)
getWaves_stitch(SIMD, code, jumps, gfxv, latency_map, hitcount_map, bIsAuto)
analysed_filenames.append(name)
EVENTS.append(perfevents)
@@ -524,6 +533,12 @@ if __name__ == "__main__":
code[k][-2] = int(hitcount_map[k])
code[k][-1] = int(latency_map[k])
if CSV_MODE:
if mpi_root:
from att_to_csv import dump_csv
dump_csv(code)
quit()
gc.collect()
print("Min time:", min_event_time)
+18
مشاهده پرونده
@@ -0,0 +1,18 @@
#!/usr/bin/env python3
import numpy as np
import csv
import os
def dump_csv(code):
outpath = os.getenv("OUT_FILE_NAME")
if outpath is None:
outpath = "att_output.csv"
if ".csv" not in outpath:
outpath += ".csv"
with open(outpath, 'w') as f:
writer = csv.writer(f)
writer.writerow(['Line', 'Instruction', 'Hitcount', 'Cycles', 'Addr', 'C++ Reference'])
[writer.writerow([m[5], m[0], m[7], m[8], hex(m[6]), m[3]]) for m in code]
#[writer.writerow(m) for m in code]
@@ -0,0 +1,226 @@
/* Copyright (c) 2022 Advanced Micro Devices, Inc.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE. */
#include "code_printing.hpp"
#include <algorithm>
#include <fstream>
#include <iomanip>
#include <iostream>
#include <map>
#include <memory>
#include <mutex>
#include <optional>
#include <string>
#include <type_traits>
#include <unordered_map>
#include <vector>
#include <cstdarg>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <sys/types.h>
#include <sys/stat.h>
#include <fcntl.h>
#include <unistd.h>
#include <hsa/amd_hsa_elf.h>
#include "../utils.h"
#include <cxxabi.h>
#include <elfutils/libdw.h>
#include <sys/mman.h>
code_object_decoder_t::code_object_decoder_t(const char* codeobj_data, uint64_t codeobj_size) {
buffer = std::vector<char>{};
buffer.resize(codeobj_size);
std::memcpy(buffer.data(), codeobj_data, codeobj_size);
m_fd = -1;
#if defined(_GNU_SOURCE) && defined(MFD_ALLOW_SEALING) && defined(MFD_CLOEXEC)
m_fd = ::memfd_create(m_uri.c_str(), MFD_ALLOW_SEALING | MFD_CLOEXEC);
#endif
if (m_fd == -1) // If fail, attempt under /tmp
m_fd = ::open("/tmp", O_TMPFILE | O_RDWR, 0666);
if (m_fd == -1) {
printf("could not create a temporary file for code object\n");
return;
}
if (size_t size = ::write(m_fd, buffer.data(), buffer.size()); size != buffer.size()) {
printf("could not write to the temporary file\n");
return;
}
::lseek(m_fd, 0, SEEK_SET);
fsync(m_fd);
m_line_number_map = {};
std::unique_ptr<Dwarf, void (*)(Dwarf*)> dbg(dwarf_begin(m_fd, DWARF_C_READ),
[](Dwarf* dbg) { dwarf_end(dbg); });
/*if (!dbg) {
rocprofiler::warning("Error opening Dwarf!\n");
return;
} */
if (dbg) {
Dwarf_Off cu_offset{0}, next_offset;
size_t header_size;
while (!dwarf_nextcu(dbg.get(), cu_offset, &next_offset, &header_size, nullptr, nullptr,
nullptr)) {
Dwarf_Die die;
if (!dwarf_offdie(dbg.get(), cu_offset + header_size, &die)) continue;
Dwarf_Lines* lines;
size_t line_count;
if (dwarf_getsrclines(&die, &lines, &line_count)) continue;
for (size_t i = 0; i < line_count; ++i) {
Dwarf_Addr addr;
int line_number;
if (Dwarf_Line* line = dwarf_onesrcline(lines, i))
if (!dwarf_lineaddr(line, &addr) && !dwarf_lineno(line, &line_number) && line_number) {
m_line_number_map.emplace(
addr, std::make_pair(dwarf_linesrc(line, nullptr, nullptr), line_number));
}
}
cu_offset = next_offset;
}
// load_symbol_map();
}
disassemble_kernels();
}
code_object_decoder_t::~code_object_decoder_t() {
if (m_fd) ::close(m_fd);
}
std::optional<code_object_decoder_t::symbol_info_t> code_object_decoder_t::find_symbol(
uint64_t address) {
/* Load the symbol table. */
if (auto it = m_symbol_map.upper_bound(address); it != m_symbol_map.begin()) {
if (auto&& [symbol_value, symbol] = *std::prev(it); address < (symbol_value + symbol.second)) {
std::string symbol_name = symbol.first;
if (int status; auto* demangled_name =
abi::__cxa_demangle(symbol_name.c_str(), nullptr, nullptr, &status)) {
symbol_name = demangled_name;
free(demangled_name);
}
return symbol_info_t{std::move(symbol_name), symbol_value, symbol.second};
}
}
return {};
}
/*
void code_object_decoder_t::load_symbol_map() {
std::unique_ptr<Elf, void (*)(Elf *)> elf (
elf_begin(m_fd, ELF_C_READ, nullptr),
[](Elf *elf){ elf_end(elf); });
if (!elf) {
rocprofiler::warning("Error opening ELF!\n");
return;
}
Elf64_Ehdr *ehdr = elf64_getehdr(elf.get());
if (!ehdr) {
printf("elf64_getehdr failed\n");
return;
}
// Slurp the symbol table.
Elf_Scn *scn = nullptr;
while ((scn = elf_nextscn(elf.get(), scn)) != nullptr) {
GElf_Shdr shdr_mem;
GElf_Shdr *shdr = gelf_getshdr(scn, &shdr_mem);
if (shdr->sh_type != SHT_SYMTAB && shdr->sh_type != SHT_DYNSYM) {
continue;
}
Elf_Data *data = elf_getdata(scn, nullptr);
if (!data) continue;
size_t symbol_count = data->d_size / gelf_fsize(elf.get(), ELF_T_SYM, 1, EV_CURRENT);
for (size_t j = 0; j < symbol_count; ++j) {
GElf_Sym sym_mem;
GElf_Sym *sym = gelf_getsym(data, j, &sym_mem);
if (GELF_ST_TYPE(sym->st_info) != STT_FUNC || sym->st_shndx == SHN_UNDEF) continue;
std::string symbol_name{ elf_strptr(elf.get(), shdr->sh_link, sym->st_name) };
auto symbol_pair = std::make_pair(symbol_name, sym->st_size);
auto [it, success] = m_symbol_map.emplace(sym->st_value, symbol_pair);
// If there already was a symbol defined at this address, but this
// new symbol covers a larger address range, replace the old symbol
// with this new one.
if (!success && sym->st_size > it->second.second) it->second = symbol_pair;
}
}
} */
void code_object_decoder_t::disassemble_kernel(uint64_t addr) {
auto symbol = find_symbol(addr);
if (!symbol) {
std::cerr << "No symbol found at address 0x" << std::hex << addr << std::endl;
return;
}
// if (symbol->m_name.find("__amd_rocclr_") == 0)
// return;
std::cout << "Dumping ISA for " << symbol->m_name << std::endl;
uint64_t end_addr = addr + symbol->m_size;
while (addr < end_addr) {
char* cpp_line = nullptr;
auto it = m_line_number_map.find(addr);
if (it != m_line_number_map.end()) {
const std::string& file_name = it->second.first;
size_t line_number = it->second.second;
std::string cpp = file_name + ':' + std::to_string(line_number);
cpp_line = (char*)calloc(cpp.size() + 4, sizeof(char));
std::memcpy(cpp_line, cpp.data(), cpp.size() * sizeof(char));
}
size_t size = disassembly->ReadInstruction(addr, cpp_line);
addr += size;
}
}
void code_object_decoder_t::disassemble_kernels() {
disassembly = std::make_unique<DisassemblyInstance>(*this);
// if (m_symbol_map.begin() == m_symbol_map.end())
m_symbol_map = disassembly->GetKernelMap();
for (auto& [k, v] : m_symbol_map) disassemble_kernel(k);
}
@@ -0,0 +1,56 @@
/* Copyright (c) 2022 Advanced Micro Devices, Inc.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE. */
#pragma once
#include <map>
#include <optional>
#include <string>
#include <vector>
#include "rocprofiler.h"
#include <memory>
#include "disassembly.hpp"
class code_object_decoder_t {
struct symbol_info_t {
const std::string m_name;
uint64_t m_value;
uint64_t m_size;
};
public:
// void load_symbol_map();
std::optional<symbol_info_t> find_symbol(uint64_t address);
code_object_decoder_t(const char* codeobj_data, uint64_t codeobj_size);
~code_object_decoder_t();
void disassemble_kernel(uint64_t addr);
void disassemble_kernels();
int m_fd;
std::map<uint64_t, std::pair<std::string, size_t>> m_line_number_map;
std::map<uint64_t, std::pair<std::string, uint64_t>> m_symbol_map;
std::string m_uri;
std::vector<char> buffer;
std::vector<instruction_instance_t> instructions;
std::unique_ptr<DisassemblyInstance> disassembly;
};
+228
مشاهده پرونده
@@ -0,0 +1,228 @@
/* Copyright (c) 2022 Advanced Micro Devices, Inc.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE. */
#if !defined(_GNU_SOURCE) || !defined(_XOPEN_SOURCE)
#define _XOPEN_SOURCE 700
#endif
#include "code_printing.hpp"
#include <algorithm>
#include <fstream>
#include <iomanip>
#include <iostream>
#include <map>
#include <memory>
#include <mutex>
#include <optional>
#include <string>
#include <type_traits>
#include <unordered_map>
#include <vector>
#include <cstdarg>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <sys/mman.h>
#include <sys/types.h>
#include <sys/stat.h>
#include <fcntl.h>
#include <unistd.h>
#include <hsa/amd_hsa_elf.h>
#include "../utils.h"
#include <cxxabi.h>
#include <elfutils/libdw.h>
#define CHECK_COMGR(call) \
if (amd_comgr_status_s status = call) { \
const char* reason = ""; \
amd_comgr_status_string(status, &reason); \
std::cerr << __LINE__ << " failed: " << reason << std::endl; \
return; \
}
CodeObjectBinary::CodeObjectBinary(const std::string& uri) : m_uri(uri) {
const std::string protocol_delim{"://"};
size_t protocol_end = m_uri.find(protocol_delim);
std::string protocol = m_uri.substr(0, protocol_end);
protocol_end += protocol_delim.length();
std::transform(protocol.begin(), protocol.end(), protocol.begin(),
[](unsigned char c) { return std::tolower(c); });
std::string path;
size_t path_end = m_uri.find_first_of("#?", protocol_end);
if (path_end != std::string::npos) {
path = m_uri.substr(protocol_end, path_end++ - protocol_end);
} else {
path = m_uri.substr(protocol_end);
}
/* %-decode the string. */
std::string decoded_path;
decoded_path.reserve(path.length());
for (size_t i = 0; i < path.length(); ++i)
if (path[i] == '%' && std::isxdigit(path[i + 1]) && std::isxdigit(path[i + 2])) {
decoded_path += std::stoi(path.substr(i + 1, 2), 0, 16);
i += 2;
} else {
decoded_path += path[i];
}
/* Tokenize the query/fragment. */
std::vector<std::string> tokens;
size_t pos, last = path_end;
while ((pos = m_uri.find('&', last)) != std::string::npos) {
tokens.emplace_back(m_uri.substr(last, pos - last));
last = pos + 1;
}
if (last != std::string::npos) {
tokens.emplace_back(m_uri.substr(last));
}
/* Create a tag-value map from the tokenized query/fragment. */
std::unordered_map<std::string, std::string> params;
std::for_each(tokens.begin(), tokens.end(), [&](std::string& token) {
size_t delim = token.find('=');
if (delim != std::string::npos) {
params.emplace(token.substr(0, delim), token.substr(delim + 1));
}
});
buffer = std::vector<char>{};
try {
size_t offset{0}, size{0};
if (auto offset_it = params.find("offset"); offset_it != params.end()) {
offset = std::stoul(offset_it->second, nullptr, 0);
}
if (auto size_it = params.find("size"); size_it != params.end()) {
if (!(size = std::stoul(size_it->second, nullptr, 0))) return;
}
if (protocol != "file") {
printf("\"%s\" protocol not supported\n", protocol.c_str());
return;
}
std::ifstream file(decoded_path, std::ios::in | std::ios::binary);
if (!file) {
printf("could not open `%s'\n", decoded_path.c_str());
return;
}
if (!size) {
file.ignore(std::numeric_limits<std::streamsize>::max());
size_t bytes = file.gcount();
file.clear();
if (bytes < offset) {
printf("invalid uri `%s' (file size < offset)\n", decoded_path.c_str());
return;
}
size = bytes - offset;
}
file.seekg(offset, std::ios_base::beg);
buffer.resize(size);
file.read(&buffer[0], size);
} catch (...) {
}
}
DisassemblyInstance::DisassemblyInstance(code_object_decoder_t& decoder)
: buffer(reinterpret_cast<int64_t>(decoder.buffer.data())),
size(decoder.buffer.size()),
instructions(decoder.instructions) {
amd_comgr_create_data(AMD_COMGR_DATA_KIND_RELOCATABLE, &data);
amd_comgr_set_data(data, size, decoder.buffer.data());
char isa_name[128];
size_t isa_size = sizeof(isa_name);
amd_comgr_get_data_isa_name(data, &isa_size, isa_name);
CHECK_COMGR(amd_comgr_create_disassembly_info(
isa_name, //"amdgcn-amd-amdhsa--gfx1100",
&DisassemblyInstance::memory_callback, &DisassemblyInstance::inst_callback,
[](uint64_t address, void* user_data) {}, &info));
}
amd_comgr_status_t DisassemblyInstance::symbol_callback(amd_comgr_symbol_t symbol,
void* user_data) {
amd_comgr_symbol_type_t type;
amd_comgr_symbol_get_info(symbol, AMD_COMGR_SYMBOL_INFO_TYPE, &type);
if (type != AMD_COMGR_SYMBOL_TYPE_FUNC && type != AMD_COMGR_SYMBOL_TYPE_AMDGPU_HSA_KERNEL)
return AMD_COMGR_STATUS_SUCCESS;
uint64_t addr;
amd_comgr_symbol_get_info(symbol, AMD_COMGR_SYMBOL_INFO_VALUE, &addr);
uint64_t mem_size;
amd_comgr_symbol_get_info(symbol, AMD_COMGR_SYMBOL_INFO_SIZE, &mem_size);
uint64_t name_size;
amd_comgr_symbol_get_info(symbol, AMD_COMGR_SYMBOL_INFO_NAME_LENGTH, &name_size);
std::string name;
name.resize(name_size);
amd_comgr_symbol_get_info(symbol, AMD_COMGR_SYMBOL_INFO_NAME, name.data());
static_cast<DisassemblyInstance*>(user_data)->symbol_map[addr] = {name, mem_size};
return AMD_COMGR_STATUS_SUCCESS;
}
std::map<uint64_t, std::pair<std::string, uint64_t>>& DisassemblyInstance::GetKernelMap() {
symbol_map = std::map<uint64_t, std::pair<std::string, uint64_t>>{};
amd_comgr_iterate_symbols(data, &DisassemblyInstance::symbol_callback, this);
return symbol_map;
}
DisassemblyInstance::~DisassemblyInstance() {
amd_comgr_release_data(data);
CHECK_COMGR(amd_comgr_destroy_disassembly_info(info));
}
uint64_t DisassemblyInstance::ReadInstruction(uint64_t addr, const char* cpp_line) {
uint64_t size_read;
amd_comgr_disassemble_instruction(info, buffer + addr, (void*)this, &size_read);
instructions.back().address = addr;
instructions.back().cpp_reference = cpp_line;
return size_read;
}
uint64_t DisassemblyInstance::memory_callback(uint64_t from, char* to, uint64_t size,
void* user_data) {
DisassemblyInstance& instance = *static_cast<DisassemblyInstance*>(user_data);
size_t copysize = std::min((int64_t)size, instance.buffer + instance.size - (int64_t)from);
std::memcpy(to, (char*)from, copysize);
return copysize;
}
void DisassemblyInstance::inst_callback(const char* instruction, void* user_data) {
DisassemblyInstance& instance = *static_cast<DisassemblyInstance*>(user_data);
instance.instructions.push_back({strdup(instruction), nullptr, 0});
}
@@ -0,0 +1,59 @@
/* Copyright (c) 2022 Advanced Micro Devices, Inc.
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE. */
#pragma once
#include <string>
#include <vector>
#include <amd_comgr/amd_comgr.h>
#include <memory>
typedef struct {
const char* instruction;
const char* cpp_reference;
uint64_t address;
} instruction_instance_t;
class CodeObjectBinary {
public:
CodeObjectBinary(const std::string& uri);
std::string m_uri;
std::vector<char> buffer;
};
class DisassemblyInstance {
public:
DisassemblyInstance(class code_object_decoder_t& decoder);
~DisassemblyInstance();
uint64_t ReadInstruction(uint64_t addr, const char* cpp_line);
std::map<uint64_t, std::pair<std::string, uint64_t>>& GetKernelMap();
static uint64_t memory_callback(uint64_t from, char* to, uint64_t size, void* user_data);
static void inst_callback(const char* instruction, void* user_data);
static amd_comgr_status_t symbol_callback(amd_comgr_symbol_t symbol, void* user_data);
int64_t buffer;
int64_t size;
std::vector<instruction_instance_t>& instructions;
amd_comgr_disassembly_info_t info;
amd_comgr_data_t data;
std::map<uint64_t, std::pair<std::string, uint64_t>> symbol_map;
};
+2
مشاهده پرونده
@@ -120,6 +120,7 @@ def draw_wave_states(selections, normalize, TIMELINES):
plt.figure(figsize=(15,4))
maxtime = max([np.max((TIMELINES[k]!=0)*np.arange(0,TIMELINES[k].size)) for k in plot_indices])
maxtime = max(maxtime, 1)
timelines = [deepcopy(TIMELINES[k][:maxtime]) for k in plot_indices]
timelines = [np.pad(t, [0, maxtime-t.size]) for t in timelines]
@@ -161,6 +162,7 @@ def draw_occupancy(selections, normalize, OCCUPANCY, shadernames):
OCCUPANCY = [occ for occ in OCCUPANCY if len(occ) > 0]
maxtime = 1
delta = 1
for name, occ in zip(shadernames, OCCUPANCY):
occ_values = [0]
occ_times = [0]
+118 -35
مشاهده پرونده
@@ -6,7 +6,7 @@ if sys.version_info[0] < 3:
from collections import defaultdict
from copy import deepcopy
MAX_STITCHED_TOKENS = 10000000
MAX_STITCHED_TOKENS = 100000000
MAX_FAILED_STITCHES = 256
STACK_SIZE_LIMIT = 64
@@ -25,6 +25,7 @@ GETPC = 11
SETPC = 12
SWAPPC = 13
LANEIO = 14
PCINFO = 15
DONT_KNOW = 100
WaveInstCategory = {
@@ -46,10 +47,11 @@ WaveInstCategory = {
SETPC: "SETPC",
SWAPPC: "SWAPPC",
LANEIO: "LANEIO",
PCINFO: "PCINFO",
DONT_KNOW: "DONT_KNOW",
}
# Keeps track of register states for hipcc-generated assembly
class RegisterWatchList:
def __init__(self, labels):
self.registers = {'v'+str(k): [[] for m in range(64)] for k in range(64)}
@@ -84,7 +86,7 @@ class RegisterWatchList:
except:
pass
def swappc(self, line, line_num):
def swappc(self, line, line_num, inst_num):
try:
tokens = self.tokenize(line)
dst = tokens[1]
@@ -97,10 +99,9 @@ class RegisterWatchList:
except:
return 0
def setpc(self, line):
def setpc(self, line, inst_num):
try:
src = line.split(' ')[1].strip()
#print('Going to:', self.registers[self.range(src)[0]], src)
popped = self.registers[self.range(src)[0]][-1]
self.registers[self.range(src)[0]] = self.registers[self.range(src)[0]][:-1]
return popped
@@ -140,16 +141,57 @@ class RegisterWatchList:
except Exception as e:
pass
# Translates PC values to instructions, for auto captured ISA
class PCTranslator:
def __init__(self, code, insts):
self.code = code
self.insts = insts
self.addrmap = {code[m][-3] : m for m in range(len(code))}
def try_translate(self, tok):
pass
def range(self, r):
pass
def tokenize(self, line):
pass
def getpc(self, line, next_line):
pass
def swappc(self, line, line_num, inst_index):
try:
loc = self.addrmap[self.insts[inst_index+1][2]]
#print('Jumping to:', loc, self.code[loc])
return loc
except:
print('SWAPPC: Could not find addr', self.insts[inst_index+1][2], 'for', line)
return -1
def setpc(self, line, inst_index):
try:
loc = self.addrmap[self.insts[inst_index+1][2]]
#print('Jumping to:', loc, self.code[loc])
return loc
except:
print('SETPC: Could not find addr', self.insts[inst_index+1][2], 'for', line)
return -1
def scratch(self, line):
pass
def move(self, line):
pass
def updatelane(self, line):
pass
# Matches tokens in reverse order
def try_match_swapped(insts, code, i, line):
return insts[i+1][1] == code[line][1] and insts[i][1] == code[line+1][1]
FORK_NAMES = 1
# A successful parsed instruction
class CachedInst:
def __init__(self, inst, as_line):
self.inst_type = inst
self.as_line = as_line
self.forks = None
# A branch of the parsing tree
class Fork:
def __init__(self):
global FORK_NAMES
@@ -159,19 +201,21 @@ class Fork:
FORK_NAMES += 1
#print('Created new fork: ', self.name)
def move_down_fork(fork, insts, i): #def move_down_fork(fork : Fork, insts : list, i : int):
# Try to match sequence "insts" with the branch "fork", starting at position "i"
def move_down_fork(fork, insts, i): #(fork : Fork, insts : list, i : int):
N = min(len(insts), len(fork.insts))
while i < N:
if insts[i][1] == fork.insts[i].inst_type:
i += 1
elif i<N-1 and insts[i+1][1] == fork.insts[i].inst_type and insts[i][1] == fork.insts[i+1].inst_type:
elif i<N-1 and insts[i+1][1] == fork.insts[i].inst_type \
and insts[i][1] == fork.insts[i+1].inst_type:
i += 2
else:
#print('Failed at', i, insts[i])
return False, i
if len(fork.insts) < len(insts):
if len(fork.insts) != len(insts):
#print('Failed at the end at', i, insts[i])
return False, i
@@ -180,11 +224,11 @@ def move_down_fork(fork, insts, i): #def move_down_fork(fork : Fork, insts : lis
FORK_TREE = Fork()
# Check if there exists a previous wave with the same sequence of instructions executed
def fromDict(insts):
i = 0
N = len(insts)
cur_fork = FORK_TREE
#print('Getting from dict')
while i < N:
tillEnd, final_pos = move_down_fork(cur_fork, insts, i)
if tillEnd:
@@ -192,7 +236,6 @@ def fromDict(insts):
return True, cur_fork
i += final_pos
#print('Got fpos:', i, 'of', len(insts))
if i >= len(cur_fork.insts):
return False, cur_fork
@@ -204,7 +247,6 @@ def fromDict(insts):
bMatchFork = False
for fork in last_inst.forks:
if fork.insts[0].inst_type == insts[0][1]:
#print('Found match fork', fork.name)
cur_fork = fork
bMatchFork = True
break
@@ -217,10 +259,30 @@ def fromDict(insts):
return False, cur_fork
def stitch(insts, raw_code, jumps, gfxv):
def stitch(insts, raw_code, jumps, gfxv, bIsAuto):
bGFX9 = gfxv == 'vega'
# Try from cached result from a previous wave that have already been parsed
dict_sucess, current_fork = fromDict(insts)
if dict_sucess:
result, loopCount, mem_unroll, flight_count, maxline, pcsequence = current_fork.data
# Check if the sequence of measured PC values are equal for cached and new wave
if len(pcsequence) > 0:
pcs = [r[2] for r in insts if r[1] == PCINFO]
if len(pcs) != len(pcsequence):
dict_sucess = False
for pc1, pc2 in zip(pcs, pcsequence):
if pc1 != pc2:
dict_sucess = False
# If successful, use resulting assembly from cache
if dict_sucess:
result = [r+(asm[-1],) for r, asm in zip(insts, result)]
return result, loopCount, mem_unroll, flight_count, maxline, len(result)
result, i, line, loopCount, N = [], 0, 0, defaultdict(int), len(insts)
SMEM_INST = [] # scalar memory
VLMEM_INST = [] # vector memory load
VSMEM_INST = [] # vector memory store
@@ -236,8 +298,14 @@ def stitch(insts, raw_code, jumps, gfxv):
labels = {}
jump_map = [0]
# Clean the code and remove comments
code = [raw_code[0]]
for c in raw_code[1:]:
if bIsAuto and '; Begin ' == c[0][:len('; Begin ')]:
if '; Begin <Kernel>' in c[0]:
line = len(code)
print('Begin at:', line, c)
c = list(c)
c[0] = c[0].split(';')[0].split('//')[0].strip()
@@ -254,25 +322,22 @@ def stitch(insts, raw_code, jumps, gfxv):
jumps = {jump_map[j]+1: j for j in jumps}
# Checks if we have guaranteed ordering in memory operations
smem_ordering = 0
vlmem_ordering = 0
vsmem_ordering = 0
watchlist = RegisterWatchList(labels=labels)
num_failed_stitches = 0
loops = 0
maxline = 0
dict_sucess, current_fork = fromDict(insts)
if dict_sucess:
result, loopCount, mem_unroll, flight_count, maxline = current_fork.data
result = [r+(asm[-1],) for r, asm in zip(insts, result)]
return result, loopCount, mem_unroll, flight_count, maxline, len(insts)
watchlist = RegisterWatchList(labels=labels) if not bIsAuto else PCTranslator(code, insts)
pcsequence = []
while i < N:
loops += 1
if line >= len(code) or loops > MAX_STITCHED_TOKENS or num_failed_stitches > MAX_FAILED_STITCHES:
if line >= len(code) or loops > MAX_STITCHED_TOKENS \
or num_failed_stitches > MAX_FAILED_STITCHES:
break
maxline = max(reverse_map[line], maxline)
@@ -282,23 +347,37 @@ def stitch(insts, raw_code, jumps, gfxv):
matched = True
next = line+1
if '_mov_' in as_line[0]:
watchlist.move(as_line[0])
elif 'scratch_' in as_line[0]:
watchlist.scratch(as_line[0])
if not bIsAuto:
if '_mov_' in as_line[0]:
watchlist.move(as_line[0])
elif 'scratch_' in as_line[0]:
watchlist.scratch(as_line[0])
if as_line[1] == GETPC:
watchlist.getpc(as_line[0], code[line+1][0])
matched = inst[1] in [SALU, JUMP]
try:
watchlist.getpc(as_line[0], code[line+1][0])
matched = inst[1] in [SALU, JUMP]
except:
matched = False
elif as_line[1] == LANEIO:
watchlist.updatelane(as_line[0])
matched = inst[1] == VALU
elif as_line[1] == SETPC:
next = watchlist.setpc(as_line[0])
next = watchlist.setpc(as_line[0], i)
matched = inst[1] in [SALU, JUMP]
if bIsAuto:
matched = next >= 0
i += 1
result.append((insts[i][0], PCINFO, 0, 0, 0))
pcsequence.append(insts[i][2])
elif as_line[1] == SWAPPC:
next = watchlist.swappc(as_line[0], line)
next = watchlist.swappc(as_line[0], line, i)
matched = inst[1] in [SALU, JUMP]
if bIsAuto:
matched = next >= 0
i += 1
result.append((insts[i][0], PCINFO, 0, 0, 0))
pcsequence.append(insts[i][2])
elif inst[1] == as_line[1]:
if line in jumps:
loopCount[jumps[line]-1] += 1
@@ -366,10 +445,10 @@ def stitch(insts, raw_code, jumps, gfxv):
if 'vscnt' in as_line[0] or (bGFX9 and 'vmcnt' in as_line[0]):
try:
wait_N = int(as_line[0].split('vmcnt(')[1].split(')')[0])
wait_N = int(as_line[0].split('vscnt(')[1].split(')')[0])
except:
try:
wait_N = int(as_line[0].split('vscnt(')[1].split(')')[0])
wait_N = int(as_line[0].split('vmcnt(')[1].split(')')[0])
except:
wait_N = 0
flight_count.append([as_line[5], num_inflight, wait_N])
@@ -407,10 +486,13 @@ def stitch(insts, raw_code, jumps, gfxv):
if skipped_immed > 0 and 's_waitcnt ' in as_line[0]:
matched = True
skipped_immed -= 1
else:
elif 'scratch_' not in as_line[0]:
print('Parsing terminated at:', as_line)
break
#print(matched, as_line)
#print([WaveInstCategory[insts[i+k][1]] for k in range(10) if i+k < len(insts)])
if matched:
result.append(inst + (reverse_map[line],))
i += 1
@@ -425,8 +507,8 @@ def stitch(insts, raw_code, jumps, gfxv):
line = next
N = max(N, 1)
if len(result) != N:
print('Warning - Stitching rate: '+str(len(result) * 100 / N)+'% matched')
if i != N:
print('Warning - Stitching rate: '+str(i * 100 / N)+'% matched')
print('Leftovers:', [WaveInstCategory[insts[i+k][1]] for k in range(20) if i+k < len(insts)])
try:
print(line, code[line])
@@ -440,5 +522,6 @@ def stitch(insts, raw_code, jumps, gfxv):
line += 1
current_fork.insts = [CachedInst(inst[1], inst[-1]) for inst in result]
current_fork.data = result, loopCount, mem_unroll, flight_count, maxline
return result, loopCount, mem_unroll, flight_count, maxline, len(insts)
current_fork.data = result, loopCount, mem_unroll, flight_count, maxline, pcsequence
result = [r for r in result if r[1] != PCINFO]
return result, loopCount, mem_unroll, flight_count, maxline, len(result) if i == N else N
+2 -1
مشاهده پرونده
@@ -278,7 +278,8 @@ def view_trace(args, code, dbnames, att_filenames, bReturnLoc, OCCUPANCY, bDumpO
simd_wave_filenames[se_number] = wv_filenames
if mpi_root:
JSON_GLOBAL_DICTIONARY['code.json'] = Readable({"code": code[:allse_maxline+16], "top_n": get_top_n(code[:allse_maxline+16])})
code_sel = [c[:-3]+c[-2:] for c in code[:allse_maxline+16]]
JSON_GLOBAL_DICTIONARY['code.json'] = Readable({"code": code_sel, "top_n": get_top_n(code_sel)})
for key in simd_wave_filenames.keys():
wv_array = [[