TCC per channel metric's fix for rocprofv3 (#597)

[ROCm/rocprofiler-compute commit: 46e15dc840]
This commit is contained in:
ywang103-amd
2025-03-20 20:09:27 -04:00
committed by GitHub
parent 77ffa73e5b
commit b812f54a9b
3 changed files with 167 additions and 41 deletions
@@ -58,13 +58,6 @@ class rocprof_v3_profiler(RocProfCompute_Base):
trace_option = "--hip-trace"
args = [
"-E",
os.path.join(
str(config.rocprof_compute_home),
"rocprof_compute_soc",
"profile_configs",
"accum_counters.yaml",
),
# v3 requires output directory argument
"-d",
self.get_args().path + "/" + "out",
@@ -27,11 +27,13 @@ import math
import os
import re
import shutil
import threading
from abc import ABC, abstractmethod
from collections import OrderedDict
from pathlib import Path
import numpy as np
import pandas as pd
import yaml
from rocprof_compute_base import MI300_CHIP_IDS, SUPPORTED_ARCHS
@@ -42,6 +44,8 @@ from utils.utils import (
console_log,
convert_metric_id_to_panel_idx,
demarcate,
get_default_accumulate_counter_file_ymal,
using_v3,
)
@@ -317,6 +321,8 @@ class OmniSoC_Base:
workload_dir,
self.get_args().spatial_multiplexing,
self.__section_counters,
self._mspec,
self.__arch,
)
# ----------------------------------------------------
@@ -364,7 +370,9 @@ class LimitedSet:
# block limited according to perfmon config.
class CounterFile:
def __init__(self, name, perfmon_config) -> None:
self.file_name = name
name_no_extension = name.split(".")[0]
self.file_name_txt = name_no_extension + ".txt"
self.file_name_yaml = name_no_extension + ".yaml"
self.blocks = {b: LimitedSet(v) for b, v in perfmon_config.items()}
def add(self, counter) -> bool:
@@ -413,7 +421,13 @@ def parse_counters(config_text):
@demarcate
def perfmon_coalesce(
pmc_files_list, perfmon_config, workload_dir, spatial_multiplexing, section_counters
pmc_files_list,
perfmon_config,
workload_dir,
spatial_multiplexing,
section_counters,
mspec,
arch,
):
"""Sort and bucket all related performance counters to minimize required application passes"""
workload_perfmon_dir = workload_dir + "/perfmon"
@@ -447,12 +461,6 @@ def perfmon_coalesce(
else:
# Normal counters
for ctr in counters:
# v3 doesn't seem to support this counter
if using_v3():
if ctr.startswith("TCC_BUBBLE"):
continue
# Remove me later:
# v1 and v2 don't support these counters
if not using_v3():
@@ -554,7 +562,7 @@ def perfmon_coalesce(
ctrs.append(accum_name)
# Use the name of the accumulate counter as the file name
output_files.append(CounterFile(ctr_name + ".txt", perfmon_config))
output_files.append(CounterFile(ctr_name, perfmon_config))
for ctr in ctrs:
output_files[-1].add(ctr)
accu_file_count += 1
@@ -647,32 +655,142 @@ def perfmon_coalesce(
else:
# Output to files
for f in output_files:
file_name = str(Path(workload_perfmon_dir).joinpath(f.file_name))
file_name_txt = str(Path(workload_perfmon_dir).joinpath(f.file_name_txt))
file_name_yaml = str(Path(workload_perfmon_dir).joinpath(f.file_name_yaml))
pmc = []
for block_name in f.blocks.keys():
if not using_v3() and block_name == "TCC":
# Expand and interleve the TCC channel counters
# e.g. TCC_HIT[0] TCC_ATOMIC[0] ... TCC_HIT[1] TCC_ATOMIC[1] ...
channel_counters = []
for ctr in f.blocks[block_name].elements:
if "_expand" in ctr:
channel_counters.append(ctr.split("_expand")[0])
for i in range(0, perfmon_config["TCC_channels"]):
for c in channel_counters:
pmc.append("{}[{}]".format(c, i))
# Handle the rest of the TCC counters
for ctr in f.blocks[block_name].elements:
if "_expand" not in ctr:
pmc.append(ctr)
if block_name == "TCC":
if using_v3():
# TODO: implement mehcanisim for muti channel TCC counter for v3
# Expand and interleve the TCC channel counters
# e.g. TCC_HIT[0] TCC_ATOMIC[0] ... TCC_HIT[1] TCC_ATOMIC[1] ...
from utils.specs import total_xcds
channel_counters = []
xcds = total_xcds(mspec.gpu_model, mspec.compute_partition)
tcc_channel_per_xcd = int(mspec._l2_banks)
for ctr in f.blocks[block_name].elements:
if "_sum" in ctr:
channel_counters.append(ctr.split("_sum")[0])
channel_counters = (
pd.Series(channel_counters).drop_duplicates().to_list()
)
lock = threading.Lock()
def generate_yaml_config_per_pmc(
raw_counter_name,
tcc_counter_1d_index,
xcd_index,
channel_index,
):
# Ensure that the keys exist before trying to assign values
yaml_data = {}
if tcc_counter_1d_index not in yaml_data:
yaml_data[tcc_counter_1d_index] = {
"architectures": {},
"description": "",
}
if (
arch
not in yaml_data[tcc_counter_1d_index]["architectures"]
):
yaml_data[tcc_counter_1d_index]["architectures"][
arch
] = {}
yaml_data[tcc_counter_1d_index]["architectures"][arch][
"expression"
] = "select({},[DIMENSION_XCC=[{}], DIMENSION_INSTANCE=[{}]])".format(
raw_counter_name, xcd_index, channel_index
)
yaml_data[tcc_counter_1d_index]["description"] = (
"{} on {}th XCC and {}th channel".format(
raw_counter_name, xcd_index, channel_index
)
)
lock.acquire()
pmc.append(tcc_counter_1d_index)
with open(file_name_yaml, "a") as file_yaml:
yaml.dump(
yaml_data,
file_yaml,
default_flow_style=False,
allow_unicode=True,
)
lock.release()
threads_edit_yaml = []
for i in range(0, xcds):
for j in range(0, tcc_channel_per_xcd):
for c in channel_counters:
tcc_counter_1d_index = "{}[{}]".format(
c, (i * tcc_channel_per_xcd) + j
)
thread = threading.Thread(
target=generate_yaml_config_per_pmc,
args=[c, tcc_counter_1d_index, i, j],
)
threads_edit_yaml.append(thread)
thread.start()
for thread in threads_edit_yaml:
thread.join()
# Handle the rest of the TCC counters
for ctr in f.blocks[block_name].elements:
if "_expand" not in ctr and "_sum" not in ctr:
pmc.append(ctr)
else:
# Expand and interleve the TCC channel counters
# e.g. TCC_HIT[0] TCC_ATOMIC[0] ... TCC_HIT[1] TCC_ATOMIC[1] ...
channel_counters = []
for ctr in f.blocks[block_name].elements:
if "_expand" in ctr:
channel_counters.append(ctr.split("_expand")[0])
for i in range(0, perfmon_config["TCC_channels"]):
for c in channel_counters:
pmc.append("{}[{}]".format(c, i))
# Handle the rest of the TCC counters
for ctr in f.blocks[block_name].elements:
if "_expand" not in ctr:
pmc.append(ctr)
else:
for ctr in f.blocks[block_name].elements:
pmc.append(ctr)
if using_v3():
yaml_global_config_dir = (
get_default_accumulate_counter_file_ymal()
)
with open(yaml_global_config_dir, "r") as file_read:
with open(file_name_yaml, "a") as file_out:
dic_read = yaml.safe_load(file_read)
for ctr in f.blocks[block_name].elements:
if ctr in dic_read:
section_to_dump = {ctr: dic_read[ctr]}
yaml.dump(
section_to_dump,
file_out,
default_flow_style=False,
allow_unicode=True,
)
pmc.append(ctr)
else:
for ctr in f.blocks[block_name].elements:
pmc.append(ctr)
stext = "pmc: " + " ".join(pmc)
# Write counters to file
fd = open(file_name, "w")
fd = open(file_name_txt, "w")
fd.write(stext + "\n\n")
fd.write("gpu:\n")
fd.write("range:\n")
@@ -47,6 +47,23 @@ rocprof_cmd = ""
rocprof_args = ""
# TODO: This is a HACK
def using_v3():
return "ROCPROF" in os.environ.keys() and "rocprofv3" in os.environ["ROCPROF"]
# TODO: This is a HACK
def get_default_accumulate_counter_file_ymal():
"""Return the path of the default derivative counters' definatin's yaml file that we current use to store accumulated counters' defination. It will possibly be removed later on"""
return str(
config.rocprof_compute_home.joinpath(
"rocprof_compute_soc",
"profile_configs",
"accum_counters.yaml",
)
)
def demarcate(function):
def wrap_function(*args, **kwargs):
logging.trace("----- [entering function] -> %s()" % (function.__qualname__))
@@ -570,9 +587,12 @@ def run_prof(
console_debug("pmc file: %s" % path(fname).name)
path_counter_config_yaml = path(fname).with_suffix(".yaml")
# standard rocprof options
default_options = ["-i", fname]
options = default_options + profiler_options
if path_counter_config_yaml.exists():
options = ["-E", str(path_counter_config_yaml)] + options
# set required env var for mi300
new_env = None
@@ -581,12 +601,6 @@ def run_prof(
or mspec.gpu_model.lower() == "mi300x_a1"
or mspec.gpu_model.lower() == "mi300a_a0"
or mspec.gpu_model.lower() == "mi300a_a1"
) and (
path(fname).name == "pmc_perf_13.txt"
or path(fname).name == "pmc_perf_14.txt"
or path(fname).name == "pmc_perf_15.txt"
or path(fname).name == "pmc_perf_16.txt"
or path(fname).name == "pmc_perf_17.txt"
):
new_env = os.environ.copy()
new_env["ROCPROFILER_INDIVIDUAL_XCC_MODE"] = "1"
@@ -596,6 +610,7 @@ def run_prof(
is_timestamps = True
time_1 = time.time()
console_debug("rocprof command: {}".format([rocprof_cmd] + options))
# profile the app
if new_env:
success, output = capture_subprocess_output(
@@ -668,7 +683,7 @@ def run_prof(
workload_dir + "/out/pmc_1/results_" + fbase + ".csv", index=False
)
if new_env:
if new_env and not using_v3():
# flatten tcc for applicable mi300 input
f = path(workload_dir + "/out/pmc_1/results_" + fbase + ".csv")
xcds = total_xcds(mspec.gpu_model, mspec.compute_partition)