[rocprof-compute] Generalize Roofline (#325)

* per kernel analysis Roofline

* added per-kernel eval_metric calculation with display

* fixed typo

* updated tty.py show_all()

* formatting

* fixed ctest failures and updated equations

* formatting

* updated metric descriptoins

* review tweaks

* update docs

* added roofline gui analysis

* updated GUI docs

* updated print statement

* comment tweaks and ran ruff formatting
Šī revīzija ir iekļauta:
jamessiddeley-amd
2025-08-20 09:58:08 -04:00
revīziju iesūtīja GitHub
vecāks 71b725f307
revīzija 5840940caa
26 mainīti faili ar 2612 papildinājumiem un 158 dzēšanām
@@ -103,6 +103,7 @@ supported_call = {
"STD": "to_std",
# functions apply to whole column of df or a single value
"TO_INT": "to_int",
"SUM": "to_sum",
# Support the below with 2 inputs
"ROUND": "to_round",
"QUANTILE": "to_quantile",
@@ -196,6 +197,19 @@ def to_int(a):
raise Exception("to_int: unsupported type.")
def to_sum(a):
if str(type(a)) == "<class 'NoneType'>":
return np.nan
elif np.isnan(a).all():
return np.nan
elif a.empty:
return np.nan
elif isinstance(a, pd.core.series.Series):
return a.sum()
else:
raise Exception("to_sum: unsupported type.")
def to_round(a, b):
if isinstance(a, pd.core.series.Series):
return a.round(b)
@@ -755,7 +769,7 @@ def build_metric_value_string(dfs, dfs_type, normal_unit, profiling_config):
@demarcate
def eval_metric(dfs, dfs_type, sys_info, raw_pmc_df, debug, config):
def eval_metric(dfs, dfs_type, sys_info, empirical_peaks_df, raw_pmc_df, debug, config):
"""
Execute the expr string for each metric in the df.
"""
@@ -860,6 +874,30 @@ def eval_metric(dfs, dfs_type, sys_info, raw_pmc_df, debug, config):
"wave_size is not available in sysinfo.csv, please provide the correct "
"value using --specs-correction"
)
if not empirical_peaks_df.empty:
peak_data_row = empirical_peaks_df.iloc[0]
for metric_name in empirical_peaks_df.columns:
var_name = f"ammolite__{metric_name}_empirical_peak"
locals()[var_name] = peak_data_row[metric_name]
else:
default_peaks = [
"MFMAF64Flops",
"MFMAF32Flops",
"MFMAF16Flops",
"MFMABF16Flops",
"MFMAF8Flops",
"MFMAI8Ops",
"HBMBw",
"L2Bw",
"L1Bw",
"LDSBw",
"MFMA_FLOPs_F6F4",
]
# set values to 0 if no no empirical peaks from roofline.csv are provided
for peak_name in default_peaks:
var_name = f"ammolite__{peak_name}_empirical_peak"
exec(f"{var_name} = 0", globals(), locals())
# TODO: fix all $normUnit in Unit column or title
# build and eval all derived build-in global variables
@@ -958,8 +996,7 @@ def eval_metric(dfs, dfs_type, sys_info, raw_pmc_df, debug, config):
except TypeError:
console_warning(
"Skipping entry. Encountered a missing "
"counter\n{} has been assigned to None\n{}"
.format(
"counter\n{} has been assigned to None\n{}".format(
expr,
np.nan,
)
@@ -984,8 +1021,14 @@ def eval_metric(dfs, dfs_type, sys_info, raw_pmc_df, debug, config):
row[expr] = ""
else:
row[expr] = out
except TypeError:
row[expr] = ""
except (TypeError, NameError) as e:
if "empirical_peak" in str(e):
console_warning(
f"Missing empirical peak data: {e}. Using empty value."
)
row[expr] = ""
else:
row[expr] = ""
except AttributeError as ae:
if (
str(ae)
@@ -1043,8 +1086,7 @@ def apply_filters(workload, dir, is_gui, debug):
for kernel_id in workload.filter_kernel_ids:
if kernel_id >= len(kernels_df["Kernel_Name"]):
console_error(
"{} is an invalid kernel id. Please enter an id between 0-{}"
.format(
"{} is an invalid kernel id. Please enter an id between 0-{}".format(
kernel_id,
len(kernels_df["Kernel_Name"]) - 1,
)
@@ -1579,6 +1621,7 @@ def load_table_data(workload, dir, is_gui, args, config, skipKernelTop=False):
workload.dfs,
workload.dfs_type,
workload.sys_info.iloc[0],
workload.roofline_peaks,
apply_filters(workload, dir, is_gui, args.debug),
args.debug,
config,
@@ -23,11 +23,15 @@
##############################################################################
import csv
from dataclasses import dataclass
from pathlib import Path
import pandas as pd
from utils.logger import console_debug
from utils.parser import apply_filters, eval_metric
################################################
# Global vars
@@ -154,8 +158,7 @@ def get_color(catagory):
# Plot BW at each cache level
# -------------------------------------------------------------------------------------
def calc_ceilings(roofline_parameters, dtype, benchmark_data):
"""Given benchmarking data, calculate ceilings
(or peak performance) for empirical roofline"""
"""Given benchmarking data, calculate ceilings (or peak performance) for empirical roofline"""
# TODO: This is where filtering by memory level will need to occur for standalone
graphPoints = {"hbm": [], "l2": [], "l1": [], "lds": [], "valu": [], "mfma": []}
@@ -186,7 +189,7 @@ def calc_ceilings(roofline_parameters, dtype, benchmark_data):
if dtype in PEAK_OPS_DATATYPES:
x2 = peakOps / peakBw
y2 = peakOps # noqa: F841
y2 = peakOps
# Plot MFMA lines (NOTE: Assuming MI200 soc)
x1_mfma = peakOps / peakBw
@@ -220,9 +223,9 @@ def calc_ceilings(roofline_parameters, dtype, benchmark_data):
graphPoints[cacheHierarchy[i].lower()].append([y1, peakY])
graphPoints[cacheHierarchy[i].lower()].append(peakBw)
# ---------------------------------------------------------------------------------
# -------------------------------------------------------------------------------------
# Plot computing roof
# ---------------------------------------------------------------------------------
# -------------------------------------------------------------------------------------
if dtype in PEAK_OPS_DATATYPES:
# Plot FMA roof
x0 = XMAX
@@ -254,9 +257,151 @@ def calc_ceilings(roofline_parameters, dtype, benchmark_data):
# Overlay application performance
# -------------------------------------------------------------------------------------
# Calculate relevant metrics for ai calculation
def calc_ai(mspec, sort_type, ret_df):
"""Given counter data, calculate arithmetic intensity
for each kernel in the application."""
def calc_ai_analyze(workload, mspec, sort_type, config, arch_config):
"""
Calculate per-kernel metrics and AI points with Roofline yamls using eval_metric.
"""
console_debug("calc_ai_analyze: Starting calc_ai analysis using Roofline yamls")
plot_points = {
"ai_l1": [[], []],
"ai_l2": [[], []],
"ai_hbm": [[], []],
"kernelNames": [],
}
workload.roofline_metrics = {}
filtered_pmc = apply_filters(workload, workload.path, is_gui=False, debug=False)
kernel_ids_to_process = []
kernel_top_table_id = 1
if workload.filter_kernel_ids:
kernel_ids_to_process = workload.filter_kernel_ids
else:
if kernel_top_table_id in workload.dfs:
kernel_top_df = workload.dfs[kernel_top_table_id]
kernel_ids_to_process = kernel_top_df.index.tolist()
console_debug(
"roofline", f"Found {len(kernel_ids_to_process)} kernels to process"
)
if not kernel_ids_to_process:
console_warning("No kernels found to process for roofline")
return plot_points
for kernel_id in kernel_ids_to_process:
if kernel_top_table_id in workload.dfs:
kernel_top_df = workload.dfs[kernel_top_table_id]
if kernel_id in kernel_top_df.index:
kernel_name = kernel_top_df.loc[kernel_id, "Kernel_Name"]
else:
continue
else:
continue
console_debug("roofline", f"Processing kernel {kernel_id}: {kernel_name[:50]}")
# filter PMC data for specific kernel
kernel_pmc_df = filtered_pmc[
filtered_pmc["pmc_perf"]["Kernel_Name"] == kernel_name
]
if kernel_pmc_df.empty:
console_debug("roofline", f"No PMC data for kernel {kernel_id}")
continue
kernel_only_data = {"pmc_perf": kernel_pmc_df["pmc_perf"]}
kernel_dfs = {}
kernel_dfs_type = {}
for table_id in [401, 402]:
if table_id in arch_config.dfs:
kernel_dfs[table_id] = arch_config.dfs[table_id].copy()
kernel_dfs_type[table_id] = arch_config.dfs_type[table_id]
# eval metrics for single kernel only
eval_metric(
kernel_dfs,
kernel_dfs_type,
workload.sys_info.iloc[0],
workload.roofline_peaks,
kernel_only_data,
debug=False,
config=config,
)
# DEBUG
if 402 in kernel_dfs:
console_debug("roofline", f"Table 402 for kernel {kernel_id}:")
for idx, row in kernel_dfs[402].iterrows():
console_debug(
"roofline", f" {row.get('Metric', '')}: {row.get('Value', '')}"
)
ai_hbm = ai_l2 = ai_l1 = performance = 0
if 402 in kernel_dfs:
for idx, row in kernel_dfs[402].iterrows():
metric = row.get("Metric", "")
value = row.get("Value", 0)
if metric == "AI HBM":
ai_hbm = value if value and value != "" else 0
elif metric == "AI L2":
ai_l2 = value if value and value != "" else 0
elif metric == "AI L1":
ai_l1 = value if value and value != "" else 0
elif metric == "Performance (GFLOPs)":
performance = value if value and value != "" else 0
console_debug(
"roofline",
f"Kernel {kernel_id}: AI_HBM={ai_hbm:.2f}, AI_L2={ai_l2:.2f}, AI_L1={ai_l1:.2f}, Performance={performance:.2e} GFLOP/s",
)
# add to plot points if we have valid data
if performance > 0:
if ai_hbm > 0:
plot_points["ai_hbm"][0].append(ai_hbm)
plot_points["ai_hbm"][1].append(performance)
if ai_l2 > 0:
plot_points["ai_l2"][0].append(ai_l2)
plot_points["ai_l2"][1].append(performance)
if ai_l1 > 0:
plot_points["ai_l1"][0].append(ai_l1)
plot_points["ai_l1"][1].append(performance)
plot_points["kernelNames"].append(f"K{kernel_id}")
console_debug("roofline", f"Added kernel {kernel_id} to plot points")
else:
console_debug(
"roofline", f"Skipping kernel {kernel_id} - no performance data"
)
# store metrics for display
workload.roofline_metrics[kernel_id] = {
"name": kernel_name,
"ai_table": kernel_dfs.get(401, pd.DataFrame()),
"calc_table": kernel_dfs.get(402, pd.DataFrame()),
}
console_debug(
"roofline", f"Generated {len(plot_points['kernelNames'])} plot points"
)
console_debug("roofline", f"Plot points: {plot_points}")
return plot_points
def calc_ai_profile(mspec, sort_type, ret_df):
"""Given counter data, calculate arithmetic intensity for each kernel in the application.
Leverage hard-coded equations to calculate AI values.
Used during profiling stage to generate roofline PDF, since Roofline yamls are not available
in the profiling stage."""
console_debug(
"calc_ai_profile: Starting legacy roofline calculation (from roofline_calc)"
)
df = ret_df["pmc_perf"]
# Sort by top kernels or top dispatches?
df = df.sort_values(by=["Kernel_Name"])
@@ -463,7 +608,9 @@ def calc_ai(mspec, sort_type, ret_df):
calls += 1
if sort_type == "kernels" and (at_end or (kernelName != next_kernelName)):
if sort_type == "kernels" and (
at_end == True or (kernelName != next_kernelName)
):
myList.append(
AI_Data(
kernelName,
@@ -538,8 +685,9 @@ def calc_ai(mspec, sort_type, ret_df):
while i < TOP_N and i != len(myList):
if myList[i].total_flops == 0:
console_debug(
"No flops counted for {}, arithmetic intensities will not "
"display on plots.".format(myList[i].KernelName)
"No flops counted for {}, arithmetic intensities will not display on plots.".format(
myList[i].KernelName
)
)
kernelNames.append(myList[i].KernelName)
@@ -548,40 +696,28 @@ def calc_ai(mspec, sort_type, ret_df):
if myList[i].L1cache_data
else intensities["ai_l1"].append(0)
)
# print(
# "cur_ai_L1",
# myList[i].total_flops / myList[i].L1cache_data
# ) if myList[i].L1cache_data else print("null")
# print("cur_ai_L1", myList[i].total_flops/myList[i].L1cache_data) if myList[i].L1cache_data else print("null")
# print()
(
intensities["ai_l2"].append(myList[i].total_flops / myList[i].L2cache_data)
if myList[i].L2cache_data
else intensities["ai_l2"].append(0)
)
# print(
# "cur_ai_L2",
# myList[i].total_flops / myList[i].L2cache_data
# ) if myList[i].L2cache_data else print("null")
# print("cur_ai_L2", myList[i].total_flops/myList[i].L2cache_data) if myList[i].L2cache_data else print("null")
# print()
(
intensities["ai_hbm"].append(myList[i].total_flops / myList[i].hbm_data)
if myList[i].hbm_data
else intensities["ai_hbm"].append(0)
)
# print(
# "cur_ai_hbm",
# myList[i].total_flops / myList[i].hbm_data
# ) if myList[i].hbm_data else print("null")
# print("cur_ai_hbm", myList[i].total_flops/myList[i].hbm_data) if myList[i].hbm_data else print("null")
# print()
(
curr_perf.append(myList[i].total_flops / myList[i].avgDuration)
if myList[i].avgDuration
else curr_perf.append(0)
)
# print(
# "cur_perf",
# myList[i].total_flops / myList[i].avgDuration
# ) if myList[i].avgDuration else print("null")
# print("cur_perf", myList[i].total_flops/myList[i].avgDuration) if myList[i].avgDuration else print("null")
i += 1
@@ -590,7 +726,7 @@ def calc_ai(mspec, sort_type, ret_df):
for i in intensities:
values = intensities[i]
color = get_color(i) # noqa: F841
color = get_color(i)
x = []
y = []
for entryIndx in range(0, len(values)):
@@ -622,8 +758,7 @@ def constuct_roof(roofline_parameters, dtype):
# -----------------------------------------------------
# Initialize roofline data dictionary from roofline.csv
# -----------------------------------------------------
# TODO: consider changing this to an ordered dict for consistency over py versions
benchmark_data = {}
benchmark_data = {} # TODO: consider changing this to an ordered dict for consistency over py versions
headers = []
try:
with open(benchmark_results, "r") as csvfile:
@@ -641,7 +776,7 @@ def constuct_roof(roofline_parameters, dtype):
rowCount += 1
csvfile.close()
except Exception:
except:
graphPoints = {
"hbm": [None, None, None],
"l2": [None, None, None],
@@ -83,6 +83,7 @@ supported_field = [
"Avg",
"Pct of Peak",
"Peak",
"Peak (Empirical)",
"Count",
"Mean",
"Pct",
@@ -32,6 +32,7 @@ from tabulate import tabulate
import config
from utils import mem_chart, parser
from utils.kernel_name_shortener import kernel_name_shortener
from utils.logger import console_error, console_log, console_warning
from utils.utils import convert_metric_id_to_panel_info
@@ -146,6 +147,108 @@ def show_all(args, runs, archConfigs, output, profiling_config, roof_plot=None):
continue
ss = "" # store content of all data_source from one panel
if panel_id == 400:
has_roofline_style = any(
data_source.get(type, {}).get("cli_style") == "Roofline"
for data_source in panel["data source"]
for type in data_source
)
if has_roofline_style and (
not args.filter_metrics or "4" in args.filter_metrics
):
print("\n" + "=" * 80, file=output)
print("4. Roofline", file=output)
print("=" * 80, file=output)
for run_path, workload in runs.items():
if (
hasattr(workload, "roofline_metrics")
and workload.roofline_metrics
):
print(
"\n(4.1) Per-Kernel Roofline Metrics and (4.2) AI Plot Points",
file=output,
)
print("-" * 80, file=output)
kernel_top_df = workload.dfs.get(1, pd.DataFrame())
if not kernel_top_df.empty:
kernel_name_shortener(kernel_top_df, args.kernel_verbose)
for i, (kernel_id, metrics) in enumerate(
workload.roofline_metrics.items()
):
if (
not kernel_top_df.empty
and kernel_id in kernel_top_df.index
):
kernel_name = kernel_top_df.loc[
kernel_id, "Kernel_Name"
]
kernel_pct = (
kernel_top_df.loc[kernel_id, "Pct"]
if "Pct" in kernel_top_df.columns
else 0
)
else:
kernel_name = metrics.get("name", f"Kernel {kernel_id}")
kernel_pct = 0
display_name = (
kernel_name[:80] + "..."
if len(kernel_name) > 80
else kernel_name
)
print(
f"\nKernel {kernel_id}: {display_name} ({kernel_pct:.1f}%)",
file=output,
)
base_indent = " "
table_indent_prefix = f"{base_indent}| "
tables = {
401: (
"4.1 Roofline Rate Metrics:",
metrics.get("ai_table", pd.DataFrame()),
),
402: (
"4.2 Roofline AI Plot Points:",
metrics.get("calc_table", pd.DataFrame()),
),
}
print(f"{base_indent}|")
for table_id, (table_name, df) in tables.items():
if df.empty:
continue
print(f"{base_indent}├─ {table_name}", file=output)
display_df = df.copy()
for col in hidden_cols:
if col in display_df.columns:
display_df = display_df.drop(columns=[col])
table_string = get_table_string(
display_df, transpose=False, decimal=args.decimal
)
indented_table_string = textwrap.indent(
table_string, table_indent_prefix
)
print(indented_table_string, file=output)
else:
print("\nNo per-kernel metrics available", file=output)
# Show the roofline plot
if roof_plot:
show_roof_plot(roof_plot)
continue
for data_source in panel["data source"]:
for type, table_config in data_source.items():
# If block filtering was used during analysis, then don't use profiling
@@ -172,16 +275,6 @@ def show_all(args, runs, archConfigs, output, profiling_config, roof_plot=None):
)
continue
# Show roofline
# Check if we have filter_metrics for analyze stage:
# no filter_metrics = show all,
# filter_metrics containing "4" = user requesting roofline chart
if panel_id == 400 and (
not args.filter_metrics or "4" in args.filter_metrics
):
show_roof_plot(roof_plot)
continue
# Metrics baseline comparison mode
# We cannot guarantee that all runs have the same metrics.
# Only show common metrics.
@@ -454,7 +547,7 @@ def show_roof_plot(roof_plot):
# TODO: short term solution to display roofline plot
print("\n" + "-" * 80)
print("4. Roofline")
print("4.1 Roofline")
print("4.3 Roofline Plot")
if roof_plot:
print(roof_plot)
else:
@@ -745,7 +745,7 @@ def run_prof(
config.rocprof_compute_home
/ "rocprof_compute_soc"
/ "profile_configs"
/ f"counter_defs.yaml",
/ "counter_defs.yaml",
"r",
) as file:
counter_defs = yaml.safe_load(file)