Python formatting

Signed-off-by: colramos-amd <colramos@amd.com>


[ROCm/rocprofiler-compute commit: d6f45411eb]
이 커밋은 다음에 포함됨:
colramos-amd
2024-03-01 11:52:31 -06:00
커밋한 사람 Cole Ramos
부모 abefd57924
커밋 9ff07b51ff
19개의 변경된 파일688개의 추가작업 그리고 380개의 파일을 삭제
+3 -3
파일 보기
@@ -227,9 +227,9 @@ def build_bar_chart(display_df, table_config, barchart_elements, norm_filt):
).update_xaxes(range=[0, 1638])
) # append second GB/s chart
else:
key = 'Avg'
if table_config['id'] in [1101]:
key = 'Pct of Peak'
key = "Avg"
if table_config["id"] in [1101]:
key = "Pct of Peak"
d_figs.append(
px.bar(
display_df,
+12 -9
파일 보기
@@ -79,11 +79,11 @@ supported_denom = {
build_in_vars = {
"GRBM_GUI_ACTIVE_PER_XCD": "(GRBM_GUI_ACTIVE / $num_xcd)",
"GRBM_COUNT_PER_XCD": "(GRBM_COUNT / $num_xcd)",
"GRBM_SPI_BUSY_PER_XCD" : "(GRBM_SPI_BUSY / $num_xcd)",
"GRBM_SPI_BUSY_PER_XCD": "(GRBM_SPI_BUSY / $num_xcd)",
"numActiveCUs": "TO_INT(MIN((((ROUND(AVG(((4 * SQ_BUSY_CU_CYCLES) / $GRBM_GUI_ACTIVE_PER_XCD)), \
0) / $max_waves_per_cu) * 8) + MIN(MOD(ROUND(AVG(((4 * SQ_BUSY_CU_CYCLES) \
/ $GRBM_GUI_ACTIVE_PER_XCD)), 0), $max_waves_per_cu), 8)), $cu_per_gpu))",
"kernelBusyCycles": "ROUND(AVG((((End_Timestamp - Start_Timestamp) / 1000) * $max_sclk)), 0)"
"kernelBusyCycles": "ROUND(AVG((((End_Timestamp - Start_Timestamp) / 1000) * $max_sclk)), 0)",
}
supported_call = {
@@ -685,11 +685,11 @@ def eval_metric(dfs, dfs_type, sys_info, raw_pmc_df, debug):
ammolite__se_per_gpu = sys_info.se_per_gpu
ammolite__pipes_per_gpu = sys_info.pipes_per_gpu
ammolite__cu_per_gpu = sys_info.cu_per_gpu
ammolite__simd_per_cu = sys_info.simd_per_cu # not used
ammolite__simd_per_cu = sys_info.simd_per_cu # not used
ammolite__sqc_per_gpu = sys_info.sqc_per_gpu
ammolite__lds_banks_per_cu = sys_info.lds_banks_per_cu
ammolite__cur_sclk = sys_info.cur_sclk # not used
ammolite__mclk = sys_info.cur_mclk # not used
ammolite__mclk = sys_info.cur_mclk # not used
ammolite__max_sclk = sys_info.max_sclk
ammolite__max_waves_per_cu = sys_info.max_waves_per_cu
ammolite__hbm_bw = sys_info.hbm_bw
@@ -924,7 +924,9 @@ def load_kernel_top(workload, dir):
if file.exists():
tmp[id] = pd.read_csv(file)
else:
logging.info("Warning: Issue loading top kernels. Check pmc_kernel_top.csv")
logging.info(
"Warning: Issue loading top kernels. Check pmc_kernel_top.csv"
)
# NB: Special case for sysinfo. Probably room for improvement in this whole function design
elif "from_csv_columnwise" in df.columns and id == 101:
tmp[id] = workload.sys_info.transpose()
@@ -977,7 +979,8 @@ def build_comparable_columns(time_unit):
return comparable_columns
def correct_sys_info(mspec, specs_correction:dict):
def correct_sys_info(mspec, specs_correction: dict):
"""
Correct system spec items manually
"""
@@ -987,8 +990,8 @@ def correct_sys_info(mspec, specs_correction:dict):
for k, v in pairs.items():
if not hasattr(mspec, str(k)):
error(f"Invalid specs correction '{k}'. Please use --specs option to peak valid specs")
error(
f"Invalid specs correction '{k}'. Please use --specs option to peak valid specs"
)
setattr(mspec, str(k), v)
return mspec.get_class_members()
+376 -212
파일 보기
@@ -53,6 +53,7 @@ VERSION_LOC = [
"version-utils",
]
def detect_arch(_rocminfo):
from omniperf_base import SUPPORTED_ARCHS
@@ -68,12 +69,15 @@ def detect_arch(_rocminfo):
else:
return (gpu_arch, idx1)
def generate_machine_specs(args, sysinfo:dict=None):
def generate_machine_specs(args, sysinfo: dict = None):
if not sysinfo is None:
sysinfo_ver = str(sysinfo['version'])
version = get_version(config.omniperf_home)['version']
if sysinfo_ver != version[:version.find(".")]:
logging.warning("WARNING: Detected mismatch in sysinfo versioning. You may need to reprofile to update data.")
sysinfo_ver = str(sysinfo["version"])
version = get_version(config.omniperf_home)["version"]
if sysinfo_ver != version[: version.find(".")]:
logging.warning(
"WARNING: Detected mismatch in sysinfo versioning. You may need to reprofile to update data."
)
return MachineSpecs(**sysinfo)
# read timestamp info
now = datetime.now()
@@ -87,7 +91,9 @@ def generate_machine_specs(args, sysinfo:dict=None):
vData = get_version(config.omniperf_home)
version = vData["version"]
# NB: Just taking major as specs version. May want to make this more specific in the future
specs_version = version[:version.find(".")] # version will always follow 'major.minor.patch' format
specs_version = version[
: version.find(".")
] # version will always follow 'major.minor.patch' format
##########################################
## A. Machine Specs
@@ -97,25 +103,28 @@ def generate_machine_specs(args, sysinfo:dict=None):
version = path("/proc/version").read_text()
os_release = path("/etc/os-release").read_text()
cpu_model = search(r"^model name\s*: (.*?)$", cpuinfo)
sbios = path("/sys/class/dmi/id/bios_vendor").read_text().strip() + \
path("/sys/class/dmi/id/bios_version").read_text().strip()
sbios = (
path("/sys/class/dmi/id/bios_vendor").read_text().strip()
+ path("/sys/class/dmi/id/bios_version").read_text().strip()
)
linux_kernel_version = search(r"version (\S*)", version)
amd_gpu_kernel_version = "" #TODO: Extract amdgpu kernel version
amd_gpu_kernel_version = "" # TODO: Extract amdgpu kernel version
cpu_memory = search(r"MemTotal:\s*(\S*)", meminfo)
gpu_memory = "" #TODO: Extract gpu memory
gpu_memory = "" # TODO: Extract gpu memory
linux_distro = search(r'PRETTY_NAME="(.*?)"', os_release)
if linux_distro is None:
linux_distro = ""
rocm_version = get_rocm_ver().strip()
#FIXME: use device
vbios = search(
r"VBIOS version: (.*?)$", run(["rocm-smi", "-v"], exit_on_error=True))
# FIXME: use device
vbios = search(r"VBIOS version: (.*?)$", run(["rocm-smi", "-v"], exit_on_error=True))
compute_partition = search(
r"Compute Partition:\s*(\w+)", run(["rocm-smi", "--showcomputepartition"]))
r"Compute Partition:\s*(\w+)", run(["rocm-smi", "--showcomputepartition"])
)
if compute_partition is None:
compute_partition = "NA"
memory_partition = search(
r"Memory Partition:\s*(\w+)", run(["rocm-smi", "--showmemorypartition"]))
r"Memory Partition:\s*(\w+)", run(["rocm-smi", "--showmemorypartition"])
)
if memory_partition is None:
memory_partition = "NA"
@@ -126,29 +135,34 @@ def generate_machine_specs(args, sysinfo:dict=None):
rocminfo_full = run(["rocminfo"])
_rocminfo = rocminfo_full.split("\n")
gpu_arch, idx = detect_arch(_rocminfo)
_rocminfo = _rocminfo[idx + 1 :] # update rocminfo for target section
specs = MachineSpecs(version=specs_version,
timestamp=timestamp,
_rocminfo=_rocminfo,
hostname=hostname,
cpu_model=cpu_model,
sbios=sbios,
linux_kernel_version=linux_kernel_version,
amd_gpu_kernel_version=amd_gpu_kernel_version,
cpu_memory=cpu_memory,
gpu_memory=gpu_memory,
linux_distro=linux_distro,
rocm_version=rocm_version,
vbios=vbios,
compute_partition=compute_partition,
memory_partition=memory_partition,
gpu_arch=gpu_arch)
_rocminfo = _rocminfo[idx + 1 :] # update rocminfo for target section
specs = MachineSpecs(
version=specs_version,
timestamp=timestamp,
_rocminfo=_rocminfo,
hostname=hostname,
cpu_model=cpu_model,
sbios=sbios,
linux_kernel_version=linux_kernel_version,
amd_gpu_kernel_version=amd_gpu_kernel_version,
cpu_memory=cpu_memory,
gpu_memory=gpu_memory,
linux_distro=linux_distro,
rocm_version=rocm_version,
vbios=vbios,
compute_partition=compute_partition,
memory_partition=memory_partition,
gpu_arch=gpu_arch,
)
# Load above SoC specs via module import
try:
soc_module = importlib.import_module('omniperf_soc.soc_'+ specs.gpu_arch)
soc_module = importlib.import_module("omniperf_soc.soc_" + specs.gpu_arch)
except ModuleNotFoundError as e:
error("Arch %s marked as supported, but couldn't find class implementation %s." % (specs.gpu_arch, e))
soc_class = getattr(soc_module, specs.gpu_arch+'_soc')
error(
"Arch %s marked as supported, but couldn't find class implementation %s."
% (specs.gpu_arch, e)
)
soc_class = getattr(soc_module, specs.gpu_arch + "_soc")
soc_obj = soc_class(args, specs)
# Update arch specific specs
specs.total_l2_chan: str = total_l2_banks(
@@ -157,6 +171,7 @@ def generate_machine_specs(args, sysinfo:dict=None):
specs.hbm_bw: str = str(int(specs.max_mclk) / 1000 * 32 * specs.get_hbm_channels())
return specs
@dataclass(kw_only=True)
class MachineSpecs:
##########################################
@@ -168,66 +183,135 @@ class MachineSpecs:
# _are_ included in profiling/analysis, so we mark them as 'optional'
# in the metadata to avoid erroring out on missing fields on
# serialization
workload_name: str = field(default=None, metadata={
'doc': 'The name of the workload data was collected for.',
'name': 'Workload Name',
'optional': True})
command: str = field(default=None, metadata={
'doc': 'The command the workload was executed with.',
'name': 'Command',
'optional': True})
ip_blocks: str = field(default=None, metadata={
'doc': 'The hardware blocks profiling information was collected for.',
'name': 'IP Blocks',
'optional': True
})
timestamp: str = field(default=None, metadata={
'doc': 'The time (in local system time) when data was collected',
'name': 'Timestamp'})
version: str = field(default=None, metadata={
'doc': 'The version of the machine specification file format.',
'name': 'MachineSpecs Version',
'intable': False})
timestamp: str = field(default=None, metadata={
'doc': 'The time (in local system time) when data was collected',
'name': 'Timestamp'})
workload_name: str = field(
default=None,
metadata={
"doc": "The name of the workload data was collected for.",
"name": "Workload Name",
"optional": True,
},
)
command: str = field(
default=None,
metadata={
"doc": "The command the workload was executed with.",
"name": "Command",
"optional": True,
},
)
ip_blocks: str = field(
default=None,
metadata={
"doc": "The hardware blocks profiling information was collected for.",
"name": "IP Blocks",
"optional": True,
},
)
timestamp: str = field(
default=None,
metadata={
"doc": "The time (in local system time) when data was collected",
"name": "Timestamp",
},
)
version: str = field(
default=None,
metadata={
"doc": "The version of the machine specification file format.",
"name": "MachineSpecs Version",
"intable": False,
},
)
timestamp: str = field(
default=None,
metadata={
"doc": "The time (in local system time) when data was collected",
"name": "Timestamp",
},
)
_rocminfo: list = field(default=None)
##########################################
## A. Machine Specs
##########################################
hostname: str = field(default=None, metadata={'doc': 'The hostname of the machine.',
'name': 'Hostname'})
cpu_model: str = field(default=None, metadata={'doc': 'The model name of the CPU used.',
'name': 'CPU Model'})
sbios: str = field(default=None, metadata={'doc': 'The system management bios version and vendor.',
'name': "SBIOS"})
linux_distro: str = field(default=None, metadata={'doc': 'The Linux distribution installed on the machine.',
'name': 'Linux Distribution'})
linux_kernel_version: str = field(default=None,
metadata={'doc': 'The Linux kernel version running on the machine.',
'name': 'Linux Kernel Version'})
amd_gpu_kernel_version: str = field(default=None, metadata={
'doc': '[RESERVED] The version of the AMDGPU driver installed on the machine. Unimplemented.',
'name': "AMD GPU Kernel Version"})
cpu_memory: str = field(default=None, metadata={
'doc': 'The total amount of memory available to the CPU.', 'unit': 'KB',
'name': 'CPU Memory'})
gpu_memory: str = field(default=None, metadata={
'doc': '[RESERVED] The total amount of memory available to accelerators/GPUs in the system. Unimplemented.',
'unit': 'KB',
'name': 'GPU Memory'})
rocm_version: str = field(default=None,
metadata={'doc': 'The ROCm version used during data-collection.',
'name': "ROCm Version"})
vbios: str = field(default=None,
metadata={'doc': 'The version of the accelerators/GPUs video bios in the system.',
'name': 'VBIOS'})
compute_partition: str = field(default=None,
metadata={'doc': 'The compute partitioning mode active on the accelerators/GPUs in the system (MI300 only).',
'name': 'Compute Partition'})
memory_partition: str = field(default=None,
metadata={'doc': 'The memory partitioning mode active on the accelerators/GPUs in the system (MI300 only).',
'name': 'Memory Partition'})
hostname: str = field(
default=None, metadata={"doc": "The hostname of the machine.", "name": "Hostname"}
)
cpu_model: str = field(
default=None,
metadata={"doc": "The model name of the CPU used.", "name": "CPU Model"},
)
sbios: str = field(
default=None,
metadata={
"doc": "The system management bios version and vendor.",
"name": "SBIOS",
},
)
linux_distro: str = field(
default=None,
metadata={
"doc": "The Linux distribution installed on the machine.",
"name": "Linux Distribution",
},
)
linux_kernel_version: str = field(
default=None,
metadata={
"doc": "The Linux kernel version running on the machine.",
"name": "Linux Kernel Version",
},
)
amd_gpu_kernel_version: str = field(
default=None,
metadata={
"doc": "[RESERVED] The version of the AMDGPU driver installed on the machine. Unimplemented.",
"name": "AMD GPU Kernel Version",
},
)
cpu_memory: str = field(
default=None,
metadata={
"doc": "The total amount of memory available to the CPU.",
"unit": "KB",
"name": "CPU Memory",
},
)
gpu_memory: str = field(
default=None,
metadata={
"doc": "[RESERVED] The total amount of memory available to accelerators/GPUs in the system. Unimplemented.",
"unit": "KB",
"name": "GPU Memory",
},
)
rocm_version: str = field(
default=None,
metadata={
"doc": "The ROCm version used during data-collection.",
"name": "ROCm Version",
},
)
vbios: str = field(
default=None,
metadata={
"doc": "The version of the accelerators/GPUs video bios in the system.",
"name": "VBIOS",
},
)
compute_partition: str = field(
default=None,
metadata={
"doc": "The compute partitioning mode active on the accelerators/GPUs in the system (MI300 only).",
"name": "Compute Partition",
},
)
memory_partition: str = field(
default=None,
metadata={
"doc": "The memory partitioning mode active on the accelerators/GPUs in the system (MI300 only).",
"name": "Memory Partition",
},
)
##########################################
## B. SoC Specs
@@ -235,93 +319,168 @@ class MachineSpecs:
gpu_model: str = field(
default=None,
metadata={
'doc': 'The product name of the accelerators/GPUs in the system.',
'name': 'GPU Model'})
"doc": "The product name of the accelerators/GPUs in the system.",
"name": "GPU Model",
},
)
gpu_arch: str = field(
default=None,
metadata={
'doc': 'The architecture name of the accelerators/GPUs in the system,\n'
'as used by (e.g.,) the AMDGPU backed of LLVM.',
'name': 'GPU Arch'})
gpu_l1: str = field(default=None, metadata={'doc':
"The size of the vL1D cache (per compute-unit) on the accelerators/GPUs in the system in KiB",
'name': 'GPU L1'})
gpu_l2: str = field(default=None, metadata={'doc':
"The size of the vL1D cache (per compute-unit) on the accelerators/GPUs in the system in KiB",
'name': 'GPU L2'})
cu_per_gpu: str = field(default=None, metadata={'doc':
"The total number of compute units per accelerator/GPU in the system. On systems with configurable\n"
"partitioning, (e.g., MI300) this is the total number of compute units in a partition.",
'name': 'CU per GPU'})
simd_per_cu: str = field(default=None, metadata={'doc':
"The number of SIMD processors in a compute unit for the accelerators/GPUs in the system.",
'name': 'SIMD per CU'})
se_per_gpu: str = field(default=None, metadata={'doc':
'The number of shader engines on the accelerators/GPUs in the system. On systems with configurable\n'
'partitioning, (e.g., MI300) this is the total number of shader engines in a partition.',
'name': 'SE per GPU'})
wave_size: str = field(default=None, metadata={'doc':
'The number work-items in a wavefront on the accelerators/GPUs in the system.',
'name': 'Wave Size'})
workgroup_max_size: str = field(default=None, metadata={'doc':
'The maximum number of work-items in a workgroup on the accelerators/GPUs in the system.',
'name': 'Workgroup Max Size'})
max_waves_per_cu: str = field(default=None, metadata={'doc':
'The maximum number of wavefronts that can be resident on a compute unit on the\n'
'accelerators/GPUs in the system',
'name': 'Max Waves per CU'})
max_sclk: str = field(default=None, metadata={'doc':
'The maximum engine (compute-unit) clock rate of the accelerators/GPUs in the system.',
'name': 'Max SCLK',
'unit': 'MHz'})
max_mclk: str = field(default=None, metadata={'doc':
'The maximum memory clock rate of the accelerators/GPUs in the system.',
'name': 'Max MCLK',
'unit': 'MHz'})
cur_sclk: str = field(default=None, metadata={'doc':
'[RESERVED] The current engine (compute unit) clock rate of the accelerators/GPUs in the system. Unused.',
'name': 'Cur SCLK',
'unit': 'MHz'})
cur_mclk: str = field(default=None, metadata={'doc':
'[RESERVED] The current memory clock rate of the accelerators/GPUs in the system. Unused.',
'name': 'Cur MCLK',
'unit': 'MHz'})
_l2_banks: str = None # NB: This only used in flatten_tcc_info_across_hbm_stacks()
total_l2_chan: str = field(default=None, metadata={
'doc': 'The maximum number of L2 cache channels on the accelerators/GPUs in the system. On systems with\n'
'configurable partitioning, (e.g., MI300) this is the total number of L2 cache channels in a partition.',
'name': 'Total L2 Channels'})
lds_banks_per_cu: str = field(default=None, metadata={'doc':
'The number of banks in the LDS for a compute unit on the accelerators/GPUs in the system.',
'name': 'LDS Banks per CU'})
sqc_per_gpu: str = field(default=None, metadata={
'doc': 'The number of L1I/sL1D caches on the accelerators/GPUs in the system. On systems with\n'
'configurable partitioning, (e.g., MI300) this is the total number of L1I/sL1D caches in a partition.',
'name': 'SQC per GPU'})
pipes_per_gpu: str = field(default=None, metadata={
'doc': 'The number of scheduler-pipes on the accelerators/GPUs in the system.',
'name': 'Pipes per GPU'})
hbm_bw: str = field(default=None, metadata={
'doc': 'The peak theoretical HBM bandwidth for the accelerators/GPUs in the system. On systems with\n'
'configurable partitioning, (e.g., MI300) this is the peak theoretical HBM bandwidth for a partition.',
'name': 'HBM BW',
'unit': 'MB/s'})
num_xcd: str = field(default=None, metadata={
'doc': 'The total number of accelerator complex dies in a compute partition on the accelerators/GPUs in the\n'
'system. For accelerators without partitioning (i.e., pre-MI300), this is considered to be one.',
'name': 'Num XCDs',
'unit': 'MB/s'})
"doc": "The architecture name of the accelerators/GPUs in the system,\n"
"as used by (e.g.,) the AMDGPU backed of LLVM.",
"name": "GPU Arch",
},
)
gpu_l1: str = field(
default=None,
metadata={
"doc": "The size of the vL1D cache (per compute-unit) on the accelerators/GPUs in the system in KiB",
"name": "GPU L1",
},
)
gpu_l2: str = field(
default=None,
metadata={
"doc": "The size of the vL1D cache (per compute-unit) on the accelerators/GPUs in the system in KiB",
"name": "GPU L2",
},
)
cu_per_gpu: str = field(
default=None,
metadata={
"doc": "The total number of compute units per accelerator/GPU in the system. On systems with configurable\n"
"partitioning, (e.g., MI300) this is the total number of compute units in a partition.",
"name": "CU per GPU",
},
)
simd_per_cu: str = field(
default=None,
metadata={
"doc": "The number of SIMD processors in a compute unit for the accelerators/GPUs in the system.",
"name": "SIMD per CU",
},
)
se_per_gpu: str = field(
default=None,
metadata={
"doc": "The number of shader engines on the accelerators/GPUs in the system. On systems with configurable\n"
"partitioning, (e.g., MI300) this is the total number of shader engines in a partition.",
"name": "SE per GPU",
},
)
wave_size: str = field(
default=None,
metadata={
"doc": "The number work-items in a wavefront on the accelerators/GPUs in the system.",
"name": "Wave Size",
},
)
workgroup_max_size: str = field(
default=None,
metadata={
"doc": "The maximum number of work-items in a workgroup on the accelerators/GPUs in the system.",
"name": "Workgroup Max Size",
},
)
max_waves_per_cu: str = field(
default=None,
metadata={
"doc": "The maximum number of wavefronts that can be resident on a compute unit on the\n"
"accelerators/GPUs in the system",
"name": "Max Waves per CU",
},
)
max_sclk: str = field(
default=None,
metadata={
"doc": "The maximum engine (compute-unit) clock rate of the accelerators/GPUs in the system.",
"name": "Max SCLK",
"unit": "MHz",
},
)
max_mclk: str = field(
default=None,
metadata={
"doc": "The maximum memory clock rate of the accelerators/GPUs in the system.",
"name": "Max MCLK",
"unit": "MHz",
},
)
cur_sclk: str = field(
default=None,
metadata={
"doc": "[RESERVED] The current engine (compute unit) clock rate of the accelerators/GPUs in the system. Unused.",
"name": "Cur SCLK",
"unit": "MHz",
},
)
cur_mclk: str = field(
default=None,
metadata={
"doc": "[RESERVED] The current memory clock rate of the accelerators/GPUs in the system. Unused.",
"name": "Cur MCLK",
"unit": "MHz",
},
)
_l2_banks: str = None # NB: This only used in flatten_tcc_info_across_hbm_stacks()
total_l2_chan: str = field(
default=None,
metadata={
"doc": "The maximum number of L2 cache channels on the accelerators/GPUs in the system. On systems with\n"
"configurable partitioning, (e.g., MI300) this is the total number of L2 cache channels in a partition.",
"name": "Total L2 Channels",
},
)
lds_banks_per_cu: str = field(
default=None,
metadata={
"doc": "The number of banks in the LDS for a compute unit on the accelerators/GPUs in the system.",
"name": "LDS Banks per CU",
},
)
sqc_per_gpu: str = field(
default=None,
metadata={
"doc": "The number of L1I/sL1D caches on the accelerators/GPUs in the system. On systems with\n"
"configurable partitioning, (e.g., MI300) this is the total number of L1I/sL1D caches in a partition.",
"name": "SQC per GPU",
},
)
pipes_per_gpu: str = field(
default=None,
metadata={
"doc": "The number of scheduler-pipes on the accelerators/GPUs in the system.",
"name": "Pipes per GPU",
},
)
hbm_bw: str = field(
default=None,
metadata={
"doc": "The peak theoretical HBM bandwidth for the accelerators/GPUs in the system. On systems with\n"
"configurable partitioning, (e.g., MI300) this is the peak theoretical HBM bandwidth for a partition.",
"name": "HBM BW",
"unit": "MB/s",
},
)
num_xcd: str = field(
default=None,
metadata={
"doc": "The total number of accelerator complex dies in a compute partition on the accelerators/GPUs in the\n"
"system. For accelerators without partitioning (i.e., pre-MI300), this is considered to be one.",
"name": "Num XCDs",
"unit": "MB/s",
},
)
def get_hbm_channels(self):
hbmchannels = int(self.total_l2_chan)
if (
self.gpu_model.lower() == "mi300a_a0"
or self.gpu_model.lower() == "mi300a_a1"
self.gpu_model.lower() == "mi300a_a0" or self.gpu_model.lower() == "mi300a_a1"
) and self.memory_partition.lower() == "nps1":
# we have an extra 32 channels for the CCD
hbmchannels += 32
return hbmchannels
def get_class_members(self):
all_populated = True
data = {}
@@ -332,20 +491,25 @@ class MachineSpecs:
value = getattr(self, name)
if value is None:
# check if we've marked it optional
if field.metadata and 'optional' in field.metadata and field.metadata['optional']:
if (
field.metadata
and "optional" in field.metadata
and field.metadata["optional"]
):
pass
else:
#TODO: use proper logging function when that's merged
# TODO: use proper logging function when that's merged
logging.warning(
f"WARNING: Incomplete class definition for {self.gpu_arch}. "
f"Expecting populated {name} but detected None.")
f"Expecting populated {name} but detected None."
)
all_populated = False
data[name] = value
if not all_populated:
error("Missing specs fields for %s" % self.gpu_arch)
return pd.DataFrame(data, index=[0])
def __repr__(self):
topstr = "Machine Specifications: describing the state of the machine that Omniperf data was collected on.\n"
data = []
@@ -356,32 +520,30 @@ class MachineSpecs:
value = getattr(self, name)
if field.metadata:
# check out of table before any re-naming for pretty-printing
if 'intable' in field.metadata and not field.metadata['intable']:
if name == 'version':
topstr += f'Output version: {value}\n'
if "intable" in field.metadata and not field.metadata["intable"]:
if name == "version":
topstr += f"Output version: {value}\n"
else:
error(f"Unknown out of table printing field: {name}")
continue
if 'name' in field.metadata:
name = field.metadata['name']
if 'unit' in field.metadata:
_data['Unit'] = field.metadata['unit']
if 'doc' in field.metadata:
_data['Description'] = field.metadata['doc']
_data['Spec'] = name
_data['Value'] = value
if "name" in field.metadata:
name = field.metadata["name"]
if "unit" in field.metadata:
_data["Unit"] = field.metadata["unit"]
if "doc" in field.metadata:
_data["Description"] = field.metadata["doc"]
_data["Spec"] = name
_data["Value"] = value
data.append(_data)
df = pd.DataFrame(data)
columns = ['Spec', 'Value']
if 'Description' in df.columns:
columns += ['Description']
if 'Unit' in df.columns:
columns += ['Unit']
columns = ["Spec", "Value"]
if "Description" in df.columns:
columns += ["Description"]
if "Unit" in df.columns:
columns += ["Unit"]
df = df[columns]
df = df.fillna('')
return (
topstr +
get_table_string(df, transpose=False, decimal=2))
df = df.fillna("")
return topstr + get_table_string(df, transpose=False, decimal=2)
def get_rocm_ver():
@@ -403,14 +565,20 @@ def get_rocm_ver():
rocm_ver = ROCM_VER_USER
else:
_rocm_path = os.getenv("ROCM_PATH", "/opt/rocm")
error("Unable to detect a complete local ROCm installation.\nThe expected %s/.info/ versioning directory is missing. Please ensure you have valid ROCm installation." % _rocm_path)
error(
"Unable to detect a complete local ROCm installation.\nThe expected %s/.info/ versioning directory is missing. Please ensure you have valid ROCm installation."
% _rocm_path
)
return rocm_ver
def run(cmd,exit_on_error=False):
def run(cmd, exit_on_error=False):
try:
p = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
except FileNotFoundError as e:
error(f"Unable to parse specs. Can't find ROCm asset: {e.filename}\nTry passing a path to an existing workload results in 'analyze' mode.")
error(
f"Unable to parse specs. Can't find ROCm asset: {e.filename}\nTry passing a path to an existing workload results in 'analyze' mode."
)
if exit_on_error:
if cmd[0] == "rocm-smi":
@@ -434,35 +602,28 @@ def total_l2_banks(archname, L2Banks, memory_partition):
# Fixme: support all supported partitioning mode
# Fixme: "name" is a bad name!
totalL2Banks = L2Banks
if (
archname.lower() == "mi300a_a0"
or archname.lower() == "mi300a_a1"
):
totalL2Banks = L2Banks * get_hbm_stack_num(
archname, memory_partition)
elif (
archname.lower() == "mi300x_a0"
or archname.lower() == "mi300x_a1"
):
totalL2Banks = L2Banks * get_hbm_stack_num(
archname, memory_partition)
if archname.lower() == "mi300a_a0" or archname.lower() == "mi300a_a1":
totalL2Banks = L2Banks * get_hbm_stack_num(archname, memory_partition)
elif archname.lower() == "mi300x_a0" or archname.lower() == "mi300x_a1":
totalL2Banks = L2Banks * get_hbm_stack_num(archname, memory_partition)
return str(totalL2Banks)
def total_sqc(archname, numCUs, numSEs):
cu_per_se = float(numCUs) / float(numSEs)
sq_per_se = cu_per_se / 2
if archname.lower() in ['mi50', 'mi100']:
if archname.lower() in ["mi50", "mi100"]:
sq_per_se = cu_per_se / 3
sq_per_se = ceil(sq_per_se)
return int(sq_per_se) * int(numSEs)
def total_xcds(archname, compute_partition):
# check MI300 has a valid compute partition
mi300a_archs = ["mi300a_a0", "mi300a_a1"]
mi300x_archs = ["mi300x_a0", "mi300x_a1"]
if archname.lower() in mi300a_archs + mi300x_archs \
and compute_partition == "NA":
error("Invalid compute partition found for {}".format(archname))
if archname.lower() in mi300a_archs + mi300x_archs and compute_partition == "NA":
error("Invalid compute partition found for {}".format(archname))
if archname.lower() not in mi300a_archs + mi300x_archs:
return 1
# from the whitepaper
@@ -484,8 +645,11 @@ def total_xcds(archname, compute_partition):
if compute_partition.lower() == "cpx":
if archname.lower() in mi300x_archs:
return 2
error("Unknown compute partition / arch found for {} / {}".format(
compute_partition, archname))
error(
"Unknown compute partition / arch found for {} / {}".format(
compute_partition, archname
)
)
if __name__ == "__main__":
+10 -6
파일 보기
@@ -51,10 +51,12 @@ def string_multiple_lines(source, width, max_rows):
def get_table_string(df, transpose=False, decimal=2):
return tabulate(df.transpose() if transpose else df,
headers="keys",
tablefmt="fancy_grid",
floatfmt="." + str(decimal) + "f")
return tabulate(
df.transpose() if transpose else df,
headers="keys",
tablefmt="fancy_grid",
floatfmt="." + str(decimal) + "f",
)
def show_all(args, runs, archConfigs, output):
@@ -221,9 +223,11 @@ def show_all(args, runs, archConfigs, output):
# df when load it, because we need those items in column.
# For metric_table, we only need to show the data in column
# fash for now.
transpose = (type != "raw_csv_table"
transpose = (
type != "raw_csv_table"
and "columnwise" in table_config
and table_config["columnwise"] == True)
and table_config["columnwise"] == True
)
ss += (
get_table_string(df, transpose=transpose, decimal=args.decimal)
+ "\n"
+21 -13
파일 보기
@@ -198,10 +198,11 @@ def capture_subprocess_output(subprocess_args, new_env=None):
return (success, output)
def run_prof(fname, profiler_options, workload_dir, mspec):
fbase = os.path.splitext(os.path.basename(fname))[0]
logging.debug("pmc file: %s" % str(os.path.basename(fname)))
# standard rocprof options
@@ -210,7 +211,12 @@ def run_prof(fname, profiler_options, workload_dir, mspec):
# set required env var for mi300
new_env = None
if (mspec.gpu_model.lower() == "mi300x_a0" or mspec.gpu_model.lower() == "mi300x_a1" or mspec.gpu_model.lower() == "mi300a_a0" or mspec.gpu_model.lower() == "mi300a_a1") and (
if (
mspec.gpu_model.lower() == "mi300x_a0"
or mspec.gpu_model.lower() == "mi300x_a1"
or mspec.gpu_model.lower() == "mi300a_a0"
or mspec.gpu_model.lower() == "mi300a_a1"
) and (
os.path.basename(fname) == "pmc_perf_13.txt"
or os.path.basename(fname) == "pmc_perf_14.txt"
or os.path.basename(fname) == "pmc_perf_15.txt"
@@ -233,9 +239,7 @@ def run_prof(fname, profiler_options, workload_dir, mspec):
# flatten tcc for applicable mi300 input
f = path(workload_dir + "/out/pmc_1/results_" + fbase + ".csv")
hbm_stack_num = get_hbm_stack_num(mspec.gpu_model, mspec.memory_partition)
df = flatten_tcc_info_across_hbm_stacks(
f, hbm_stack_num, int(mspec._l2_banks)
)
df = flatten_tcc_info_across_hbm_stacks(f, hbm_stack_num, int(mspec._l2_banks))
df.to_csv(f, index=False)
if os.path.exists(workload_dir + "/out"):
@@ -295,13 +299,16 @@ def replace_timestamps(workload_dir):
)
logging.warning(warning + "\n")
def gen_sysinfo(workload_name, workload_dir, ip_blocks, app_cmd, skip_roof, roof_only, mspec):
def gen_sysinfo(
workload_name, workload_dir, ip_blocks, app_cmd, skip_roof, roof_only, mspec
):
df = mspec.get_class_members()
# Append workload information to machine specs
df['command'] = app_cmd
df['workload_name'] = workload_name
df["command"] = app_cmd
df["workload_name"] = workload_name
blocks = []
if ip_blocks == None:
t = ["SQ", "LDS", "SQC", "TA", "TD", "TCP", "TCC", "SPI", "CPC", "CPF"]
@@ -310,11 +317,12 @@ def gen_sysinfo(workload_name, workload_dir, ip_blocks, app_cmd, skip_roof, roof
blocks += ip_blocks
if mspec.gpu_arch == "gfx90a" and (not skip_roof):
blocks.append("roofline")
df['ip_blocks'] = "|".join(blocks)
df["ip_blocks"] = "|".join(blocks)
# Save csv
df.to_csv(workload_dir + "/" + "sysinfo.csv", index=False)
def detect_roofline(mspec):
rocm_ver = mspec.rocm_version[:1]
@@ -380,9 +388,9 @@ def run_rocscope(args, fname):
logging.error(result.stderr.decode("ascii"))
sys.exit(1)
def mibench(args, mspec):
"""Run roofline microbenchmark to generate peak BW and FLOP measurements.
"""
"""Run roofline microbenchmark to generate peak BW and FLOP measurements."""
logging.info("[roofline] No roofline data found. Generating...")
distro_map = {"platform:el8": "rhel8", "15.3": "sle15sp3", "20.04": "ubuntu20_04"}