Update csv headers, and SystemExit codes

Signed-off-by: JoseSantosAMD <Jose.Santos@amd.com>
This commit is contained in:
JoseSantosAMD
2024-02-02 20:04:11 -06:00
committed by Karl W. Schulz
parent 7556316f02
commit 717a21cf84
2842 changed files with 5653 additions and 3927 deletions
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBuffer.kd",0,0,0,935069,935074,33554432,256,0,0,8,32,6464,0x0,0x7fb1e8e04180,504417,504417,524288,6291456,791548,101536232,12076606776836515,12076607020949616,12076607021273775,12076607021386317
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,935069,935074,32768,256,0,0,24,24,12480,0x0,0x7fb1e8e35100,27457,27457,512,8192,8659,1112808,12076607036018836,12076607036328752,12076607036334672,12076607036343359
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,4,935069,935074,4194304,256,0,0,24,24,12928,0x7fb2f49b3900,0x7fb1e8e35140,221451,221451,65536,917504,137700,17627108,12076607036410164,12076607036634350,12076607036769870,12076607036773860
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBuffer.kd 0 0 0 935069 935074 33554432 256 0 0 8 32 6464 0x0 0x7fb1e8e04180 504417 504417 524288 6291456 791548 101536232 12076606776836515 12076607020949616 12076607021273775 12076607021386317
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 935069 935074 32768 256 0 0 24 24 12480 0x0 0x7fb1e8e35100 27457 27457 512 8192 8659 1112808 12076607036018836 12076607036328752 12076607036334672 12076607036343359
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 4 935069 935074 4194304 256 0 0 24 24 12928 0x7fb2f49b3900 0x7fb1e8e35140 221451 221451 65536 917504 137700 17627108 12076607036410164 12076607036634350 12076607036769870 12076607036773860
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBuffer.kd",0,0,0,936874,936879,33554432,256,0,0,8,32,6464,0x0,0x7f608a404180,0,0,0,12076632988700397,12076633233142956,12076633233467755,12076633233577518
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,936874,936879,32768,256,0,0,24,24,12480,0x0,0x7f608a435100,0,0,0,12076633248282761,12076633248580944,12076633248587504,12076633248592868
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,4,936874,936879,4194304,256,0,0,24,24,12928,0x7f61960b3900,0x7f608a435140,0,0,0,12076633248655084,12076633248875663,12076633249007503,12076633249011156
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBuffer.kd 0 0 0 936874 936879 33554432 256 0 0 8 32 6464 0x0 0x7f608a404180 0 0 0 12076632988700397 12076633233142956 12076633233467755 12076633233577518
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 936874 936879 32768 256 0 0 24 24 12480 0x0 0x7f608a435100 0 0 0 12076633248282761 12076633248580944 12076633248587504 12076633248592868
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 4 936874 936879 4194304 256 0 0 24 24 12928 0x7f61960b3900 0x7f608a435140 0 0 0 12076633248655084 12076633248875663 12076633249007503 12076633249011156
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBuffer.kd",0,0,0,936496,936501,33554432,256,0,0,8,32,6464,0x0,0x7f17e8404180,4194304,3122666,398904176,12076630480743351,12076630718390515,12076630718714354,12076630718822006
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,936496,936501,32768,256,0,0,24,24,12480,0x0,0x7f17e8435100,512,21524,2754256,12076630733442442,12076630733749898,12076630733756458,12076630733761836
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,4,936496,936501,4194304,256,0,0,24,24,12928,0x7f18f4030900,0x7f17e8435140,65536,146806,18775720,12076630733826656,12076630734045577,12076630734179017,12076630734183330
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBuffer.kd 0 0 0 936496 936501 33554432 256 0 0 8 32 6464 0x0 0x7f17e8404180 4194304 3122666 398904176 12076630480743351 12076630718390515 12076630718714354 12076630718822006
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 936496 936501 32768 256 0 0 24 24 12480 0x0 0x7f17e8435100 512 21524 2754256 12076630733442442 12076630733749898 12076630733756458 12076630733761836
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 4 936496 936501 4194304 256 0 0 24 24 12928 0x7f18f4030900 0x7f17e8435140 65536 146806 18775720 12076630733826656 12076630734045577 12076630734179017 12076630734183330
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBuffer.kd",0,0,0,936686,936691,33554432,256,0,0,8,32,6464,0x0,0x7fa3b8c04180,1048576,11096456,1419911252,12076631726864889,12076631974434106,12076631974755065,12076631974864438
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,936686,936691,32768,256,0,0,24,24,12480,0x0,0x7fa3b8c35100,4096,107784,13788356,12076631989518086,12076631989806097,12076631989812497,12076631989817693
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,4,936686,936691,4194304,256,0,0,24,24,12928,0x7fa4c491a900,0x7fa3b8c35140,524288,10667896,1365491840,12076631989877063,12076631990100976,12076631990238896,12076631990242964
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBuffer.kd 0 0 0 936686 936691 33554432 256 0 0 8 32 6464 0x0 0x7fa3b8c04180 1048576 11096456 1419911252 12076631726864889 12076631974434106 12076631974755065 12076631974864438
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 936686 936691 32768 256 0 0 24 24 12480 0x0 0x7fa3b8c35100 4096 107784 13788356 12076631989518086 12076631989806097 12076631989812497 12076631989817693
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 4 936686 936691 4194304 256 0 0 24 24 12928 0x7fa4c491a900 0x7fa3b8c35140 524288 10667896 1365491840 12076631989877063 12076631990100976 12076631990238896 12076631990242964
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBuffer.kd",0,0,0,937062,937067,33554432,256,0,0,8,32,6464,0x0,0x7f30a5604180,499674,499674,16219,3997400,524288,371768834,3801967,0,1501859028,12076634246773666,12076634492293789,12076634492615709,12076634492722711
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,937062,937067,32768,256,0,0,24,24,12480,0x0,0x7f30a5635100,27467,27467,20214,219744,512,1126759,75366,0,4521000,12076634507281913,12076634507609674,12076634507615914,12076634507624380
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,4,937062,937067,4194304,256,0,0,24,24,12928,0x7f31b1191900,0x7f30a5635140,217853,217853,21633,1742832,65536,133671007,1576099,0,536498696,12076634507695903,12076634507934314,12076634508067113,12076634508071402
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBuffer.kd 0 0 0 937062 937067 33554432 256 0 0 8 32 6464 0x0 0x7f30a5604180 499674 499674 16219 3997400 524288 371768834 3801967 0 1501859028 12076634246773666 12076634492293789 12076634492615709 12076634492722711
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 937062 937067 32768 256 0 0 24 24 12480 0x0 0x7f30a5635100 27467 27467 20214 219744 512 1126759 75366 0 4521000 12076634507281913 12076634507609674 12076634507615914 12076634507624380
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 4 937062 937067 4194304 256 0 0 24 24 12928 0x7f31b1191900 0x7f30a5635140 217853 217853 21633 1742832 65536 133671007 1576099 0 536498696 12076634507695903 12076634507934314 12076634508067113 12076634508071402
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id
Dispatch_ID,Kernel_Name,GPU_ID
0,__amd_rocclr_fillBuffer.kd,0
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID
2 0 __amd_rocclr_fillBuffer.kd 0
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0
File diff suppressed because one or more lines are too long
@@ -1,4 +1,4 @@
KernelName,Count,Sum(ns),Mean(ns),Median(ns),Pct
Kernel_Name,Count,Sum(ns),Mean(ns),Median(ns),Pct
"void benchmark_func<HIP_vector_type<float, 2u>, 256, 8u, 512u>(HIP_vector_type<float, 2u>, HIP_vector_type<float, 2u>*) [clone .kd]",1,6059957.0,6059957.0,6059957.0,9.169486064060209
"void benchmark_func<int, 256, 8u, 512u>(int, int*) [clone .kd]",1,4525889.0,4525889.0,4525889.0,6.848245971544582
"void benchmark_func<HIP_vector_type<float, 2u>, 256, 8u, 256u>(HIP_vector_type<float, 2u>, HIP_vector_type<float, 2u>*) [clone .kd]",1,3056459.0,3056459.0,3056459.0,4.624811398145466
1 KernelName Kernel_Name Count Sum(ns) Mean(ns) Median(ns) Pct
2 void benchmark_func<HIP_vector_type<float, 2u>, 256, 8u, 512u>(HIP_vector_type<float, 2u>, HIP_vector_type<float, 2u>*) [clone .kd] 1 6059957.0 6059957.0 6059957.0 9.169486064060209
3 void benchmark_func<int, 256, 8u, 512u>(int, int*) [clone .kd] 1 4525889.0 4525889.0 4525889.0 6.848245971544582
4 void benchmark_func<HIP_vector_type<float, 2u>, 256, 8u, 256u>(HIP_vector_type<float, 2u>, HIP_vector_type<float, 2u>*) [clone .kd] 1 3056459.0 3056459.0 3056459.0 4.624811398145466
@@ -1,7 +1,7 @@
Metric,Count,Unit
VALU - Vector,,Instr per wave
VMEM,,Instr per wave
LDS,0.0,Instr per wave
LDS_Per_Workgroup,0.0,Instr per wave
VALU - MFMA,,Instr per wave
SALU,46.862275449101794,Instr per wave
SMEM,1.8083832335329342,Instr per wave
1 Metric Count Unit
2 VALU - Vector Instr per wave
3 VMEM Instr per wave
4 LDS LDS_Per_Workgroup 0.0 Instr per wave
5 VALU - MFMA Instr per wave
6 SALU 46.862275449101794 Instr per wave
7 SMEM 1.8083832335329342 Instr per wave
@@ -23,4 +23,4 @@ gfx908
mi100
48
1228.8
SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
SQ|LDS_Per_Workgroup|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
1 Info
23 mi100
24 48
25 1228.8
26 SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF SQ|LDS_Per_Workgroup|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
@@ -1,12 +1,12 @@
Metric,Avg,Min,Max,Unit
Wave Cycles,33617.72709436474,2860.667678833008,547916.0441894531,Cycles/wave
LDS Instrs,0.0,0.0,0.0,Instr per wave
LDS_Per_Workgroup Instrs,0.0,0.0,0.0,Instr per wave
Bandwidth,0.0,0.0,0.0,Bytes per wave
Bank Conficts/Access,,,,Conflicts/access
Index Accesses,0.0,0.0,0.0,Cycles per wave
Dispatch_ID Accesses,0.0,0.0,0.0,Cycles per wave
Atomic Cycles,0.0,0.0,0.0,Cycles per wave
Bank Conflict,0.0,0.0,0.0,Cycles per wave
Addr Conflict,0.0,0.0,0.0,Cycles per wave
Unaligned Stall,0.0,0.0,0.0,Cycles per wave
Mem Violations,0.0,0.0,0.0, per wave
LDS Latency,,,,Cycles
LDS_Per_Workgroup Latency,,,,Cycles
1 Metric Avg Min Max Unit
2 Wave Cycles 33617.72709436474 2860.667678833008 547916.0441894531 Cycles/wave
3 LDS Instrs LDS_Per_Workgroup Instrs 0.0 0.0 0.0 Instr per wave
4 Bandwidth 0.0 0.0 0.0 Bytes per wave
5 Bank Conficts/Access Conflicts/access
6 Index Accesses Dispatch_ID Accesses 0.0 0.0 0.0 Cycles per wave
7 Atomic Cycles 0.0 0.0 0.0 Cycles per wave
8 Bank Conflict 0.0 0.0 0.0 Cycles per wave
9 Addr Conflict 0.0 0.0 0.0 Cycles per wave
10 Unaligned Stall 0.0 0.0 0.0 Cycles per wave
11 Mem Violations 0.0 0.0 0.0 per wave
12 LDS Latency LDS_Per_Workgroup Latency Cycles
@@ -6,16 +6,16 @@ SMEM,2.0,smem_
VALU,735.0,valu_
MFMA,,mfma_
VMEM,8.0,vmem_
LDS,0.0,lds_
LDS_Per_Workgroup,0.0,lds_
GWS,0.0,gws_
BR,22.0,br_
VGPR,24.0,vgpr_
SGPR,24.0,sgpr_
LDS Allocation,0.0,lds_alloc_
LDS_Per_Workgroup Allocation,0.0,lds_alloc_
Scratch Allocation,0.0,scratch_alloc_
Wavefronts,67894.0,wavefronts_
Workgroups,16973.0,workgroups_
LDS Req,0.0,lds_req_
LDS_Per_Workgroup Req,0.0,lds_req_
IL1 Fetch,163.0,il1_fetch_
IL1 Hit,100.0,il1_hit_
IL1_L2 Rd,0.0,il1_l2_req_
@@ -46,10 +46,10 @@ Fabric_L2 Wr,0.0,l2_fabric_wr_
Fabric_l2 Atomic,0.0,l2_fabric_atom_
HBM Rd,44.0,hbm_rd_
HBM Wr,0.0,hbm_wr_
LDS Util,0.0,lds_util_
LDS_Per_Workgroup Util,0.0,lds_util_
VL1 Coalesce,70.0,vl1_coales_
VL1 Stall,31.0,vl1_stall_
LDS Lat,,lds_lat_
LDS_Per_Workgroup Lat,,lds_lat_
vL1D Lat,,sl1_lat_
IL1 Lat,,il1_lat_
Wave Occupancy,34.0,wave_occ_
1 Metric Value Alias
6 VALU 735.0 valu_
7 MFMA mfma_
8 VMEM 8.0 vmem_
9 LDS LDS_Per_Workgroup 0.0 lds_
10 GWS 0.0 gws_
11 BR 22.0 br_
12 VGPR 24.0 vgpr_
13 SGPR 24.0 sgpr_
14 LDS Allocation LDS_Per_Workgroup Allocation 0.0 lds_alloc_
15 Scratch Allocation 0.0 scratch_alloc_
16 Wavefronts 67894.0 wavefronts_
17 Workgroups 16973.0 workgroups_
18 LDS Req LDS_Per_Workgroup Req 0.0 lds_req_
19 IL1 Fetch 163.0 il1_fetch_
20 IL1 Hit 100.0 il1_hit_
21 IL1_L2 Rd 0.0 il1_l2_req_
46 Fabric_l2 Atomic 0.0 l2_fabric_atom_
47 HBM Rd 44.0 hbm_rd_
48 HBM Wr 0.0 hbm_wr_
49 LDS Util LDS_Per_Workgroup Util 0.0 lds_util_
50 VL1 Coalesce 70.0 vl1_coales_
51 VL1 Stall 31.0 vl1_stall_
52 LDS Lat LDS_Per_Workgroup Lat lds_lat_
53 vL1D Lat sl1_lat_
54 IL1 Lat il1_lat_
55 Wave Occupancy 34.0 wave_occ_
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id
Dispatch_ID,Kernel_Name,GPU_ID
0,__amd_rocclr_fillBuffer.kd,0
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID
2 0 __amd_rocclr_fillBuffer.kd 0
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0
@@ -12,8 +12,8 @@ VALU Util,59.71163511528944,Pct,100,59.71163511528944
MFMA Util,,Pct,100,
VALU Active Threads/Wave,63.967840400497316,Threads,64,99.94975062577706
IPC - Issue,0.8437262750969097,Instr/cycle,5,16.874525501938194
LDS BW,0.0,Gb/sec,23070.72,0.0
LDS Bank Conflict,,Conflicts/access,32,
LDS_Per_Workgroup BW,0.0,Gb/sec,23070.72,0.0
LDS_Per_Workgroup Bank Conflict,,Conflicts/access,32,
Instr Cache Hit Rate,99.99318745374036,Pct,100,99.99318745374036
Instr Cache BW,1406.6059705869952,Gb/s,4614.144,30.484656971845595
Scalar L1D Cache Hit Rate,99.35620448519533,Pct,100,99.35620448519533
1 Metric Value Unit Peak PoP
12 MFMA Util Pct 100
13 VALU Active Threads/Wave 63.967840400497316 Threads 64 99.94975062577706
14 IPC - Issue 0.8437262750969097 Instr/cycle 5 16.874525501938194
15 LDS BW LDS_Per_Workgroup BW 0.0 Gb/sec 23070.72 0.0
16 LDS Bank Conflict LDS_Per_Workgroup Bank Conflict Conflicts/access 32
17 Instr Cache Hit Rate 99.99318745374036 Pct 100 99.99318745374036
18 Instr Cache BW 1406.6059705869952 Gb/s 4614.144 30.484656971845595
19 Scalar L1D Cache Hit Rate 99.35620448519533 Pct 100 99.35620448519533
@@ -6,7 +6,7 @@ Scratch Stall,0.0,0,0,Cycles
Insufficient SIMD Waveslots,64892.16766467066,0,774244,Simd
Insufficient SIMD VGPRs,527589.6886227545,0,30743007,Simd
Insufficient SIMD SGPRs,0.0,0,0,Simd
Insufficient CU LDS,0.0,0,0,Cu
Insufficient CU LDS_Per_Workgroup,0.0,0,0,Cu
Insufficient CU Barries,0.0,0,0,Cu
Insufficient Bulky Resource,0.0,0,0,Cu
Reach CU Threadgroups Limit,0.0,0,0,Cycles
1 Metric Avg Min Max Unit
6 Insufficient SIMD Waveslots 64892.16766467066 0 774244 Simd
7 Insufficient SIMD VGPRs 527589.6886227545 0 30743007 Simd
8 Insufficient SIMD SGPRs 0.0 0 0 Simd
9 Insufficient CU LDS Insufficient CU LDS_Per_Workgroup 0.0 0 0 Cu
10 Insufficient CU Barries 0.0 0 0 Cu
11 Insufficient Bulky Resource 0.0 0 0 Cu
12 Reach CU Threadgroups Limit 0.0 0 0 Cycles
@@ -6,5 +6,5 @@ Saved Wavefronts,0.0,0,0,Wavefronts
Restored Wavefronts,0.0,0,0,Wavefronts
VGPRs,23.976047904191617,8,36,Registers
SGPRs,24.047904191616766,24,32,Registers
LDS Allocation,0.0,0,0,Bytes
LDS_Per_Workgroup Allocation,0.0,0,0,Bytes
Scratch Allocation,0.0,0,0,Bytes
1 Metric Avg Min Max Unit
6 Restored Wavefronts 0.0 0 0 Wavefronts
7 VGPRs 23.976047904191617 8 36 Registers
8 SGPRs 24.047904191616766 24 32 Registers
9 LDS Allocation LDS_Per_Workgroup Allocation 0.0 0 0 Bytes
10 Scratch Allocation 0.0 0 0 Bytes
+1 -1
View File
@@ -1,2 +1,2 @@
workload_name,host_name,host_cpu,host_distro,host_kernel,host_rocmver,date,gpu_soc,numSE,numCU,numSIMD,waveSize,maxWavesPerCU,maxWorkgroupSize,L1,L2,sclk,mclk,cur_sclk,cur_mclk,L2Banks,name,numSQC,hbmBW,ip_blocks
invdev,vesuvius,AMD EPYC 7542 32-Core Processor,,4.18.0-240.15.1.el8_3.x86_64,4.2.0-21,Thu Sep 22 11:36:21 2022 (CDT),gfx908,8,120,4,64,40,1024,16,,1502,1200,300,1200,32,mi100,48,1228.8,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
invdev,vesuvius,AMD EPYC 7542 32-Core Processor,,4.18.0-240.15.1.el8_3.x86_64,4.2.0-21,Thu Sep 22 11:36:21 2022 (CDT),gfx908,8,120,4,64,40,1024,16,,1502,1200,300,1200,32,mi100,48,1228.8,SQ|LDS_Per_Workgroup|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
1 workload_name host_name host_cpu host_distro host_kernel host_rocmver date gpu_soc numSE numCU numSIMD waveSize maxWavesPerCU maxWorkgroupSize L1 L2 sclk mclk cur_sclk cur_mclk L2Banks name numSQC hbmBW ip_blocks
2 invdev vesuvius AMD EPYC 7542 32-Core Processor 4.18.0-240.15.1.el8_3.x86_64 4.2.0-21 Thu Sep 22 11:36:21 2022 (CDT) gfx908 8 120 4 64 40 1024 16 1502 1200 300 1200 32 mi100 48 1228.8 SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF SQ|LDS_Per_Workgroup|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
+1 -1
View File
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBuffer.kd",0,0,0,937113,937118,33554432,256,0,0,8,32,6464,0x0,0x7f12eac04180,12076635381227990,12076635381273678,12076635381598795,12076635381689318
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,937113,937118,32768,256,0,0,24,24,12480,0x0,0x7f12eac35100,12076635396787522,12076635396887551,12076635396894271,12076635396915901
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,4,937113,937118,4194304,256,0,0,24,24,12928,0x7f141aca3900,0x7f12eac35140,12076635396946107,12076635396958910,12076635397089629,12076635397093471
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBuffer.kd 0 0 0 937113 937118 33554432 256 0 0 8 32 6464 0x0 0x7f12eac04180 12076635381227990 12076635381273678 12076635381598795 12076635381689318
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 937113 937118 32768 256 0 0 24 24 12480 0x0 0x7f12eac35100 12076635396787522 12076635396887551 12076635396894271 12076635396915901
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 4 937113 937118 4194304 256 0 0 24 24 12928 0x7f141aca3900 0x7f12eac35140 12076635396946107 12076635396958910 12076635397089629 12076635397093471
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBufferAligned.kd",0,0,0,236503,236503,33554432,256,0,0,4,32,4160,0x0,0x7fe83c204280,385832,385832,524288,4718592,682094,76386864,17833173257611,17832460487290,17833324063793,17833324177862
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,236503,236503,32768,256,0,0,12,24,13888,0x0,0x7fe83c223f80,34124,34124,512,8192,5540,624020,17833329373303,17833324063793,17833329503317,17833329508120
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,5,236503,236503,4194304,256,0,0,12,24,14336,0x7fe83f16b380,0x7fe83c223fc0,166451,166451,65536,917504,140662,15737800,17833329547519,17833329503317,17833329887317,17833329890091
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBufferAligned.kd 0 0 0 236503 236503 33554432 256 0 0 4 32 4160 0x0 0x7fe83c204280 385832 385832 524288 4718592 682094 76386864 17833173257611 17832460487290 17833324063793 17833324177862
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 236503 236503 32768 256 0 0 12 24 13888 0x0 0x7fe83c223f80 34124 34124 512 8192 5540 624020 17833329373303 17833324063793 17833329503317 17833329508120
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 5 236503 236503 4194304 256 0 0 12 24 14336 0x7fe83f16b380 0x7fe83c223fc0 166451 166451 65536 917504 140662 15737800 17833329547519 17833329503317 17833329887317 17833329890091
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBufferAligned.kd",0,0,0,238351,238351,33554432,256,0,0,4,32,4160,0x0,0x7ff888c04280,0,0,0,17852547584485,17851838027568,17852691164011,17852691281181
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,238351,238351,32768,256,0,0,12,24,13888,0x0,0x7ff888c23f80,0,0,0,17852696444493,17852691164011,17852696572335,17852696576630
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,5,238351,238351,4194304,256,0,0,12,24,14336,0x7ff88bbf2380,0x7ff888c23fc0,0,0,0,17852696612159,17852696572335,17852696936336,17852696938901
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBufferAligned.kd 0 0 0 238351 238351 33554432 256 0 0 4 32 4160 0x0 0x7ff888c04280 0 0 0 17852547584485 17851838027568 17852691164011 17852691281181
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 238351 238351 32768 256 0 0 12 24 13888 0x0 0x7ff888c23f80 0 0 0 17852696444493 17852691164011 17852696572335 17852696576630
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 5 238351 238351 4194304 256 0 0 12 24 14336 0x7ff88bbf2380 0x7ff888c23fc0 0 0 0 17852696612159 17852696572335 17852696936336 17852696938901
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBufferAligned.kd",0,0,0,236281,236281,33554432,256,0,0,4,32,4160,0x0,0x7f45e2204280,3670016,2905706,325081704,17832236843728,17803198665120,17832381629870,17832381744199
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,236281,236281,32768,256,0,0,12,24,13888,0x0,0x7f45e2223f80,512,96664,10851288,17832386900391,17832381629870,17832387026519,17832387030968
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,5,236281,236281,4194304,256,0,0,12,24,14336,0x7f45e517a380,0x7f45e2223fc0,65536,639926,71704992,17832387064037,17832387026519,17832387402360,17832387404639
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBufferAligned.kd 0 0 0 236281 236281 33554432 256 0 0 4 32 4160 0x0 0x7f45e2204280 3670016 2905706 325081704 17832236843728 17803198665120 17832381629870 17832381744199
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 236281 236281 32768 256 0 0 12 24 13888 0x0 0x7f45e2223f80 512 96664 10851288 17832386900391 17832381629870 17832387026519 17832387030968
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 5 236281 236281 4194304 256 0 0 12 24 14336 0x7f45e517a380 0x7f45e2223fc0 65536 639926 71704992 17832387064037 17832387026519 17832387402360 17832387404639
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBufferAligned.kd",0,0,0,238129,238129,33554432,256,0,0,4,32,4160,0x0,0x7fe2c7e04280,524288,5497472,615656480,17851610507915,17849318958956,17851757718801,17851757807931
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,238129,238129,32768,256,0,0,12,24,13888,0x0,0x7fe2c7e23f80,4096,55920,6257708,17851762985423,17851757718801,17851763112570,17851763117229
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,5,238129,238129,4194304,256,0,0,12,24,14336,0x7fe2d67ec380,0x7fe2c7e23fc0,524288,10960198,1227561992,17851763151509,17851763112570,17851763483290,17851763485710
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBufferAligned.kd 0 0 0 238129 238129 33554432 256 0 0 4 32 4160 0x0 0x7fe2c7e04280 524288 5497472 615656480 17851610507915 17849318958956 17851757718801 17851757807931
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 238129 238129 32768 256 0 0 12 24 13888 0x0 0x7fe2c7e23f80 4096 55920 6257708 17851762985423 17851757718801 17851763112570 17851763117229
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 5 238129 238129 4194304 256 0 0 12 24 14336 0x7fe2d67ec380 0x7fe2c7e23fc0 524288 10960198 1227561992 17851763151509 17851763112570 17851763483290 17851763485710
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBufferAligned.kd",0,0,0,236725,236725,33554432,256,0,0,4,32,4160,0x0,0x7fc8d4204280,388118,388118,8813,3104952,524288,245379700,3015215,0,997822932,17834114824366,17833406340767,17834261594719,17834261709089
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,236725,236725,32768,256,0,0,12,24,13888,0x0,0x7fc8d4223f80,34226,34226,30380,273816,512,1732870,162674,0,6944808,17834266862991,17834261594719,17834267004326,17834267009117
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,5,236725,236725,4194304,256,0,0,12,24,14336,0x7fc8d71bf380,0x7fc8d4223fc0,166068,166068,13553,1328552,65536,78338107,1221027,0,315083700,17834267052826,17834267004326,17834267413606,17834267416207
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBufferAligned.kd 0 0 0 236725 236725 33554432 256 0 0 4 32 4160 0x0 0x7fc8d4204280 388118 388118 8813 3104952 524288 245379700 3015215 0 997822932 17834114824366 17833406340767 17834261594719 17834261709089
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 236725 236725 32768 256 0 0 12 24 13888 0x0 0x7fc8d4223f80 34226 34226 30380 273816 512 1732870 162674 0 6944808 17834266862991 17834261594719 17834267004326 17834267009117
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 5 236725 236725 4194304 256 0 0 12 24 14336 0x7fc8d71bf380 0x7fc8d4223fc0 166068 166068 13553 1328552 65536 78338107 1221027 0 315083700 17834267052826 17834267004326 17834267413606 17834267416207
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id
Dispatch_ID,Kernel_Name,GPU_ID
0,__amd_rocclr_fillBufferAligned.kd,0
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID
2 0 __amd_rocclr_fillBufferAligned.kd 0
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0
File diff suppressed because one or more lines are too long
@@ -1,4 +1,4 @@
KernelName,Count,Sum(ns),Mean(ns),Median(ns),Pct
Kernel_Name,Count,Sum(ns),Mean(ns),Median(ns),Pct
"void benchmark_func<int, 256, 8u, 512u>(int, int*) [clone .kd]",1,3357122.0,3357122.0,3357122.0,7.840630919671416
"void benchmark_func<HIP_vector_type<float, 2u>, 256, 8u, 512u>(HIP_vector_type<float, 2u>, HIP_vector_type<float, 2u>*) [clone .kd]",1,1724001.0,1724001.0,1724001.0,4.026441560999106
"void benchmark_func<double, 256, 8u, 512u>(double, double*) [clone .kd]",1,1714881.0,1714881.0,1714881.0,4.005141604075466
1 KernelName Kernel_Name Count Sum(ns) Mean(ns) Median(ns) Pct
2 void benchmark_func<int, 256, 8u, 512u>(int, int*) [clone .kd] 1 3357122.0 3357122.0 3357122.0 7.840630919671416
3 void benchmark_func<HIP_vector_type<float, 2u>, 256, 8u, 512u>(HIP_vector_type<float, 2u>, HIP_vector_type<float, 2u>*) [clone .kd] 1 1724001.0 1724001.0 1724001.0 4.026441560999106
4 void benchmark_func<double, 256, 8u, 512u>(double, double*) [clone .kd] 1 1714881.0 1714881.0 1714881.0 4.005141604075466
@@ -1,7 +1,7 @@
Metric,Count,Unit
VALU - Vector,494.8622754491018,Instr per wave
VMEM,7.958083832335329,Instr per wave
LDS,0.0,Instr per wave
LDS_Per_Workgroup,0.0,Instr per wave
VALU - MFMA,0.0,Instr per wave
SALU,24.520958083832337,Instr per wave
SMEM,1.8023952095808384,Instr per wave
1 Metric Count Unit
2 VALU - Vector 494.8622754491018 Instr per wave
3 VMEM 7.958083832335329 Instr per wave
4 LDS LDS_Per_Workgroup 0.0 Instr per wave
5 VALU - MFMA 0.0 Instr per wave
6 SALU 24.520958083832337 Instr per wave
7 SMEM 1.8023952095808384 Instr per wave
@@ -23,4 +23,4 @@ gfx90a
mi200
56
1638.4
roofline|SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
roofline|SQ|LDS_Per_Workgroup|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
1 Info
23 mi200
24 56
25 1638.4
26 roofline|SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF roofline|SQ|LDS_Per_Workgroup|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
@@ -1,12 +1,12 @@
Metric,Avg,Min,Max,Unit
Wave Cycles,18296.269474052384,1894.2776184082031,258229.23516845703,Cycles/wave
LDS Instrs,0.0,0.0,0.0,Instr per wave
LDS_Per_Workgroup Instrs,0.0,0.0,0.0,Instr per wave
Bandwidth,0.0,0.0,0.0,Bytes per wave
Bank Conficts/Access,,,,Conflicts/access
Index Accesses,0.0,0.0,0.0,Cycles per wave
Dispatch_ID Accesses,0.0,0.0,0.0,Cycles per wave
Atomic Cycles,0.0,0.0,0.0,Cycles per wave
Bank Conflict,0.0,0.0,0.0,Cycles per wave
Addr Conflict,0.0,0.0,0.0,Cycles per wave
Unaligned Stall,0.0,0.0,0.0,Cycles per wave
Mem Violations,0.0,0.0,0.0, per wave
LDS Latency,,,,Cycles
LDS_Per_Workgroup Latency,,,,Cycles
1 Metric Avg Min Max Unit
2 Wave Cycles 18296.269474052384 1894.2776184082031 258229.23516845703 Cycles/wave
3 LDS Instrs LDS_Per_Workgroup Instrs 0.0 0.0 0.0 Instr per wave
4 Bandwidth 0.0 0.0 0.0 Bytes per wave
5 Bank Conficts/Access Conflicts/access
6 Index Accesses Dispatch_ID Accesses 0.0 0.0 0.0 Cycles per wave
7 Atomic Cycles 0.0 0.0 0.0 Cycles per wave
8 Bank Conflict 0.0 0.0 0.0 Cycles per wave
9 Addr Conflict 0.0 0.0 0.0 Cycles per wave
10 Unaligned Stall 0.0 0.0 0.0 Cycles per wave
11 Mem Violations 0.0 0.0 0.0 per wave
12 LDS Latency LDS_Per_Workgroup Latency Cycles
@@ -6,16 +6,16 @@ SMEM,2.0,smem_
VALU,495.0,valu_
MFMA,0.0,mfma_
VMEM,8.0,vmem_
LDS,0.0,lds_
LDS_Per_Workgroup,0.0,lds_
GWS,0.0,gws_
BR,10.0,br_
VGPR,12.0,vgpr_
SGPR,24.0,sgpr_
LDS Allocation,0.0,lds_alloc_
LDS_Per_Workgroup Allocation,0.0,lds_alloc_
Scratch Allocation,0.0,scratch_alloc_
Wavefronts,67894.0,wavefronts_
Workgroups,16973.0,workgroups_
LDS Req,0.0,lds_req_
LDS_Per_Workgroup Req,0.0,lds_req_
IL1 Fetch,128.0,il1_fetch_
IL1 Hit,100.0,il1_hit_
IL1_L2 Rd,0.0,il1_l2_req_
@@ -46,10 +46,10 @@ Fabric_L2 Wr,0.0,l2_fabric_wr_
Fabric_l2 Atomic,0.0,l2_fabric_atom_
HBM Rd,44.0,hbm_rd_
HBM Wr,0.0,hbm_wr_
LDS Util,0.0,lds_util_
LDS_Per_Workgroup Util,0.0,lds_util_
VL1 Coalesce,70.0,vl1_coales_
VL1 Stall,2.0,vl1_stall_
LDS Lat,,lds_lat_
LDS_Per_Workgroup Lat,,lds_lat_
vL1D Lat,,sl1_lat_
IL1 Lat,,il1_lat_
Wave Occupancy,25.0,wave_occ_
1 Metric Value Alias
6 VALU 495.0 valu_
7 MFMA 0.0 mfma_
8 VMEM 8.0 vmem_
9 LDS LDS_Per_Workgroup 0.0 lds_
10 GWS 0.0 gws_
11 BR 10.0 br_
12 VGPR 12.0 vgpr_
13 SGPR 24.0 sgpr_
14 LDS Allocation LDS_Per_Workgroup Allocation 0.0 lds_alloc_
15 Scratch Allocation 0.0 scratch_alloc_
16 Wavefronts 67894.0 wavefronts_
17 Workgroups 16973.0 workgroups_
18 LDS Req LDS_Per_Workgroup Req 0.0 lds_req_
19 IL1 Fetch 128.0 il1_fetch_
20 IL1 Hit 100.0 il1_hit_
21 IL1_L2 Rd 0.0 il1_l2_req_
46 Fabric_l2 Atomic 0.0 l2_fabric_atom_
47 HBM Rd 44.0 hbm_rd_
48 HBM Wr 0.0 hbm_wr_
49 LDS Util LDS_Per_Workgroup Util 0.0 lds_util_
50 VL1 Coalesce 70.0 vl1_coales_
51 VL1 Stall 2.0 vl1_stall_
52 LDS Lat LDS_Per_Workgroup Lat lds_lat_
53 vL1D Lat sl1_lat_
54 IL1 Lat il1_lat_
55 Wave Occupancy 25.0 wave_occ_
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id
Dispatch_ID,Kernel_Name,GPU_ID
0,__amd_rocclr_fillBufferAligned.kd,0
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID
2 0 __amd_rocclr_fillBufferAligned.kd 0
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0
@@ -12,8 +12,8 @@ VALU Util,58.3677767471036,Pct,100,58.3677767471036
MFMA Util,0.0,Pct,100,0.0
VALU Active Threads/Wave,64.0,Threads,64,100.0
IPC - Issue,1.0,Instr/cycle,5,20.0
LDS BW,0.0,Gb/sec,22630.4,0.0
LDS Bank Conflict,,Conflicts/access,32,
LDS_Per_Workgroup BW,0.0,Gb/sec,22630.4,0.0
LDS_Per_Workgroup Bank Conflict,,Conflicts/access,32,
Instr Cache Hit Rate,99.99259859947313,Pct,100,99.99259859947313
Instr Cache BW,1671.978257589768,Gb/s,6092.8,27.441870036596768
Scalar L1D Cache Hit Rate,99.3485588612334,Pct,100,99.3485588612334
1 Metric Value Unit Peak PoP
12 MFMA Util 0.0 Pct 100 0.0
13 VALU Active Threads/Wave 64.0 Threads 64 100.0
14 IPC - Issue 1.0 Instr/cycle 5 20.0
15 LDS BW LDS_Per_Workgroup BW 0.0 Gb/sec 22630.4 0.0
16 LDS Bank Conflict LDS_Per_Workgroup Bank Conflict Conflicts/access 32
17 Instr Cache Hit Rate 99.99259859947313 Pct 100 99.99259859947313
18 Instr Cache BW 1671.978257589768 Gb/s 6092.8 27.441870036596768
19 Scalar L1D Cache Hit Rate 99.3485588612334 Pct 100 99.3485588612334
@@ -6,7 +6,7 @@ Scratch Stall,0.0,0,0,Cycles
Insufficient SIMD Waveslots,12526180.185628742,0,370304750,Simd
Insufficient SIMD VGPRs,0.0,0,0,Simd
Insufficient SIMD SGPRs,0.0,0,0,Simd
Insufficient CU LDS,0.0,0,0,Cu
Insufficient CU LDS_Per_Workgroup,0.0,0,0,Cu
Insufficient CU Barries,0.0,0,0,Cu
Insufficient Bulky Resource,0.0,0,0,Cu
Reach CU Threadgroups Limit,0.0,0,0,Cycles
1 Metric Avg Min Max Unit
6 Insufficient SIMD Waveslots 12526180.185628742 0 370304750 Simd
7 Insufficient SIMD VGPRs 0.0 0 0 Simd
8 Insufficient SIMD SGPRs 0.0 0 0 Simd
9 Insufficient CU LDS Insufficient CU LDS_Per_Workgroup 0.0 0 0 Cu
10 Insufficient CU Barries 0.0 0 0 Cu
11 Insufficient Bulky Resource 0.0 0 0 Cu
12 Reach CU Threadgroups Limit 0.0 0 0 Cycles
@@ -6,5 +6,5 @@ Saved Wavefronts,0.0,0,0,Wavefronts
Restored Wavefronts,0.0,0,0,Wavefronts
VGPRs,11.784431137724551,4,16,Registers
SGPRs,24.047904191616766,24,32,Registers
LDS Allocation,0.0,0,0,Bytes
LDS_Per_Workgroup Allocation,0.0,0,0,Bytes
Scratch Allocation,0.0,0,0,Bytes
1 Metric Avg Min Max Unit
6 Restored Wavefronts 0.0 0 0 Wavefronts
7 VGPRs 11.784431137724551 4 16 Registers
8 SGPRs 24.047904191616766 24 32 Registers
9 LDS Allocation LDS_Per_Workgroup Allocation 0.0 0 0 Bytes
10 Scratch Allocation 0.0 0 0 Bytes
+1 -1
View File
@@ -1,2 +1,2 @@
workload_name,host_name,host_cpu,host_distro,host_kernel,host_rocmver,date,gpu_soc,numSE,numCU,numSIMD,waveSize,maxWavesPerCU,maxWorkgroupSize,L1,L2,sclk,mclk,cur_sclk,cur_mclk,L2Banks,name,numSQC,hbmBW,ip_blocks
invdev,06fa5f860366,AMD EPYC 7282 16-Core Processor,Ubuntu 20.04.5 LTS,5.11.0-27-generic,5.1.3-66,Wed Sep 28 20:50:32 2022 (),gfx90a,8,104,4,64,32,1024,16,8192,1700,1600,800,1600,32,mi200,56,1638.4,roofline|SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
invdev,06fa5f860366,AMD EPYC 7282 16-Core Processor,Ubuntu 20.04.5 LTS,5.11.0-27-generic,5.1.3-66,Wed Sep 28 20:50:32 2022 (),gfx90a,8,104,4,64,32,1024,16,8192,1700,1600,800,1600,32,mi200,56,1638.4,roofline|SQ|LDS_Per_Workgroup|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
1 workload_name host_name host_cpu host_distro host_kernel host_rocmver date gpu_soc numSE numCU numSIMD waveSize maxWavesPerCU maxWorkgroupSize L1 L2 sclk mclk cur_sclk cur_mclk L2Banks name numSQC hbmBW ip_blocks
2 invdev 06fa5f860366 AMD EPYC 7282 16-Core Processor Ubuntu 20.04.5 LTS 5.11.0-27-generic 5.1.3-66 Wed Sep 28 20:50:32 2022 () gfx90a 8 104 4 64 32 1024 16 8192 1700 1600 800 1600 32 mi200 56 1638.4 roofline|SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF roofline|SQ|LDS_Per_Workgroup|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF
+1 -1
View File
@@ -1,4 +1,4 @@
Index,KernelName,gpu-id,queue-id,queue-index,pid,tid,grd,wgr,lds,scr,vgpr,sgpr,fbar,sig,obj,DispatchNs,BeginNs,EndNs,CompleteNs
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,grd,Workgroup_Size,LDS_Per_Workgroup,scr,vgpr,SGPR,fbar,sig,obj,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"__amd_rocclr_fillBufferAligned.kd",0,0,0,238434,238434,33554432,256,0,0,4,32,4160,0x0,0x7fd647c04280,17853365225848,17853365250419,17853365490099,17853365576439
1,"void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd]",0,0,2,238434,238434,32768,256,0,0,12,24,13888,0x0,0x7fd647c23f80,17853370260633,17853370276021,17853370289461,17853370306822
2,"void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd]",0,0,5,238434,238434,4194304,256,0,0,12,24,14336,0x7fd66e582380,0x7fd647c23fc0,17853370311482,17853370360501,17853370452981,17853370455238
1 Index Dispatch_ID KernelName Kernel_Name gpu-id GPU_ID queue-id queue-index pid tid grd wgr Workgroup_Size lds LDS_Per_Workgroup scr vgpr sgpr SGPR fbar sig obj DispatchNs BeginNs Start_Timestamp EndNs End_Timestamp CompleteNs
2 0 __amd_rocclr_fillBufferAligned.kd 0 0 0 238434 238434 33554432 256 0 0 4 32 4160 0x0 0x7fd647c04280 17853365225848 17853365250419 17853365490099 17853365576439
3 1 void benchmark_func<short, 256, 8u, 0u>(short, short*) [clone .kd] 0 0 2 238434 238434 32768 256 0 0 12 24 13888 0x0 0x7fd647c23f80 17853370260633 17853370276021 17853370289461 17853370306822
4 2 void benchmark_func<float, 256, 8u, 0u>(float, float*) [clone .kd] 0 0 5 238434 238434 4194304 256 0 0 12 24 14336 0x7fd66e582380 0x7fd647c23fc0 17853370311482 17853370360501 17853370452981 17853370455238