Pytest add mi200 to analyze workloads (#334)

* Updated links in documentation. (#328)

Updated to reflect new GitHub organization.
Fixed broken links to GitHub pages.

Signed-off-by: David Galiffi <David.Galiffi@amd.com>

* update branch for 2.x documentation builds

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* update checkout action and use concurrency instead of cancel-workflow-action

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* test addition of user option for container launch

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* remove --user option for container, try chown instead

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* fixing yaml syntax

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* reorder job step - start with checkout

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* restore missing run directive

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* Update workloads to include log.txt
Add missing MI200 workloads

Signed-off-by: Jose Santos <josantos@amd.com>

* Signed-off-by: Jose Santos <josantos@amd.com>
Add vcopy workload for tests

* Change exit codes for caught failures

Signed-off-by: Jose Santos <josantos@amd.com>

* reformat

Signed-off-by: Jose Santos <josantos@amd.com>

* Add pytest-xdist for pytest -n

Signed-off-by: Jose Santos <josantos@amd.com>

---------

Signed-off-by: David Galiffi <David.Galiffi@amd.com>
Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>
Signed-off-by: Jose Santos <josantos@amd.com>
Co-authored-by: David Galiffi <David.Galiffi@amd.com>
Co-authored-by: Karl W. Schulz <karl.schulz@amd.com>
Cette révision appartient à :
JoseSantosAMD
2024-03-25 10:20:31 -05:00
révisé par Cole Ramos
Parent 482fd6f2ca
révision da506ad9b5
1122 fichiers modifiés avec 37938 ajouts et 3563 suppressions
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,137526,137526,1048576,256,0,0,8,8,16,64,0x0,0x7ff7d60b8e80,48136,48136,16384,65536,12547,1608968,194163174117052,194175534630063,194175534654063,194163181897573
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,137526,137526,1048576,256,0,0,8,8,16,64,0x0,0x7ff7d60b8e80,43175,43175,16384,65536,8463,1048596,194163181922491,194175534674223,194175534693583,194163182207089
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,137526,137526,1048576,256,0,0,8,8,16,64,0x0,0x7ff7d60b8e80,44071,44071,16384,65536,8088,1048576,194163182239821,194175534779503,194175534798543,194163182401738
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1233276,1233276,1048576,256,0,0,8,8,16,64,0x0,0x7efe2639cec0,47520,47520,16384,65536,13262,1711364,1410088190828131,1410100033478037,1410100033502197,1410088198583377
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1233276,1233276,1048576,256,0,0,8,8,16,64,0x0,0x7efe2639cec0,41636,41636,16384,65536,8455,1048580,1410088198612933,1410100033593877,1410100033612597,1410088198932796
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1233276,1233276,1048576,256,0,0,8,8,16,64,0x0,0x7efe2639cec0,41636,41636,16384,65536,8143,1048580,1410088198965097,1410100033638517,1410100033657077,1410088199142181
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 137526 1233276 137526 1233276 1048576 256 0 0 8 8 16 64 0x0 0x7ff7d60b8e80 0x7efe2639cec0 48136 47520 48136 47520 16384 65536 12547 13262 1608968 1711364 194163174117052 1410088190828131 194175534630063 1410100033478037 194175534654063 1410100033502197 194163181897573 1410088198583377
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 137526 1233276 137526 1233276 1048576 256 0 0 8 8 16 64 0x0 0x7ff7d60b8e80 0x7efe2639cec0 43175 41636 43175 41636 16384 65536 8463 8455 1048596 1048580 194163181922491 1410088198612933 194175534674223 1410100033593877 194175534693583 1410100033612597 194163182207089 1410088198932796
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 137526 1233276 137526 1233276 1048576 256 0 0 8 8 16 64 0x0 0x7ff7d60b8e80 0x7efe2639cec0 44071 41636 44071 41636 16384 65536 8088 8143 1048576 1048580 194163182239821 1410088198965097 194175534779503 1410100033638517 194175534798543 1410100033657077 194163182401738 1410088199142181
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,137714,137714,1048576,256,0,0,8,8,16,64,0x0,0x7f3cb4fe8e80,0,0,0,194163677587142,194175534630063,194175534654063,194163685530482
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,137714,137714,1048576,256,0,0,8,8,16,64,0x0,0x7f3cb4fe8e80,0,0,0,194163685548716,194175534674223,194175534693583,194163685853083
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,137714,137714,1048576,256,0,0,8,8,16,64,0x0,0x7f3cb4fe8e80,0,0,0,194163685881677,194175534779503,194175534798543,194163686052871
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1233465,1233465,1048576,256,0,0,8,8,16,64,0x0,0x7fcfea990ec0,0,0,0,1410088676805908,1410100033478037,1410100033502197,1410088684337091
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1233465,1233465,1048576,256,0,0,8,8,16,64,0x0,0x7fcfea990ec0,0,0,0,1410088684356949,1410100033593877,1410100033612597,1410088684680419
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1233465,1233465,1048576,256,0,0,8,8,16,64,0x0,0x7fcfea990ec0,0,0,0,1410088684708933,1410100033638517,1410100033657077,1410088684889954
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 137714 1233465 137714 1233465 1048576 256 0 0 8 8 16 64 0x0 0x7f3cb4fe8e80 0x7fcfea990ec0 0 0 0 194163677587142 1410088676805908 194175534630063 1410100033478037 194175534654063 1410100033502197 194163685530482 1410088684337091
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 137714 1233465 137714 1233465 1048576 256 0 0 8 8 16 64 0x0 0x7f3cb4fe8e80 0x7fcfea990ec0 0 0 0 194163685548716 1410088684356949 194175534674223 1410100033593877 194175534693583 1410100033612597 194163685853083 1410088684680419
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 137714 1233465 137714 1233465 1048576 256 0 0 8 8 16 64 0x0 0x7f3cb4fe8e80 0x7fcfea990ec0 0 0 0 194163685881677 1410088684708933 194175534779503 1410100033638517 194175534798543 1410100033657077 194163686052871 1410088684889954
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,137904,137904,1048576,256,0,0,8,8,16,64,0x0,0x7f071b3dce80,65536,188508,24142904,194164172303568,194175534630063,194175534654063,194164179881035
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,137904,137904,1048576,256,0,0,8,8,16,64,0x0,0x7f071b3dce80,65536,169170,21619920,194164179900282,194175534674223,194175534693583,194164180246938
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,137904,137904,1048576,256,0,0,8,8,16,64,0x0,0x7f071b3dce80,65536,168242,21557048,194164180274780,194175534779503,194175534798543,194164180454190
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1233652,1233652,1048576,256,0,0,8,8,16,64,0x0,0x7f35fba6cec0,65536,215772,27659752,1410089161300999,1410100033478037,1410100033502197,1410089168886174
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1233652,1233652,1048576,256,0,0,8,8,16,64,0x0,0x7f35fba6cec0,65536,218704,27996088,1410089168906202,1410100033593877,1410100033612597,1410089169214454
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1233652,1233652,1048576,256,0,0,8,8,16,64,0x0,0x7f35fba6cec0,65536,216244,27719712,1410089169242256,1410100033638517,1410100033657077,1410089169412888
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 137904 1233652 137904 1233652 1048576 256 0 0 8 8 16 64 0x0 0x7f071b3dce80 0x7f35fba6cec0 65536 188508 215772 24142904 27659752 194164172303568 1410089161300999 194175534630063 1410100033478037 194175534654063 1410100033502197 194164179881035 1410089168886174
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 137904 1233652 137904 1233652 1048576 256 0 0 8 8 16 64 0x0 0x7f071b3dce80 0x7f35fba6cec0 65536 169170 218704 21619920 27996088 194164179900282 1410089168906202 194175534674223 1410100033593877 194175534693583 1410100033612597 194164180246938 1410089169214454
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 137904 1233652 137904 1233652 1048576 256 0 0 8 8 16 64 0x0 0x7f071b3dce80 0x7f35fba6cec0 65536 168242 216244 21557048 27719712 194164180274780 1410089169242256 194175534779503 1410100033638517 194175534798543 1410100033657077 194164180454190 1410089169412888
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,138091,138091,1048576,256,0,0,8,8,16,64,0x0,0x7f290d802e80,32768,655161,83871140,194164678008099,194175534630063,194175534654063,194164686165985
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,138091,138091,1048576,256,0,0,8,8,16,64,0x0,0x7f290d802e80,32768,672735,86120852,194164686184760,194175534674223,194175534693583,194164686503344
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,138091,138091,1048576,256,0,0,8,8,16,64,0x0,0x7f290d802e80,32768,657649,84175780,194164686531978,194175534779503,194175534798543,194164686707520
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1233840,1233840,1048576,256,0,0,8,8,16,64,0x0,0x7fc449ed8ec0,32768,655957,83965736,1410089647501356,1410100033478037,1410100033502197,1410089655200997
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1233840,1233840,1048576,256,0,0,8,8,16,64,0x0,0x7fc449ed8ec0,32768,665445,85174904,1410089655222728,1410100033593877,1410100033612597,1410089655499350
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1233840,1233840,1048576,256,0,0,8,8,16,64,0x0,0x7fc449ed8ec0,32768,657926,84218544,1410089655527864,1410100033638517,1410100033657077,1410089655705819
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 138091 1233840 138091 1233840 1048576 256 0 0 8 8 16 64 0x0 0x7f290d802e80 0x7fc449ed8ec0 32768 655161 655957 83871140 83965736 194164678008099 1410089647501356 194175534630063 1410100033478037 194175534654063 1410100033502197 194164686165985 1410089655200997
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 138091 1233840 138091 1233840 1048576 256 0 0 8 8 16 64 0x0 0x7f290d802e80 0x7fc449ed8ec0 32768 672735 665445 86120852 85174904 194164686184760 1410089655222728 194175534674223 1410100033593877 194175534693583 1410100033612597 194164686503344 1410089655499350
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 138091 1233840 138091 1233840 1048576 256 0 0 8 8 16 64 0x0 0x7f290d802e80 0x7fc449ed8ec0 32768 657649 657926 84175780 84218544 194164686531978 1410089655527864 194175534779503 1410100033638517 194175534798543 1410100033657077 194164686707520 1410089655705819
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,138275,138275,1048576,256,0,0,8,8,16,64,0x0,0x7f5d7a4c8e80,47598,47598,16289,380792,16384,24789627,229303,0,99653036,194165181026564,194175534630063,194175534654063,194165189084451
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,138275,138275,1048576,256,0,0,8,8,16,64,0x0,0x7f5d7a4c8e80,42421,42421,14055,339376,16384,24679834,225411,0,99219160,194165189106292,194175534674223,194175534693583,194165189431618
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,138275,138275,1048576,256,0,0,8,8,16,64,0x0,0x7f5d7a4c8e80,41925,41925,13806,335408,16384,24141356,225187,0,97056972,194165189469360,194175534779503,194175534798543,194165189654570
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1234024,1234024,1048576,256,0,0,8,8,16,64,0x0,0x7f90cce04ec0,48770,48770,18412,390168,16384,24851481,236800,0,99947332,1410090131932426,1410100033478037,1410100033502197,1410090139651645
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1234024,1234024,1048576,256,0,0,8,8,16,64,0x0,0x7f90cce04ec0,42136,42136,13157,337096,16384,24727121,229507,0,99439364,1410090139682352,1410100033593877,1410100033612597,1410090139991095
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1234024,1234024,1048576,256,0,0,8,8,16,64,0x0,0x7f90cce04ec0,42472,42472,13191,339784,16384,23653826,229247,0,95156176,1410090140029487,1410100033638517,1410100033657077,1410090140209627
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 138275 1234024 138275 1234024 1048576 256 0 0 8 8 16 64 0x0 0x7f5d7a4c8e80 0x7f90cce04ec0 47598 48770 47598 48770 16289 18412 380792 390168 16384 24789627 24851481 229303 236800 0 99653036 99947332 194165181026564 1410090131932426 194175534630063 1410100033478037 194175534654063 1410100033502197 194165189084451 1410090139651645
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 138275 1234024 138275 1234024 1048576 256 0 0 8 8 16 64 0x0 0x7f5d7a4c8e80 0x7f90cce04ec0 42421 42136 42421 42136 14055 13157 339376 337096 16384 24679834 24727121 225411 229507 0 99219160 99439364 194165189106292 1410090139682352 194175534674223 1410100033593877 194175534693583 1410100033612597 194165189431618 1410090139991095
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 138275 1234024 138275 1234024 1048576 256 0 0 8 8 16 64 0x0 0x7f5d7a4c8e80 0x7f90cce04ec0 41925 42472 41925 42472 13806 13191 335408 339784 16384 24141356 23653826 225187 229247 0 97056972 95156176 194165189469360 1410090140029487 194175534779503 1410100033638517 194175534798543 1410100033657077 194165189654570 1410090140209627
+702
Voir le fichier
@@ -0,0 +1,702 @@
Omniperf version: 2.0.0-RC1
Profiler choice: rocprofv1
Path: /home1/josantos/omniperf/tests/workloads/kernel_substr/MI100
Target: MI100
Command: ./tests/vcopy -n 1048576 -b 256 -i 3
Kernel Selection: ['vecCopy']
Dispatch Selection: None
IP Blocks: All
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Collecting Performance Counters
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/SQ_IFETCH_LEVEL.txt
|-> [rocprof] RPL: on '240321_155234' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/SQ_IFETCH_LEVEL.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155234_1233116'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155234_1233116/input0_results_240321_155234'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155234_1233116/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 6 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, SQ_WAVES, SQ_IFETCH, SQ_IFETCH_LEVEL, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155234_1233116/input0_results_240321_155234
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/SQ_IFETCH_LEVEL.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/SQ_INST_LEVEL_LDS.txt
|-> [rocprof] RPL: on '240321_155234' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/SQ_INST_LEVEL_LDS.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155234_1233305'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155234_1233305/input0_results_240321_155234'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155234_1233305/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_LDS, SQ_INST_LEVEL_LDS, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155234_1233305/input0_results_240321_155234
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/SQ_INST_LEVEL_LDS.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/SQ_INST_LEVEL_SMEM.txt
|-> [rocprof] RPL: on '240321_155235' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/SQ_INST_LEVEL_SMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155235_1233492'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155235_1233492/input0_results_240321_155235'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155235_1233492/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_SMEM, SQ_INST_LEVEL_SMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155235_1233492/input0_results_240321_155235
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/SQ_INST_LEVEL_SMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/SQ_INST_LEVEL_VMEM.txt
|-> [rocprof] RPL: on '240321_155235' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/SQ_INST_LEVEL_VMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155235_1233680'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155235_1233680/input0_results_240321_155235'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155235_1233680/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_VMEM, SQ_INST_LEVEL_VMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155235_1233680/input0_results_240321_155235
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/SQ_INST_LEVEL_VMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/SQ_LEVEL_WAVES.txt
|-> [rocprof] RPL: on '240321_155236' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/SQ_LEVEL_WAVES.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155236_1233864'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155236_1233864/input0_results_240321_155236'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155236_1233864/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 9 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, CPC_ME1_BUSY_FOR_PACKET_DECODE, SQ_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQ_BUSY_CYCLES, SQ_LEVEL_WAVES, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155236_1233864/input0_results_240321_155236
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/SQ_LEVEL_WAVES.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_0.txt
|-> [rocprof] RPL: on '240321_155236' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_0.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155236_1234050'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155236_1234050/input0_results_240321_155236'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155236_1234050/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 28 metrics
|-> [rocprof] SQ_CYCLES, SQ_BUSY_CYCLES, SQ_BUSY_CU_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQC_TC_INST_REQ, SQC_TC_DATA_READ_REQ, SQC_TC_DATA_WRITE_REQ, GRBM_COUNT, GRBM_GUI_ACTIVE, TCP_GATE_EN1_sum, TCP_GATE_EN2_sum, TCP_TD_TCP_STALL_CYCLES_sum, TCP_TCR_TCP_STALL_CYCLES_sum, TA_TA_BUSY_sum, TA_BUFFER_WAVEFRONTS_sum, TD_TD_BUSY_sum, TD_TC_STALL_sum, SPI_CSN_WINDOW_VALID, SPI_CSN_BUSY, CPC_CPC_STAT_BUSY, CPC_CPC_STAT_IDLE, CPF_CPF_STAT_BUSY, CPF_CPF_STAT_STALL, TCC_CYCLE_sum, TCC_BUSY_sum, TCC_PROBE_sum, TCC_PROBE_ALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155236_1234050/input0_results_240321_155236
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_0.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_1.txt
|-> [rocprof] RPL: on '240321_155237' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_1.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155237_1234236'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155237_1234236/input0_results_240321_155237'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155237_1234236/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 27 metrics
|-> [rocprof] SQC_TC_DATA_ATOMIC_REQ, SQC_TC_STALL, SQC_TC_REQ, SQC_DCACHE_REQ_READ_16, SQC_ICACHE_REQ, SQC_ICACHE_HITS, SQC_ICACHE_MISSES, SQC_ICACHE_MISSES_DUPLICATE, GRBM_SPI_BUSY, TCP_READ_TAGCONFLICT_STALL_CYCLES_sum, TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum, TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum, TCP_TA_TCP_STATE_READ_sum, TA_BUFFER_READ_WAVEFRONTS_sum, TA_BUFFER_WRITE_WAVEFRONTS_sum, TD_COALESCABLE_WAVEFRONT_sum, TD_LOAD_WAVEFRONT_sum, SPI_CSN_NUM_THREADGROUPS, SPI_CSN_WAVE, CPC_CPC_TCIU_BUSY, CPC_CPC_TCIU_IDLE, CPF_CPF_TCIU_BUSY, CPF_CPF_TCIU_STALL, TCC_NC_REQ_sum, TCC_UC_REQ_sum, TCC_CC_REQ_sum, TCC_RW_REQ_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155237_1234236/input0_results_240321_155237
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_1.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_10.txt
|-> [rocprof] RPL: on '240321_155237' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_10.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155237_1234425'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155237_1234425/input0_results_240321_155237'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155237_1234425/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 1 metrics
|-> [rocprof] TCC_EA_ATOMIC_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155237_1234425/input0_results_240321_155237
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_10.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_11.txt
|-> [rocprof] RPL: on '240321_155238' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_11.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155238_1234614'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155238_1234614/input0_results_240321_155238'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155238_1234614/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_ATOMIC[0], TCC_CYCLE[0], TCC_EA_ATOMIC[0], TCC_EA_ATOMIC_LEVEL[0], TCC_ATOMIC[1], TCC_CYCLE[1], TCC_EA_ATOMIC[1], TCC_EA_ATOMIC_LEVEL[1], TCC_ATOMIC[2], TCC_CYCLE[2], TCC_EA_ATOMIC[2], TCC_EA_ATOMIC_LEVEL[2], TCC_ATOMIC[3], TCC_CYCLE[3], TCC_EA_ATOMIC[3], TCC_EA_ATOMIC_LEVEL[3], TCC_ATOMIC[4], TCC_CYCLE[4], TCC_EA_ATOMIC[4], TCC_EA_ATOMIC_LEVEL[4], TCC_ATOMIC[5], TCC_CYCLE[5], TCC_EA_ATOMIC[5], TCC_EA_ATOMIC_LEVEL[5], TCC_ATOMIC[6], TCC_CYCLE[6], TCC_EA_ATOMIC[6], TCC_EA_ATOMIC_LEVEL[6], TCC_ATOMIC[7], TCC_CYCLE[7], TCC_EA_ATOMIC[7], TCC_EA_ATOMIC_LEVEL[7], TCC_ATOMIC[8], TCC_CYCLE[8], TCC_EA_ATOMIC[8], TCC_EA_ATOMIC_LEVEL[8], TCC_ATOMIC[9], TCC_CYCLE[9], TCC_EA_ATOMIC[9], TCC_EA_ATOMIC_LEVEL[9], TCC_ATOMIC[10], TCC_CYCLE[10], TCC_EA_ATOMIC[10], TCC_EA_ATOMIC_LEVEL[10], TCC_ATOMIC[11], TCC_CYCLE[11], TCC_EA_ATOMIC[11], TCC_EA_ATOMIC_LEVEL[11], TCC_ATOMIC[12], TCC_CYCLE[12], TCC_EA_ATOMIC[12], TCC_EA_ATOMIC_LEVEL[12], TCC_ATOMIC[13], TCC_CYCLE[13], TCC_EA_ATOMIC[13], TCC_EA_ATOMIC_LEVEL[13], TCC_ATOMIC[14], TCC_CYCLE[14], TCC_EA_ATOMIC[14], TCC_EA_ATOMIC_LEVEL[14], TCC_ATOMIC[15], TCC_CYCLE[15], TCC_EA_ATOMIC[15], TCC_EA_ATOMIC_LEVEL[15], TCC_ATOMIC[16], TCC_CYCLE[16], TCC_EA_ATOMIC[16], TCC_EA_ATOMIC_LEVEL[16], TCC_ATOMIC[17], TCC_CYCLE[17], TCC_EA_ATOMIC[17], TCC_EA_ATOMIC_LEVEL[17], TCC_ATOMIC[18], TCC_CYCLE[18], TCC_EA_ATOMIC[18], TCC_EA_ATOMIC_LEVEL[18], TCC_ATOMIC[19], TCC_CYCLE[19], TCC_EA_ATOMIC[19], TCC_EA_ATOMIC_LEVEL[19], TCC_ATOMIC[20], TCC_CYCLE[20], TCC_EA_ATOMIC[20], TCC_EA_ATOMIC_LEVEL[20], TCC_ATOMIC[21], TCC_CYCLE[21], TCC_EA_ATOMIC[21], TCC_EA_ATOMIC_LEVEL[21], TCC_ATOMIC[22], TCC_CYCLE[22], TCC_EA_ATOMIC[22], TCC_EA_ATOMIC_LEVEL[22], TCC_ATOMIC[23], TCC_CYCLE[23], TCC_EA_ATOMIC[23], TCC_EA_ATOMIC_LEVEL[23], TCC_ATOMIC[24], TCC_CYCLE[24], TCC_EA_ATOMIC[24], TCC_EA_ATOMIC_LEVEL[24], TCC_ATOMIC[25], TCC_CYCLE[25], TCC_EA_ATOMIC[25], TCC_EA_ATOMIC_LEVEL[25], TCC_ATOMIC[26], TCC_CYCLE[26], TCC_EA_ATOMIC[26], TCC_EA_ATOMIC_LEVEL[26], TCC_ATOMIC[27], TCC_CYCLE[27], TCC_EA_ATOMIC[27], TCC_EA_ATOMIC_LEVEL[27], TCC_ATOMIC[28], TCC_CYCLE[28], TCC_EA_ATOMIC[28], TCC_EA_ATOMIC_LEVEL[28], TCC_ATOMIC[29], TCC_CYCLE[29], TCC_EA_ATOMIC[29], TCC_EA_ATOMIC_LEVEL[29], TCC_ATOMIC[30], TCC_CYCLE[30], TCC_EA_ATOMIC[30], TCC_EA_ATOMIC_LEVEL[30], TCC_ATOMIC[31], TCC_CYCLE[31], TCC_EA_ATOMIC[31], TCC_EA_ATOMIC_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155238_1234614/input0_results_240321_155238
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_11.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_12.txt
|-> [rocprof] RPL: on '240321_155238' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_12.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155238_1234809'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155238_1234809/input0_results_240321_155238'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155238_1234809/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ[0], TCC_EA_RDREQ_32B[0], TCC_EA_RDREQ_DRAM_CREDIT_STALL[0], TCC_EA_RDREQ_GMI_CREDIT_STALL[0], TCC_EA_RDREQ[1], TCC_EA_RDREQ_32B[1], TCC_EA_RDREQ_DRAM_CREDIT_STALL[1], TCC_EA_RDREQ_GMI_CREDIT_STALL[1], TCC_EA_RDREQ[2], TCC_EA_RDREQ_32B[2], TCC_EA_RDREQ_DRAM_CREDIT_STALL[2], TCC_EA_RDREQ_GMI_CREDIT_STALL[2], TCC_EA_RDREQ[3], TCC_EA_RDREQ_32B[3], TCC_EA_RDREQ_DRAM_CREDIT_STALL[3], TCC_EA_RDREQ_GMI_CREDIT_STALL[3], TCC_EA_RDREQ[4], TCC_EA_RDREQ_32B[4], TCC_EA_RDREQ_DRAM_CREDIT_STALL[4], TCC_EA_RDREQ_GMI_CREDIT_STALL[4], TCC_EA_RDREQ[5], TCC_EA_RDREQ_32B[5], TCC_EA_RDREQ_DRAM_CREDIT_STALL[5], TCC_EA_RDREQ_GMI_CREDIT_STALL[5], TCC_EA_RDREQ[6], TCC_EA_RDREQ_32B[6], TCC_EA_RDREQ_DRAM_CREDIT_STALL[6], TCC_EA_RDREQ_GMI_CREDIT_STALL[6], TCC_EA_RDREQ[7], TCC_EA_RDREQ_32B[7], TCC_EA_RDREQ_DRAM_CREDIT_STALL[7], TCC_EA_RDREQ_GMI_CREDIT_STALL[7], TCC_EA_RDREQ[8], TCC_EA_RDREQ_32B[8], TCC_EA_RDREQ_DRAM_CREDIT_STALL[8], TCC_EA_RDREQ_GMI_CREDIT_STALL[8], TCC_EA_RDREQ[9], TCC_EA_RDREQ_32B[9], TCC_EA_RDREQ_DRAM_CREDIT_STALL[9], TCC_EA_RDREQ_GMI_CREDIT_STALL[9], TCC_EA_RDREQ[10], TCC_EA_RDREQ_32B[10], TCC_EA_RDREQ_DRAM_CREDIT_STALL[10], TCC_EA_RDREQ_GMI_CREDIT_STALL[10], TCC_EA_RDREQ[11], TCC_EA_RDREQ_32B[11], TCC_EA_RDREQ_DRAM_CREDIT_STALL[11], TCC_EA_RDREQ_GMI_CREDIT_STALL[11], TCC_EA_RDREQ[12], TCC_EA_RDREQ_32B[12], TCC_EA_RDREQ_DRAM_CREDIT_STALL[12], TCC_EA_RDREQ_GMI_CREDIT_STALL[12], TCC_EA_RDREQ[13], TCC_EA_RDREQ_32B[13], TCC_EA_RDREQ_DRAM_CREDIT_STALL[13], TCC_EA_RDREQ_GMI_CREDIT_STALL[13], TCC_EA_RDREQ[14], TCC_EA_RDREQ_32B[14], TCC_EA_RDREQ_DRAM_CREDIT_STALL[14], TCC_EA_RDREQ_GMI_CREDIT_STALL[14], TCC_EA_RDREQ[15], TCC_EA_RDREQ_32B[15], TCC_EA_RDREQ_DRAM_CREDIT_STALL[15], TCC_EA_RDREQ_GMI_CREDIT_STALL[15], TCC_EA_RDREQ[16], TCC_EA_RDREQ_32B[16], TCC_EA_RDREQ_DRAM_CREDIT_STALL[16], TCC_EA_RDREQ_GMI_CREDIT_STALL[16], TCC_EA_RDREQ[17], TCC_EA_RDREQ_32B[17], TCC_EA_RDREQ_DRAM_CREDIT_STALL[17], TCC_EA_RDREQ_GMI_CREDIT_STALL[17], TCC_EA_RDREQ[18], TCC_EA_RDREQ_32B[18], TCC_EA_RDREQ_DRAM_CREDIT_STALL[18], TCC_EA_RDREQ_GMI_CREDIT_STALL[18], TCC_EA_RDREQ[19], TCC_EA_RDREQ_32B[19], TCC_EA_RDREQ_DRAM_CREDIT_STALL[19], TCC_EA_RDREQ_GMI_CREDIT_STALL[19], TCC_EA_RDREQ[20], TCC_EA_RDREQ_32B[20], TCC_EA_RDREQ_DRAM_CREDIT_STALL[20], TCC_EA_RDREQ_GMI_CREDIT_STALL[20], TCC_EA_RDREQ[21], TCC_EA_RDREQ_32B[21], TCC_EA_RDREQ_DRAM_CREDIT_STALL[21], TCC_EA_RDREQ_GMI_CREDIT_STALL[21], TCC_EA_RDREQ[22], TCC_EA_RDREQ_32B[22], TCC_EA_RDREQ_DRAM_CREDIT_STALL[22], TCC_EA_RDREQ_GMI_CREDIT_STALL[22], TCC_EA_RDREQ[23], TCC_EA_RDREQ_32B[23], TCC_EA_RDREQ_DRAM_CREDIT_STALL[23], TCC_EA_RDREQ_GMI_CREDIT_STALL[23], TCC_EA_RDREQ[24], TCC_EA_RDREQ_32B[24], TCC_EA_RDREQ_DRAM_CREDIT_STALL[24], TCC_EA_RDREQ_GMI_CREDIT_STALL[24], TCC_EA_RDREQ[25], TCC_EA_RDREQ_32B[25], TCC_EA_RDREQ_DRAM_CREDIT_STALL[25], TCC_EA_RDREQ_GMI_CREDIT_STALL[25], TCC_EA_RDREQ[26], TCC_EA_RDREQ_32B[26], TCC_EA_RDREQ_DRAM_CREDIT_STALL[26], TCC_EA_RDREQ_GMI_CREDIT_STALL[26], TCC_EA_RDREQ[27], TCC_EA_RDREQ_32B[27], TCC_EA_RDREQ_DRAM_CREDIT_STALL[27], TCC_EA_RDREQ_GMI_CREDIT_STALL[27], TCC_EA_RDREQ[28], TCC_EA_RDREQ_32B[28], TCC_EA_RDREQ_DRAM_CREDIT_STALL[28], TCC_EA_RDREQ_GMI_CREDIT_STALL[28], TCC_EA_RDREQ[29], TCC_EA_RDREQ_32B[29], TCC_EA_RDREQ_DRAM_CREDIT_STALL[29], TCC_EA_RDREQ_GMI_CREDIT_STALL[29], TCC_EA_RDREQ[30], TCC_EA_RDREQ_32B[30], TCC_EA_RDREQ_DRAM_CREDIT_STALL[30], TCC_EA_RDREQ_GMI_CREDIT_STALL[30], TCC_EA_RDREQ[31], TCC_EA_RDREQ_32B[31], TCC_EA_RDREQ_DRAM_CREDIT_STALL[31], TCC_EA_RDREQ_GMI_CREDIT_STALL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155238_1234809/input0_results_240321_155238
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_12.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_13.txt
|-> [rocprof] RPL: on '240321_155239' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_13.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155239_1234998'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155239_1234998/input0_results_240321_155239'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155239_1234998/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ_IO_CREDIT_STALL[0], TCC_EA_RDREQ_LEVEL[0], TCC_EA_WRREQ[0], TCC_EA_WRREQ_64B[0], TCC_EA_RDREQ_IO_CREDIT_STALL[1], TCC_EA_RDREQ_LEVEL[1], TCC_EA_WRREQ[1], TCC_EA_WRREQ_64B[1], TCC_EA_RDREQ_IO_CREDIT_STALL[2], TCC_EA_RDREQ_LEVEL[2], TCC_EA_WRREQ[2], TCC_EA_WRREQ_64B[2], TCC_EA_RDREQ_IO_CREDIT_STALL[3], TCC_EA_RDREQ_LEVEL[3], TCC_EA_WRREQ[3], TCC_EA_WRREQ_64B[3], TCC_EA_RDREQ_IO_CREDIT_STALL[4], TCC_EA_RDREQ_LEVEL[4], TCC_EA_WRREQ[4], TCC_EA_WRREQ_64B[4], TCC_EA_RDREQ_IO_CREDIT_STALL[5], TCC_EA_RDREQ_LEVEL[5], TCC_EA_WRREQ[5], TCC_EA_WRREQ_64B[5], TCC_EA_RDREQ_IO_CREDIT_STALL[6], TCC_EA_RDREQ_LEVEL[6], TCC_EA_WRREQ[6], TCC_EA_WRREQ_64B[6], TCC_EA_RDREQ_IO_CREDIT_STALL[7], TCC_EA_RDREQ_LEVEL[7], TCC_EA_WRREQ[7], TCC_EA_WRREQ_64B[7], TCC_EA_RDREQ_IO_CREDIT_STALL[8], TCC_EA_RDREQ_LEVEL[8], TCC_EA_WRREQ[8], TCC_EA_WRREQ_64B[8], TCC_EA_RDREQ_IO_CREDIT_STALL[9], TCC_EA_RDREQ_LEVEL[9], TCC_EA_WRREQ[9], TCC_EA_WRREQ_64B[9], TCC_EA_RDREQ_IO_CREDIT_STALL[10], TCC_EA_RDREQ_LEVEL[10], TCC_EA_WRREQ[10], TCC_EA_WRREQ_64B[10], TCC_EA_RDREQ_IO_CREDIT_STALL[11], TCC_EA_RDREQ_LEVEL[11], TCC_EA_WRREQ[11], TCC_EA_WRREQ_64B[11], TCC_EA_RDREQ_IO_CREDIT_STALL[12], TCC_EA_RDREQ_LEVEL[12], TCC_EA_WRREQ[12], TCC_EA_WRREQ_64B[12], TCC_EA_RDREQ_IO_CREDIT_STALL[13], TCC_EA_RDREQ_LEVEL[13], TCC_EA_WRREQ[13], TCC_EA_WRREQ_64B[13], TCC_EA_RDREQ_IO_CREDIT_STALL[14], TCC_EA_RDREQ_LEVEL[14], TCC_EA_WRREQ[14], TCC_EA_WRREQ_64B[14], TCC_EA_RDREQ_IO_CREDIT_STALL[15], TCC_EA_RDREQ_LEVEL[15], TCC_EA_WRREQ[15], TCC_EA_WRREQ_64B[15], TCC_EA_RDREQ_IO_CREDIT_STALL[16], TCC_EA_RDREQ_LEVEL[16], TCC_EA_WRREQ[16], TCC_EA_WRREQ_64B[16], TCC_EA_RDREQ_IO_CREDIT_STALL[17], TCC_EA_RDREQ_LEVEL[17], TCC_EA_WRREQ[17], TCC_EA_WRREQ_64B[17], TCC_EA_RDREQ_IO_CREDIT_STALL[18], TCC_EA_RDREQ_LEVEL[18], TCC_EA_WRREQ[18], TCC_EA_WRREQ_64B[18], TCC_EA_RDREQ_IO_CREDIT_STALL[19], TCC_EA_RDREQ_LEVEL[19], TCC_EA_WRREQ[19], TCC_EA_WRREQ_64B[19], TCC_EA_RDREQ_IO_CREDIT_STALL[20], TCC_EA_RDREQ_LEVEL[20], TCC_EA_WRREQ[20], TCC_EA_WRREQ_64B[20], TCC_EA_RDREQ_IO_CREDIT_STALL[21], TCC_EA_RDREQ_LEVEL[21], TCC_EA_WRREQ[21], TCC_EA_WRREQ_64B[21], TCC_EA_RDREQ_IO_CREDIT_STALL[22], TCC_EA_RDREQ_LEVEL[22], TCC_EA_WRREQ[22], TCC_EA_WRREQ_64B[22], TCC_EA_RDREQ_IO_CREDIT_STALL[23], TCC_EA_RDREQ_LEVEL[23], TCC_EA_WRREQ[23], TCC_EA_WRREQ_64B[23], TCC_EA_RDREQ_IO_CREDIT_STALL[24], TCC_EA_RDREQ_LEVEL[24], TCC_EA_WRREQ[24], TCC_EA_WRREQ_64B[24], TCC_EA_RDREQ_IO_CREDIT_STALL[25], TCC_EA_RDREQ_LEVEL[25], TCC_EA_WRREQ[25], TCC_EA_WRREQ_64B[25], TCC_EA_RDREQ_IO_CREDIT_STALL[26], TCC_EA_RDREQ_LEVEL[26], TCC_EA_WRREQ[26], TCC_EA_WRREQ_64B[26], TCC_EA_RDREQ_IO_CREDIT_STALL[27], TCC_EA_RDREQ_LEVEL[27], TCC_EA_WRREQ[27], TCC_EA_WRREQ_64B[27], TCC_EA_RDREQ_IO_CREDIT_STALL[28], TCC_EA_RDREQ_LEVEL[28], TCC_EA_WRREQ[28], TCC_EA_WRREQ_64B[28], TCC_EA_RDREQ_IO_CREDIT_STALL[29], TCC_EA_RDREQ_LEVEL[29], TCC_EA_WRREQ[29], TCC_EA_WRREQ_64B[29], TCC_EA_RDREQ_IO_CREDIT_STALL[30], TCC_EA_RDREQ_LEVEL[30], TCC_EA_WRREQ[30], TCC_EA_WRREQ_64B[30], TCC_EA_RDREQ_IO_CREDIT_STALL[31], TCC_EA_RDREQ_LEVEL[31], TCC_EA_WRREQ[31], TCC_EA_WRREQ_64B[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155239_1234998/input0_results_240321_155239
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_13.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_14.txt
|-> [rocprof] RPL: on '240321_155240' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_14.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155240_1235184'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155240_1235184/input0_results_240321_155240'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155240_1235184/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_WRREQ_DRAM_CREDIT_STALL[0], TCC_EA_WRREQ_GMI_CREDIT_STALL[0], TCC_EA_WRREQ_IO_CREDIT_STALL[0], TCC_EA_WRREQ_LEVEL[0], TCC_EA_WRREQ_DRAM_CREDIT_STALL[1], TCC_EA_WRREQ_GMI_CREDIT_STALL[1], TCC_EA_WRREQ_IO_CREDIT_STALL[1], TCC_EA_WRREQ_LEVEL[1], TCC_EA_WRREQ_DRAM_CREDIT_STALL[2], TCC_EA_WRREQ_GMI_CREDIT_STALL[2], TCC_EA_WRREQ_IO_CREDIT_STALL[2], TCC_EA_WRREQ_LEVEL[2], TCC_EA_WRREQ_DRAM_CREDIT_STALL[3], TCC_EA_WRREQ_GMI_CREDIT_STALL[3], TCC_EA_WRREQ_IO_CREDIT_STALL[3], TCC_EA_WRREQ_LEVEL[3], TCC_EA_WRREQ_DRAM_CREDIT_STALL[4], TCC_EA_WRREQ_GMI_CREDIT_STALL[4], TCC_EA_WRREQ_IO_CREDIT_STALL[4], TCC_EA_WRREQ_LEVEL[4], TCC_EA_WRREQ_DRAM_CREDIT_STALL[5], TCC_EA_WRREQ_GMI_CREDIT_STALL[5], TCC_EA_WRREQ_IO_CREDIT_STALL[5], TCC_EA_WRREQ_LEVEL[5], TCC_EA_WRREQ_DRAM_CREDIT_STALL[6], TCC_EA_WRREQ_GMI_CREDIT_STALL[6], TCC_EA_WRREQ_IO_CREDIT_STALL[6], TCC_EA_WRREQ_LEVEL[6], TCC_EA_WRREQ_DRAM_CREDIT_STALL[7], TCC_EA_WRREQ_GMI_CREDIT_STALL[7], TCC_EA_WRREQ_IO_CREDIT_STALL[7], TCC_EA_WRREQ_LEVEL[7], TCC_EA_WRREQ_DRAM_CREDIT_STALL[8], TCC_EA_WRREQ_GMI_CREDIT_STALL[8], TCC_EA_WRREQ_IO_CREDIT_STALL[8], TCC_EA_WRREQ_LEVEL[8], TCC_EA_WRREQ_DRAM_CREDIT_STALL[9], TCC_EA_WRREQ_GMI_CREDIT_STALL[9], TCC_EA_WRREQ_IO_CREDIT_STALL[9], TCC_EA_WRREQ_LEVEL[9], TCC_EA_WRREQ_DRAM_CREDIT_STALL[10], TCC_EA_WRREQ_GMI_CREDIT_STALL[10], TCC_EA_WRREQ_IO_CREDIT_STALL[10], TCC_EA_WRREQ_LEVEL[10], TCC_EA_WRREQ_DRAM_CREDIT_STALL[11], TCC_EA_WRREQ_GMI_CREDIT_STALL[11], TCC_EA_WRREQ_IO_CREDIT_STALL[11], TCC_EA_WRREQ_LEVEL[11], TCC_EA_WRREQ_DRAM_CREDIT_STALL[12], TCC_EA_WRREQ_GMI_CREDIT_STALL[12], TCC_EA_WRREQ_IO_CREDIT_STALL[12], TCC_EA_WRREQ_LEVEL[12], TCC_EA_WRREQ_DRAM_CREDIT_STALL[13], TCC_EA_WRREQ_GMI_CREDIT_STALL[13], TCC_EA_WRREQ_IO_CREDIT_STALL[13], TCC_EA_WRREQ_LEVEL[13], TCC_EA_WRREQ_DRAM_CREDIT_STALL[14], TCC_EA_WRREQ_GMI_CREDIT_STALL[14], TCC_EA_WRREQ_IO_CREDIT_STALL[14], TCC_EA_WRREQ_LEVEL[14], TCC_EA_WRREQ_DRAM_CREDIT_STALL[15], TCC_EA_WRREQ_GMI_CREDIT_STALL[15], TCC_EA_WRREQ_IO_CREDIT_STALL[15], TCC_EA_WRREQ_LEVEL[15], TCC_EA_WRREQ_DRAM_CREDIT_STALL[16], TCC_EA_WRREQ_GMI_CREDIT_STALL[16], TCC_EA_WRREQ_IO_CREDIT_STALL[16], TCC_EA_WRREQ_LEVEL[16], TCC_EA_WRREQ_DRAM_CREDIT_STALL[17], TCC_EA_WRREQ_GMI_CREDIT_STALL[17], TCC_EA_WRREQ_IO_CREDIT_STALL[17], TCC_EA_WRREQ_LEVEL[17], TCC_EA_WRREQ_DRAM_CREDIT_STALL[18], TCC_EA_WRREQ_GMI_CREDIT_STALL[18], TCC_EA_WRREQ_IO_CREDIT_STALL[18], TCC_EA_WRREQ_LEVEL[18], TCC_EA_WRREQ_DRAM_CREDIT_STALL[19], TCC_EA_WRREQ_GMI_CREDIT_STALL[19], TCC_EA_WRREQ_IO_CREDIT_STALL[19], TCC_EA_WRREQ_LEVEL[19], TCC_EA_WRREQ_DRAM_CREDIT_STALL[20], TCC_EA_WRREQ_GMI_CREDIT_STALL[20], TCC_EA_WRREQ_IO_CREDIT_STALL[20], TCC_EA_WRREQ_LEVEL[20], TCC_EA_WRREQ_DRAM_CREDIT_STALL[21], TCC_EA_WRREQ_GMI_CREDIT_STALL[21], TCC_EA_WRREQ_IO_CREDIT_STALL[21], TCC_EA_WRREQ_LEVEL[21], TCC_EA_WRREQ_DRAM_CREDIT_STALL[22], TCC_EA_WRREQ_GMI_CREDIT_STALL[22], TCC_EA_WRREQ_IO_CREDIT_STALL[22], TCC_EA_WRREQ_LEVEL[22], TCC_EA_WRREQ_DRAM_CREDIT_STALL[23], TCC_EA_WRREQ_GMI_CREDIT_STALL[23], TCC_EA_WRREQ_IO_CREDIT_STALL[23], TCC_EA_WRREQ_LEVEL[23], TCC_EA_WRREQ_DRAM_CREDIT_STALL[24], TCC_EA_WRREQ_GMI_CREDIT_STALL[24], TCC_EA_WRREQ_IO_CREDIT_STALL[24], TCC_EA_WRREQ_LEVEL[24], TCC_EA_WRREQ_DRAM_CREDIT_STALL[25], TCC_EA_WRREQ_GMI_CREDIT_STALL[25], TCC_EA_WRREQ_IO_CREDIT_STALL[25], TCC_EA_WRREQ_LEVEL[25], TCC_EA_WRREQ_DRAM_CREDIT_STALL[26], TCC_EA_WRREQ_GMI_CREDIT_STALL[26], TCC_EA_WRREQ_IO_CREDIT_STALL[26], TCC_EA_WRREQ_LEVEL[26], TCC_EA_WRREQ_DRAM_CREDIT_STALL[27], TCC_EA_WRREQ_GMI_CREDIT_STALL[27], TCC_EA_WRREQ_IO_CREDIT_STALL[27], TCC_EA_WRREQ_LEVEL[27], TCC_EA_WRREQ_DRAM_CREDIT_STALL[28], TCC_EA_WRREQ_GMI_CREDIT_STALL[28], TCC_EA_WRREQ_IO_CREDIT_STALL[28], TCC_EA_WRREQ_LEVEL[28], TCC_EA_WRREQ_DRAM_CREDIT_STALL[29], TCC_EA_WRREQ_GMI_CREDIT_STALL[29], TCC_EA_WRREQ_IO_CREDIT_STALL[29], TCC_EA_WRREQ_LEVEL[29], TCC_EA_WRREQ_DRAM_CREDIT_STALL[30], TCC_EA_WRREQ_GMI_CREDIT_STALL[30], TCC_EA_WRREQ_IO_CREDIT_STALL[30], TCC_EA_WRREQ_LEVEL[30], TCC_EA_WRREQ_DRAM_CREDIT_STALL[31], TCC_EA_WRREQ_GMI_CREDIT_STALL[31], TCC_EA_WRREQ_IO_CREDIT_STALL[31], TCC_EA_WRREQ_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155240_1235184/input0_results_240321_155240
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_14.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_15.txt
|-> [rocprof] RPL: on '240321_155240' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_15.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155240_1235373'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155240_1235373/input0_results_240321_155240'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155240_1235373/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_HIT[0], TCC_MISS[0], TCC_READ[0], TCC_REQ[0], TCC_HIT[1], TCC_MISS[1], TCC_READ[1], TCC_REQ[1], TCC_HIT[2], TCC_MISS[2], TCC_READ[2], TCC_REQ[2], TCC_HIT[3], TCC_MISS[3], TCC_READ[3], TCC_REQ[3], TCC_HIT[4], TCC_MISS[4], TCC_READ[4], TCC_REQ[4], TCC_HIT[5], TCC_MISS[5], TCC_READ[5], TCC_REQ[5], TCC_HIT[6], TCC_MISS[6], TCC_READ[6], TCC_REQ[6], TCC_HIT[7], TCC_MISS[7], TCC_READ[7], TCC_REQ[7], TCC_HIT[8], TCC_MISS[8], TCC_READ[8], TCC_REQ[8], TCC_HIT[9], TCC_MISS[9], TCC_READ[9], TCC_REQ[9], TCC_HIT[10], TCC_MISS[10], TCC_READ[10], TCC_REQ[10], TCC_HIT[11], TCC_MISS[11], TCC_READ[11], TCC_REQ[11], TCC_HIT[12], TCC_MISS[12], TCC_READ[12], TCC_REQ[12], TCC_HIT[13], TCC_MISS[13], TCC_READ[13], TCC_REQ[13], TCC_HIT[14], TCC_MISS[14], TCC_READ[14], TCC_REQ[14], TCC_HIT[15], TCC_MISS[15], TCC_READ[15], TCC_REQ[15], TCC_HIT[16], TCC_MISS[16], TCC_READ[16], TCC_REQ[16], TCC_HIT[17], TCC_MISS[17], TCC_READ[17], TCC_REQ[17], TCC_HIT[18], TCC_MISS[18], TCC_READ[18], TCC_REQ[18], TCC_HIT[19], TCC_MISS[19], TCC_READ[19], TCC_REQ[19], TCC_HIT[20], TCC_MISS[20], TCC_READ[20], TCC_REQ[20], TCC_HIT[21], TCC_MISS[21], TCC_READ[21], TCC_REQ[21], TCC_HIT[22], TCC_MISS[22], TCC_READ[22], TCC_REQ[22], TCC_HIT[23], TCC_MISS[23], TCC_READ[23], TCC_REQ[23], TCC_HIT[24], TCC_MISS[24], TCC_READ[24], TCC_REQ[24], TCC_HIT[25], TCC_MISS[25], TCC_READ[25], TCC_REQ[25], TCC_HIT[26], TCC_MISS[26], TCC_READ[26], TCC_REQ[26], TCC_HIT[27], TCC_MISS[27], TCC_READ[27], TCC_REQ[27], TCC_HIT[28], TCC_MISS[28], TCC_READ[28], TCC_REQ[28], TCC_HIT[29], TCC_MISS[29], TCC_READ[29], TCC_REQ[29], TCC_HIT[30], TCC_MISS[30], TCC_READ[30], TCC_REQ[30], TCC_HIT[31], TCC_MISS[31], TCC_READ[31], TCC_REQ[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155240_1235373/input0_results_240321_155240
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_15.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_16.txt
|-> [rocprof] RPL: on '240321_155241' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_16.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155241_1235565'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155241_1235565/input0_results_240321_155241'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155241_1235565/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 96 metrics
|-> [rocprof] TCC_RW_REQ[0], TCC_TOO_MANY_EA_WRREQS_STALL[0], TCC_WRITE[0], TCC_RW_REQ[1], TCC_TOO_MANY_EA_WRREQS_STALL[1], TCC_WRITE[1], TCC_RW_REQ[2], TCC_TOO_MANY_EA_WRREQS_STALL[2], TCC_WRITE[2], TCC_RW_REQ[3], TCC_TOO_MANY_EA_WRREQS_STALL[3], TCC_WRITE[3], TCC_RW_REQ[4], TCC_TOO_MANY_EA_WRREQS_STALL[4], TCC_WRITE[4], TCC_RW_REQ[5], TCC_TOO_MANY_EA_WRREQS_STALL[5], TCC_WRITE[5], TCC_RW_REQ[6], TCC_TOO_MANY_EA_WRREQS_STALL[6], TCC_WRITE[6], TCC_RW_REQ[7], TCC_TOO_MANY_EA_WRREQS_STALL[7], TCC_WRITE[7], TCC_RW_REQ[8], TCC_TOO_MANY_EA_WRREQS_STALL[8], TCC_WRITE[8], TCC_RW_REQ[9], TCC_TOO_MANY_EA_WRREQS_STALL[9], TCC_WRITE[9], TCC_RW_REQ[10], TCC_TOO_MANY_EA_WRREQS_STALL[10], TCC_WRITE[10], TCC_RW_REQ[11], TCC_TOO_MANY_EA_WRREQS_STALL[11], TCC_WRITE[11], TCC_RW_REQ[12], TCC_TOO_MANY_EA_WRREQS_STALL[12], TCC_WRITE[12], TCC_RW_REQ[13], TCC_TOO_MANY_EA_WRREQS_STALL[13], TCC_WRITE[13], TCC_RW_REQ[14], TCC_TOO_MANY_EA_WRREQS_STALL[14], TCC_WRITE[14], TCC_RW_REQ[15], TCC_TOO_MANY_EA_WRREQS_STALL[15], TCC_WRITE[15], TCC_RW_REQ[16], TCC_TOO_MANY_EA_WRREQS_STALL[16], TCC_WRITE[16], TCC_RW_REQ[17], TCC_TOO_MANY_EA_WRREQS_STALL[17], TCC_WRITE[17], TCC_RW_REQ[18], TCC_TOO_MANY_EA_WRREQS_STALL[18], TCC_WRITE[18], TCC_RW_REQ[19], TCC_TOO_MANY_EA_WRREQS_STALL[19], TCC_WRITE[19], TCC_RW_REQ[20], TCC_TOO_MANY_EA_WRREQS_STALL[20], TCC_WRITE[20], TCC_RW_REQ[21], TCC_TOO_MANY_EA_WRREQS_STALL[21], TCC_WRITE[21], TCC_RW_REQ[22], TCC_TOO_MANY_EA_WRREQS_STALL[22], TCC_WRITE[22], TCC_RW_REQ[23], TCC_TOO_MANY_EA_WRREQS_STALL[23], TCC_WRITE[23], TCC_RW_REQ[24], TCC_TOO_MANY_EA_WRREQS_STALL[24], TCC_WRITE[24], TCC_RW_REQ[25], TCC_TOO_MANY_EA_WRREQS_STALL[25], TCC_WRITE[25], TCC_RW_REQ[26], TCC_TOO_MANY_EA_WRREQS_STALL[26], TCC_WRITE[26], TCC_RW_REQ[27], TCC_TOO_MANY_EA_WRREQS_STALL[27], TCC_WRITE[27], TCC_RW_REQ[28], TCC_TOO_MANY_EA_WRREQS_STALL[28], TCC_WRITE[28], TCC_RW_REQ[29], TCC_TOO_MANY_EA_WRREQS_STALL[29], TCC_WRITE[29], TCC_RW_REQ[30], TCC_TOO_MANY_EA_WRREQS_STALL[30], TCC_WRITE[30], TCC_RW_REQ[31], TCC_TOO_MANY_EA_WRREQS_STALL[31], TCC_WRITE[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155241_1235565/input0_results_240321_155241
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_16.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_2.txt
|-> [rocprof] RPL: on '240321_155242' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_2.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155242_1235753'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155242_1235753/input0_results_240321_155242'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155242_1235753/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 26 metrics
|-> [rocprof] SQC_DCACHE_INPUT_VALID_READYB, SQC_DCACHE_ATOMIC, SQC_DCACHE_REQ_READ_8, SQC_DCACHE_REQ, SQC_DCACHE_HITS, SQC_DCACHE_MISSES, SQC_DCACHE_MISSES_DUPLICATE, SQC_DCACHE_REQ_READ_1, TCP_VOLATILE_sum, TCP_TOTAL_ACCESSES_sum, TCP_TOTAL_READ_sum, TCP_TOTAL_WRITE_sum, TA_BUFFER_ATOMIC_WAVEFRONTS_sum, TA_BUFFER_TOTAL_CYCLES_sum, TD_ATOMIC_WAVEFRONT_sum, TD_STORE_WAVEFRONT_sum, SPI_RA_REQ_NO_ALLOC, SPI_RA_REQ_NO_ALLOC_CSN, CPC_CPC_STAT_STALL, CPC_UTCL1_STALL_ON_TRANSLATION, CPF_CPF_STAT_IDLE, CPF_CPF_TCIU_IDLE, TCC_REQ_sum, TCC_STREAMING_REQ_sum, TCC_HIT_sum, TCC_MISS_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155242_1235753/input0_results_240321_155242
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_2.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_3.txt
|-> [rocprof] RPL: on '240321_155242' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_3.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155242_1235942'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155242_1235942/input0_results_240321_155242'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155242_1235942/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 23 metrics
|-> [rocprof] SQC_DCACHE_REQ_READ_2, SQC_DCACHE_REQ_READ_4, SQ_INSTS_VMEM_WR, SQ_INSTS_VMEM_RD, SQ_INSTS_VMEM, SQ_INSTS_SALU, SQ_INSTS_VSKIPPED, SQ_INSTS_SMEM, TCP_TOTAL_ATOMIC_WITH_RET_sum, TCP_TOTAL_ATOMIC_WITHOUT_RET_sum, TCP_TOTAL_WRITEBACK_INVALIDATES_sum, TCP_TOTAL_CACHE_ACCESSES_sum, TA_BUFFER_COALESCED_READ_CYCLES_sum, TA_BUFFER_COALESCED_WRITE_CYCLES_sum, SPI_RA_RES_STALL_CSN, SPI_RA_TMP_STALL_CSN, CPC_CPC_UTCL2IU_BUSY, CPC_CPC_UTCL2IU_IDLE, CPF_CMP_UTCL1_STALL_ON_TRANSLATION, TCC_READ_sum, TCC_WRITE_sum, TCC_ATOMIC_sum, TCC_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155242_1235942/input0_results_240321_155242
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_3.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_4.txt
|-> [rocprof] RPL: on '240321_155243' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_4.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155243_1236130'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155243_1236130/input0_results_240321_155243'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155243_1236130/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 22 metrics
|-> [rocprof] SQ_INSTS_FLAT, SQ_INSTS_LDS, SQ_INSTS_GDS, SQ_INSTS_EXP_GDS, SQ_INSTS_BRANCH, SQ_INSTS_SENDMSG, SQ_INSTS, SQ_WAIT_ANY, TCP_UTCL1_TRANSLATION_MISS_sum, TCP_UTCL1_TRANSLATION_HIT_sum, TCP_UTCL1_PERMISSION_MISS_sum, TCP_UTCL1_REQUEST_sum, TA_ADDR_STALLED_BY_TC_CYCLES_sum, TA_TOTAL_WAVEFRONTS_sum, SPI_RA_WAVE_SIMD_FULL_CSN, SPI_RA_VGPR_SIMD_FULL_CSN, CPC_CPC_UTCL2IU_STALL, CPC_ME1_BUSY_FOR_PACKET_DECODE, TCC_EA_WRREQ_sum, TCC_EA_WRREQ_64B_sum, TCC_EA_WR_UNCACHED_32B_sum, TCC_EA_WRREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155243_1236130/input0_results_240321_155243
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_4.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_5.txt
|-> [rocprof] RPL: on '240321_155243' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_5.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155243_1236315'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155243_1236315/input0_results_240321_155243'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155243_1236315/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 21 metrics
|-> [rocprof] SQ_WAIT_INST_ANY, SQ_ACTIVE_INST_ANY, SQ_INSTS_VALU, SQ_ACTIVE_INST_VMEM, SQ_ACTIVE_INST_LDS, SQ_ACTIVE_INST_VALU, SQ_ACTIVE_INST_SCA, SQ_ACTIVE_INST_EXP_GDS, TCP_TCP_LATENCY_sum, TCP_TCC_READ_REQ_LATENCY_sum, TCP_TCC_WRITE_REQ_LATENCY_sum, TCP_TCC_READ_REQ_sum, TA_ADDR_STALLED_BY_TD_CYCLES_sum, TA_DATA_STALLED_BY_TC_CYCLES_sum, SPI_RA_SGPR_SIMD_FULL_CSN, SPI_RA_LDS_CU_FULL_CSN, CPC_ME1_DC0_SPI_BUSY, TCC_EA_WRREQ_STALL_sum, TCC_EA_WRREQ_IO_CREDIT_STALL_sum, TCC_EA_WRREQ_GMI_CREDIT_STALL_sum, TCC_EA_WRREQ_DRAM_CREDIT_STALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155243_1236315/input0_results_240321_155243
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_5.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_6.txt
|-> [rocprof] RPL: on '240321_155244' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_6.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155244_1236503'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155244_1236503/input0_results_240321_155244'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155244_1236503/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_ACTIVE_INST_MISC, SQ_ACTIVE_INST_FLAT, SQ_INST_CYCLES_VMEM_WR, SQ_INST_CYCLES_VMEM_RD, SQ_INST_CYCLES_SMEM, SQ_INST_CYCLES_SALU, SQ_THREAD_CYCLES_VALU, SQ_IFETCH, TCP_TCC_WRITE_REQ_sum, TCP_TCC_ATOMIC_WITH_RET_REQ_sum, TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum, TCP_TCC_NC_READ_REQ_sum, TA_FLAT_WAVEFRONTS_sum, TA_FLAT_READ_WAVEFRONTS_sum, SPI_RA_BAR_CU_FULL_CSN, SPI_RA_TGLIM_CU_FULL_CSN, TCC_EA_RDREQ_sum, TCC_EA_RDREQ_32B_sum, TCC_EA_RD_UNCACHED_32B_sum, TCC_EA_RDREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155244_1236503/input0_results_240321_155244
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_6.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_7.txt
|-> [rocprof] RPL: on '240321_155244' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_7.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155244_1236690'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155244_1236690/input0_results_240321_155244'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155244_1236690/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_LDS_BANK_CONFLICT, SQ_LDS_ADDR_CONFLICT, SQ_LDS_UNALIGNED_STALL, SQ_WAVES_EQ_64, SQ_WAVES_LT_64, SQ_WAVES_LT_48, SQ_WAVES_LT_32, SQ_WAVES_LT_16, TCP_TCC_NC_WRITE_REQ_sum, TCP_TCC_NC_ATOMIC_REQ_sum, TCP_TCC_UC_READ_REQ_sum, TCP_TCC_UC_WRITE_REQ_sum, TA_FLAT_WRITE_WAVEFRONTS_sum, TA_FLAT_ATOMIC_WAVEFRONTS_sum, SPI_RA_WVLIM_STALL_CSN, SPI_SWC_CSC_WR, TCC_EA_RDREQ_IO_CREDIT_STALL_sum, TCC_EA_RDREQ_GMI_CREDIT_STALL_sum, TCC_EA_RDREQ_DRAM_CREDIT_STALL_sum, TCC_TAG_STALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155244_1236690/input0_results_240321_155244
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_7.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_8.txt
|-> [rocprof] RPL: on '240321_155245' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_8.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155245_1236877'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155245_1236877/input0_results_240321_155245'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155245_1236877/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 17 metrics
|-> [rocprof] SQ_ITEMS, SQ_LDS_MEM_VIOLATIONS, SQ_LDS_ATOMIC_RETURN, SQ_LDS_IDX_ACTIVE, SQ_WAVES_RESTORED, SQ_WAVES_SAVED, SQ_INSTS_SMEM_NORM, TCP_TCC_UC_ATOMIC_REQ_sum, TCP_TCC_CC_READ_REQ_sum, TCP_TCC_CC_WRITE_REQ_sum, TCP_TCC_CC_ATOMIC_REQ_sum, SPI_VWC_CSC_WR, SPI_RA_BULKY_CU_FULL_CSN, TCC_NORMAL_WRITEBACK_sum, TCC_ALL_TC_OP_WB_WRITEBACK_sum, TCC_NORMAL_EVICT_sum, TCC_ALL_TC_OP_INV_EVICT_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155245_1236877/input0_results_240321_155245
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_8.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_9.txt
|-> [rocprof] RPL: on '240321_155245' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/pmc_perf_9.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155245_1237066'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155245_1237066/input0_results_240321_155245'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155245_1237066/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] TCP_TCC_RW_READ_REQ_sum, TCP_TCC_RW_WRITE_REQ_sum, TCP_TCC_RW_ATOMIC_REQ_sum, TCP_PENDING_STALL_CYCLES_sum, TCC_TOO_MANY_EA_WRREQS_STALL_sum, TCC_EA_ATOMIC_sum, TCC_EA_RDREQ_LEVEL_sum, TCC_EA_WRREQ_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155245_1237066/input0_results_240321_155245
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/pmc_perf_9.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI100/perfmon/timestamps.txt
|-> [rocprof] RPL: on '240321_155246' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI100/perfmon/timestamps.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_155246_1237253'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_155246_1237253/input0_results_240321_155246'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_155246_1237253/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 0 metrics
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_155246_1237253/input0_results_240321_155246
|-> [rocprof] File 'tests/workloads/kernel_substr/MI100/timestamps.csv' is generating
|-> [rocprof]
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID
0,"vecCopy(double*, double*, double*, int, int) ",2
1,"vecCopy(double*, double*, double*, int, int) ",2
2,"vecCopy(double*, double*, double*, int, int) ",2
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1 Dispatch_ID Kernel_Name GPU_ID
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2
Diff de fichier supprimé car une ou plusieurs lignes sont trop longues
+1 -1
Voir le fichier
@@ -1,2 +1,2 @@
workload_name,command,ip_blocks,timestamp,version,hostname,cpu_model,sbios,linux_distro,linux_kernel_version,amd_gpu_kernel_version,cpu_memory,gpu_memory,rocm_version,vbios,compute_partition,memory_partition,gpu_model,gpu_arch,gpu_l1,gpu_l2,cu_per_gpu,simd_per_cu,se_per_gpu,wave_size,workgroup_max_size,max_waves_per_cu,max_sclk,max_mclk,cur_sclk,cur_mclk,total_l2_chan,lds_banks_per_cu,sqc_per_gpu,pipes_per_gpu,hbm_bw,num_xcd
kernel_substr,./tests/vcopy -n 1048576 -b 256 -i 3,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF,Thu 07 Mar 2024 01:07:08 PM (CST),2,t008-007.hpcfund,AMD EPYC 7V13 64-Core Processor,American Megatrends Inc.0602,Rocky Linux 9.1 (Blue Onyx),5.14.0-162.18.1.el9_1.x86_64,,527651100,,6.0.2-115,113-D3431401-100,NA,NA,MI100,gfx908,16,8192,120,4,8,64,1024,40,1502,1200,1502,1200,32,32,64,4,1228.8,1
kernel_substr,./tests/vcopy -n 1048576 -b 256 -i 3,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF,Thu 21 Mar 2024 03:52:33 PM (CDT),2,t007-001.hpcfund,AMD EPYC 7V13 64-Core Processor,American Megatrends Inc.0602,Rocky Linux 9.1 (Blue Onyx),5.14.0-162.18.1.el9_1.x86_64,,527651008,,6.0.2-115,113-D3431401-100,NA,NA,MI100,gfx908,16,8192,120,4,8,64,1024,40,1502,1200,1502,1200,32,32,64,4,1228.8,1
1 workload_name command ip_blocks timestamp version hostname cpu_model sbios linux_distro linux_kernel_version amd_gpu_kernel_version cpu_memory gpu_memory rocm_version vbios compute_partition memory_partition gpu_model gpu_arch gpu_l1 gpu_l2 cu_per_gpu simd_per_cu se_per_gpu wave_size workgroup_max_size max_waves_per_cu max_sclk max_mclk cur_sclk cur_mclk total_l2_chan lds_banks_per_cu sqc_per_gpu pipes_per_gpu hbm_bw num_xcd
2 kernel_substr ./tests/vcopy -n 1048576 -b 256 -i 3 SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF Thu 07 Mar 2024 01:07:08 PM (CST) Thu 21 Mar 2024 03:52:33 PM (CDT) 2 t008-007.hpcfund t007-001.hpcfund AMD EPYC 7V13 64-Core Processor American Megatrends Inc.0602 Rocky Linux 9.1 (Blue Onyx) 5.14.0-162.18.1.el9_1.x86_64 527651100 527651008 6.0.2-115 113-D3431401-100 NA NA MI100 gfx908 16 8192 120 4 8 64 1024 40 1502 1200 1502 1200 32 32 64 4 1228.8 1
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,141682,141682,1048576,256,0,0,8,8,16,64,0x0,0x7feb54db6e80,194175534602589,194175534630063,194175534654063,194175534670638
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,141682,141682,1048576,256,0,0,8,8,16,64,0x0,0x7feb54db6e80,194175534666220,194175534674223,194175534693583,194175534776909
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,141682,141682,1048576,256,0,0,8,8,16,64,0x0,0x7feb54db6e80,194175534704713,194175534779503,194175534798543,194175534799382
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1237413,1237413,1048576,256,0,0,8,8,16,64,0x0,0x7f91afc6cec0,1410100033451844,1410100033478037,1410100033502197,1410100033513600
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1237413,1237413,1048576,256,0,0,8,8,16,64,0x0,0x7f91afc6cec0,1410100033514522,1410100033593877,1410100033612597,1410100033617336
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1237413,1237413,1048576,256,0,0,8,8,16,64,0x0,0x7f91afc6cec0,1410100033624710,1410100033638517,1410100033657077,1410100033658564
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 141682 1237413 141682 1237413 1048576 256 0 0 8 8 16 64 0x0 0x7feb54db6e80 0x7f91afc6cec0 194175534602589 1410100033451844 194175534630063 1410100033478037 194175534654063 1410100033502197 194175534670638 1410100033513600
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 141682 1237413 141682 1237413 1048576 256 0 0 8 8 16 64 0x0 0x7feb54db6e80 0x7f91afc6cec0 194175534666220 1410100033514522 194175534674223 1410100033593877 194175534693583 1410100033612597 194175534776909 1410100033617336
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 141682 1237413 141682 1237413 1048576 256 0 0 8 8 16 64 0x0 0x7feb54db6e80 0x7f91afc6cec0 194175534704713 1410100033624710 194175534779503 1410100033638517 194175534798543 1410100033657077 194175534799382 1410100033658564
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,291276,291276,1048576,256,0,0,8,0,16,64,0x0,0x7f1828160ec0,27737,27737,16384,65536,13722,1525308,198261588061508,198274458325591,198274458345911,198261603806363
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,291276,291276,1048576,256,0,0,8,0,16,64,0x0,0x7f1828160ec0,41413,41413,16384,65536,9460,1048596,198261603830830,198274458365591,198274458381111,198261604082847
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,291276,291276,1048576,256,0,0,8,0,16,64,0x0,0x7f1828160ec0,40275,40275,16384,65536,9096,1048700,198261604112302,198274458443671,198274458459991,198261604256605
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4114786,4114786,1048576,256,0,0,8,0,16,64,0x0,0x7f7efac7cec0,27808,27808,16384,65536,13399,1542552,1411747993352852,1411760919537410,1411760919557090,1411748009210180
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4114786,4114786,1048576,256,0,0,8,0,16,64,0x0,0x7f7efac7cec0,42345,42345,16384,65536,9466,1048672,1411748009231631,1411760919576130,1411760919591970,1411748009550113
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4114786,4114786,1048576,256,0,0,8,0,16,64,0x0,0x7f7efac7cec0,41809,41809,16384,65536,9261,1048648,1411748009579198,1411760919648450,1411760919664930,1411748009739572
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 291276 4114786 291276 4114786 1048576 256 0 0 8 0 16 64 0x0 0x7f1828160ec0 0x7f7efac7cec0 27737 27808 27737 27808 16384 65536 13722 13399 1525308 1542552 198261588061508 1411747993352852 198274458325591 1411760919537410 198274458345911 1411760919557090 198261603806363 1411748009210180
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 291276 4114786 291276 4114786 1048576 256 0 0 8 0 16 64 0x0 0x7f1828160ec0 0x7f7efac7cec0 41413 42345 41413 42345 16384 65536 9460 9466 1048596 1048672 198261603830830 1411748009231631 198274458365591 1411760919576130 198274458381111 1411760919591970 198261604082847 1411748009550113
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 291276 4114786 291276 4114786 1048576 256 0 0 8 0 16 64 0x0 0x7f1828160ec0 0x7f7efac7cec0 40275 41809 40275 41809 16384 65536 9096 9261 1048700 1048648 198261604112302 1411748009579198 198274458443671 1411760919648450 198274458459991 1411760919664930 198261604256605 1411748009739572
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,291478,291478,1048576,256,0,0,8,0,16,64,0x0,0x7f1f8e050ec0,0,0,0,198262073743594,198274458325591,198274458345911,198262089831408
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,291478,291478,1048576,256,0,0,8,0,16,64,0x0,0x7f1f8e050ec0,0,0,0,198262089848571,198274458365591,198274458381111,198262090150562
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,291478,291478,1048576,256,0,0,8,0,16,64,0x0,0x7f1f8e050ec0,0,0,0,198262090176932,198274458443671,198274458459991,198262090341423
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4114972,4114972,1048576,256,0,0,8,0,16,64,0x0,0x7fe2c102cec0,0,0,0,1411748492261360,1411760919537410,1411760919557090,1411748508646486
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4114972,4114972,1048576,256,0,0,8,0,16,64,0x0,0x7fe2c102cec0,0,0,0,1411748508665692,1411760919576130,1411760919591970,1411748508945812
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4114972,4114972,1048576,256,0,0,8,0,16,64,0x0,0x7fe2c102cec0,0,0,0,1411748508971250,1411760919648450,1411760919664930,1411748509120252
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 291478 4114972 291478 4114972 1048576 256 0 0 8 0 16 64 0x0 0x7f1f8e050ec0 0x7fe2c102cec0 0 0 0 198262073743594 1411748492261360 198274458325591 1411760919537410 198274458345911 1411760919557090 198262089831408 1411748508646486
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 291478 4114972 291478 4114972 1048576 256 0 0 8 0 16 64 0x0 0x7f1f8e050ec0 0x7fe2c102cec0 0 0 0 198262089848571 1411748508665692 198274458365591 1411760919576130 198274458381111 1411760919591970 198262090150562 1411748508945812
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 291478 4114972 291478 4114972 1048576 256 0 0 8 0 16 64 0x0 0x7f1f8e050ec0 0x7fe2c102cec0 0 0 0 198262090176932 1411748508971250 198274458443671 1411760919648450 198274458459991 1411760919664930 198262090341423 1411748509120252
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,291685,291685,1048576,256,0,0,8,0,16,64,0x0,0x7fddd2fc4ec0,65536,187092,20917768,198262558091736,198274458325591,198274458345911,198262574290691
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,291685,291685,1048576,256,0,0,8,0,16,64,0x0,0x7fddd2fc4ec0,65536,245200,27582200,198262574309346,198274458365591,198274458381111,198262574602852
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,291685,291685,1048576,256,0,0,8,0,16,64,0x0,0x7fddd2fc4ec0,65536,263490,29654280,198262574629011,198274458443671,198274458459991,198262574756552
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4115172,4115172,1048576,256,0,0,8,0,16,64,0x0,0x7f98c6ce0ec0,65536,191466,21343568,1411748975548556,1411760919537410,1411760919557090,1411748991614648
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4115172,4115172,1048576,256,0,0,8,0,16,64,0x0,0x7f98c6ce0ec0,65536,266474,29803832,1411748991632112,1411760919576130,1411760919591970,1411748991928322
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4115172,4115172,1048576,256,0,0,8,0,16,64,0x0,0x7f98c6ce0ec0,65536,269860,30194104,1411748991954391,1411760919648450,1411760919664930,1411748992109995
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 291685 4115172 291685 4115172 1048576 256 0 0 8 0 16 64 0x0 0x7fddd2fc4ec0 0x7f98c6ce0ec0 65536 187092 191466 20917768 21343568 198262558091736 1411748975548556 198274458325591 1411760919537410 198274458345911 1411760919557090 198262574290691 1411748991614648
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 291685 4115172 291685 4115172 1048576 256 0 0 8 0 16 64 0x0 0x7fddd2fc4ec0 0x7f98c6ce0ec0 65536 245200 266474 27582200 29803832 198262574309346 1411748991632112 198274458365591 1411760919576130 198274458381111 1411760919591970 198262574602852 1411748991928322
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 291685 4115172 291685 4115172 1048576 256 0 0 8 0 16 64 0x0 0x7fddd2fc4ec0 0x7f98c6ce0ec0 65536 263490 269860 29654280 30194104 198262574629011 1411748991954391 198274458443671 1411760919648450 198274458459991 1411760919664930 198262574756552 1411748992109995
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,291870,291870,1048576,256,0,0,8,0,16,64,0x0,0x7f5a50b30ec0,32768,288882,32369320,198263038694323,198274458325591,198274458345911,198263054914017
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,291870,291870,1048576,256,0,0,8,0,16,64,0x0,0x7f5a50b30ec0,32768,596098,66767128,198263054932602,198274458365591,198274458381111,198263055228652
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,291870,291870,1048576,256,0,0,8,0,16,64,0x0,0x7f5a50b30ec0,32768,572975,64171428,198263055254551,198274458443671,198274458459991,198263055412490
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4115370,4115370,1048576,256,0,0,8,0,16,64,0x0,0x7f400bee0ec0,32768,293469,32855948,1411749461257651,1411760919537410,1411760919557090,1411749477699425
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4115370,4115370,1048576,256,0,0,8,0,16,64,0x0,0x7f400bee0ec0,32768,562268,62950004,1411749477717519,1411760919576130,1411760919591970,1411749478030080
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4115370,4115370,1048576,256,0,0,8,0,16,64,0x0,0x7f400bee0ec0,32768,583160,65318136,1411749478056540,1411760919648450,1411760919664930,1411749478223946
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 291870 4115370 291870 4115370 1048576 256 0 0 8 0 16 64 0x0 0x7f5a50b30ec0 0x7f400bee0ec0 32768 288882 293469 32369320 32855948 198263038694323 1411749461257651 198274458325591 1411760919537410 198274458345911 1411760919557090 198263054914017 1411749477699425
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 291870 4115370 291870 4115370 1048576 256 0 0 8 0 16 64 0x0 0x7f5a50b30ec0 0x7f400bee0ec0 32768 596098 562268 66767128 62950004 198263054932602 1411749477717519 198274458365591 1411760919576130 198274458381111 1411760919591970 198263055228652 1411749478030080
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 291870 4115370 291870 4115370 1048576 256 0 0 8 0 16 64 0x0 0x7f5a50b30ec0 0x7f400bee0ec0 32768 572975 583160 64171428 65318136 198263055254551 1411749478056540 198274458443671 1411760919648450 198274458459991 1411760919664930 198263055412490 1411749478223946
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) ",2,0,0,292068,292068,1048576,256,0,0,8,0,16,64,0x0,0x7f06a0314ec0,27371,27371,10912,218976,16384,10605927,126253,0,42923692,198263526253279,198274458325591,198274458345911,198263542014035
1,"vecCopy(double*, double*, double*, int, int) ",2,0,2,292068,292068,1048576,256,0,0,8,0,16,64,0x0,0x7f06a0314ec0,41749,41749,13458,334000,16384,19718416,225192,0,79338420,198263542042509,198274458365591,198274458381111,198263542340142
2,"vecCopy(double*, double*, double*, int, int) ",2,0,4,292068,292068,1048576,256,0,0,8,0,16,64,0x0,0x7f06a0314ec0,41423,41423,13486,331392,16384,19225946,220329,0,77380152,198263542374808,198274458443671,198274458459991,198263542550089
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4115572,4115572,1048576,256,0,0,8,0,16,64,0x0,0x7f08ef3d8ec0,28866,28866,11064,230936,16384,11619166,137356,0,46984496,1411749950842366,1411760919537410,1411760919557090,1411749966997337
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4115572,4115572,1048576,256,0,0,8,0,16,64,0x0,0x7f08ef3d8ec0,41555,41555,13750,332448,16384,19544648,223581,0,78671448,1411749967023126,1411760919576130,1411760919591970,1411749967377907
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4115572,4115572,1048576,256,0,0,8,0,16,64,0x0,0x7f08ef3d8ec0,42294,42294,14213,338360,16384,19749746,227566,0,79473776,1411749967412673,1411760919648450,1411760919664930,1411749967581332
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 292068 4115572 292068 4115572 1048576 256 0 0 8 0 16 64 0x0 0x7f06a0314ec0 0x7f08ef3d8ec0 27371 28866 27371 28866 10912 11064 218976 230936 16384 10605927 11619166 126253 137356 0 42923692 46984496 198263526253279 1411749950842366 198274458325591 1411760919537410 198274458345911 1411760919557090 198263542014035 1411749966997337
3 1 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 292068 4115572 292068 4115572 1048576 256 0 0 8 0 16 64 0x0 0x7f06a0314ec0 0x7f08ef3d8ec0 41749 41555 41749 41555 13458 13750 334000 332448 16384 19718416 19544648 225192 223581 0 79338420 78671448 198263542042509 1411749967023126 198274458365591 1411760919576130 198274458381111 1411760919591970 198263542340142 1411749967377907
4 2 vecCopy(double*, double*, double*, int, int) vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 292068 4115572 292068 4115572 1048576 256 0 0 8 0 16 64 0x0 0x7f06a0314ec0 0x7f08ef3d8ec0 41423 42294 41423 42294 13486 14213 331392 338360 16384 19225946 19749746 220329 227566 0 77380152 79473776 198263542374808 1411749967412673 198274458443671 1411760919648450 198274458459991 1411760919664930 198263542550089 1411749967581332
+764
Voir le fichier
@@ -0,0 +1,764 @@
Omniperf version: 2.0.0-RC1
Profiler choice: rocprofv1
Path: /home1/josantos/omniperf/tests/workloads/kernel_substr/MI200
Target: MI200
Command: ./tests/vcopy -n 1048576 -b 256 -i 3
Kernel Selection: ['vecCopy']
Dispatch Selection: None
IP Blocks: All
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Collecting Performance Counters
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/SQ_IFETCH_LEVEL.txt
|-> [rocprof] RPL: on '240321_162015' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/SQ_IFETCH_LEVEL.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162015_4114626'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162015_4114626/input0_results_240321_162015'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162015_4114626/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 6 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, SQ_WAVES, SQ_IFETCH, SQ_IFETCH_LEVEL, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162015_4114626/input0_results_240321_162015
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/SQ_IFETCH_LEVEL.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/SQ_INST_LEVEL_LDS.txt
|-> [rocprof] RPL: on '240321_162016' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/SQ_INST_LEVEL_LDS.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162016_4114812'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162016_4114812/input0_results_240321_162016'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162016_4114812/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_LDS, SQ_INST_LEVEL_LDS, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162016_4114812/input0_results_240321_162016
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/SQ_INST_LEVEL_LDS.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/SQ_INST_LEVEL_SMEM.txt
|-> [rocprof] RPL: on '240321_162016' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/SQ_INST_LEVEL_SMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162016_4115012'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162016_4115012/input0_results_240321_162016'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162016_4115012/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_SMEM, SQ_INST_LEVEL_SMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162016_4115012/input0_results_240321_162016
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/SQ_INST_LEVEL_SMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/SQ_INST_LEVEL_VMEM.txt
|-> [rocprof] RPL: on '240321_162017' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/SQ_INST_LEVEL_VMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162017_4115210'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162017_4115210/input0_results_240321_162017'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162017_4115210/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_VMEM, SQ_INST_LEVEL_VMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162017_4115210/input0_results_240321_162017
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/SQ_INST_LEVEL_VMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/SQ_LEVEL_WAVES.txt
|-> [rocprof] RPL: on '240321_162017' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/SQ_LEVEL_WAVES.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162017_4115412'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162017_4115412/input0_results_240321_162017'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162017_4115412/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 9 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, CPC_ME1_BUSY_FOR_PACKET_DECODE, SQ_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQ_BUSY_CYCLES, SQ_LEVEL_WAVES, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162017_4115412/input0_results_240321_162017
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/SQ_LEVEL_WAVES.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_0.txt
|-> [rocprof] RPL: on '240321_162018' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_0.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162018_4115614'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162018_4115614/input0_results_240321_162018'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162018_4115614/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 28 metrics
|-> [rocprof] SQ_CYCLES, SQ_BUSY_CYCLES, SQ_WAVES, SQ_INSTS_VALU_CVT, SQ_INSTS_VMEM_WR, SQ_INSTS_VMEM_RD, SQ_INSTS_VMEM, SQ_INSTS_SALU, GRBM_COUNT, GRBM_GUI_ACTIVE, TCP_GATE_EN1_sum, TCP_GATE_EN2_sum, TCP_TD_TCP_STALL_CYCLES_sum, TCP_TCR_TCP_STALL_CYCLES_sum, TA_TA_BUSY_sum, TA_BUFFER_WAVEFRONTS_sum, TD_TD_BUSY_sum, TD_TC_STALL_sum, SPI_CSN_WINDOW_VALID, SPI_CSN_BUSY, CPC_CPC_STAT_BUSY, CPC_CPC_STAT_IDLE, CPF_CPF_STAT_BUSY, CPF_CPF_STAT_STALL, TCC_CYCLE_sum, TCC_BUSY_sum, TCC_PROBE_sum, TCC_PROBE_ALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162018_4115614/input0_results_240321_162018
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_0.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_1.txt
|-> [rocprof] RPL: on '240321_162018' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_1.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162018_4115800'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162018_4115800/input0_results_240321_162018'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162018_4115800/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 27 metrics
|-> [rocprof] SQ_INSTS_VSKIPPED, SQ_INSTS, SQ_INSTS_VALU, SQ_INSTS_VALU_ADD_F16, SQ_INSTS_VALU_MUL_F16, SQ_INSTS_VALU_FMA_F16, SQ_INSTS_VALU_TRANS_F16, SQ_INSTS_VALU_ADD_F32, GRBM_SPI_BUSY, TCP_READ_TAGCONFLICT_STALL_CYCLES_sum, TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum, TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum, TCP_TA_TCP_STATE_READ_sum, TA_BUFFER_READ_WAVEFRONTS_sum, TA_BUFFER_WRITE_WAVEFRONTS_sum, TD_SPI_STALL_sum, TD_LOAD_WAVEFRONT_sum, SPI_CSN_NUM_THREADGROUPS, SPI_CSN_WAVE, CPC_CPC_TCIU_BUSY, CPC_CPC_TCIU_IDLE, CPF_CPF_TCIU_BUSY, CPF_CPF_TCIU_STALL, TCC_NC_REQ_sum, TCC_UC_REQ_sum, TCC_CC_REQ_sum, TCC_RW_REQ_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162018_4115800/input0_results_240321_162018
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_1.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_10.txt
|-> [rocprof] RPL: on '240321_162019' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_10.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162019_4116004'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162019_4116004/input0_results_240321_162019'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162019_4116004/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] SQC_TC_DATA_WRITE_REQ, SQC_TC_DATA_ATOMIC_REQ, SQC_TC_STALL, SQC_TC_REQ, SQC_DCACHE_REQ_READ_16, SQC_ICACHE_REQ, SQC_ICACHE_HITS, SQC_ICACHE_MISSES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162019_4116004/input0_results_240321_162019
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_10.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_11.txt
|-> [rocprof] RPL: on '240321_162019' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_11.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162019_4116192'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162019_4116192/input0_results_240321_162019'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162019_4116192/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] SQC_ICACHE_MISSES_DUPLICATE, SQC_DCACHE_INPUT_VALID_READYB, SQC_DCACHE_ATOMIC, SQC_DCACHE_REQ_READ_8, SQC_DCACHE_REQ, SQC_DCACHE_HITS, SQC_DCACHE_MISSES, SQC_DCACHE_MISSES_DUPLICATE
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162019_4116192/input0_results_240321_162019
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_11.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_12.txt
|-> [rocprof] RPL: on '240321_162020' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_12.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162020_4116376'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162020_4116376/input0_results_240321_162020'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162020_4116376/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQC_DCACHE_REQ_READ_1, SQC_DCACHE_REQ_READ_2, SQC_DCACHE_REQ_READ_4
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162020_4116376/input0_results_240321_162020
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_12.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_13.txt
|-> [rocprof] RPL: on '240321_162020' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_13.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162020_4116562'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162020_4116562/input0_results_240321_162020'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162020_4116562/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_ATOMIC[0], TCC_CYCLE[0], TCC_EA_ATOMIC[0], TCC_EA_ATOMIC_LEVEL[0], TCC_ATOMIC[1], TCC_CYCLE[1], TCC_EA_ATOMIC[1], TCC_EA_ATOMIC_LEVEL[1], TCC_ATOMIC[2], TCC_CYCLE[2], TCC_EA_ATOMIC[2], TCC_EA_ATOMIC_LEVEL[2], TCC_ATOMIC[3], TCC_CYCLE[3], TCC_EA_ATOMIC[3], TCC_EA_ATOMIC_LEVEL[3], TCC_ATOMIC[4], TCC_CYCLE[4], TCC_EA_ATOMIC[4], TCC_EA_ATOMIC_LEVEL[4], TCC_ATOMIC[5], TCC_CYCLE[5], TCC_EA_ATOMIC[5], TCC_EA_ATOMIC_LEVEL[5], TCC_ATOMIC[6], TCC_CYCLE[6], TCC_EA_ATOMIC[6], TCC_EA_ATOMIC_LEVEL[6], TCC_ATOMIC[7], TCC_CYCLE[7], TCC_EA_ATOMIC[7], TCC_EA_ATOMIC_LEVEL[7], TCC_ATOMIC[8], TCC_CYCLE[8], TCC_EA_ATOMIC[8], TCC_EA_ATOMIC_LEVEL[8], TCC_ATOMIC[9], TCC_CYCLE[9], TCC_EA_ATOMIC[9], TCC_EA_ATOMIC_LEVEL[9], TCC_ATOMIC[10], TCC_CYCLE[10], TCC_EA_ATOMIC[10], TCC_EA_ATOMIC_LEVEL[10], TCC_ATOMIC[11], TCC_CYCLE[11], TCC_EA_ATOMIC[11], TCC_EA_ATOMIC_LEVEL[11], TCC_ATOMIC[12], TCC_CYCLE[12], TCC_EA_ATOMIC[12], TCC_EA_ATOMIC_LEVEL[12], TCC_ATOMIC[13], TCC_CYCLE[13], TCC_EA_ATOMIC[13], TCC_EA_ATOMIC_LEVEL[13], TCC_ATOMIC[14], TCC_CYCLE[14], TCC_EA_ATOMIC[14], TCC_EA_ATOMIC_LEVEL[14], TCC_ATOMIC[15], TCC_CYCLE[15], TCC_EA_ATOMIC[15], TCC_EA_ATOMIC_LEVEL[15], TCC_ATOMIC[16], TCC_CYCLE[16], TCC_EA_ATOMIC[16], TCC_EA_ATOMIC_LEVEL[16], TCC_ATOMIC[17], TCC_CYCLE[17], TCC_EA_ATOMIC[17], TCC_EA_ATOMIC_LEVEL[17], TCC_ATOMIC[18], TCC_CYCLE[18], TCC_EA_ATOMIC[18], TCC_EA_ATOMIC_LEVEL[18], TCC_ATOMIC[19], TCC_CYCLE[19], TCC_EA_ATOMIC[19], TCC_EA_ATOMIC_LEVEL[19], TCC_ATOMIC[20], TCC_CYCLE[20], TCC_EA_ATOMIC[20], TCC_EA_ATOMIC_LEVEL[20], TCC_ATOMIC[21], TCC_CYCLE[21], TCC_EA_ATOMIC[21], TCC_EA_ATOMIC_LEVEL[21], TCC_ATOMIC[22], TCC_CYCLE[22], TCC_EA_ATOMIC[22], TCC_EA_ATOMIC_LEVEL[22], TCC_ATOMIC[23], TCC_CYCLE[23], TCC_EA_ATOMIC[23], TCC_EA_ATOMIC_LEVEL[23], TCC_ATOMIC[24], TCC_CYCLE[24], TCC_EA_ATOMIC[24], TCC_EA_ATOMIC_LEVEL[24], TCC_ATOMIC[25], TCC_CYCLE[25], TCC_EA_ATOMIC[25], TCC_EA_ATOMIC_LEVEL[25], TCC_ATOMIC[26], TCC_CYCLE[26], TCC_EA_ATOMIC[26], TCC_EA_ATOMIC_LEVEL[26], TCC_ATOMIC[27], TCC_CYCLE[27], TCC_EA_ATOMIC[27], TCC_EA_ATOMIC_LEVEL[27], TCC_ATOMIC[28], TCC_CYCLE[28], TCC_EA_ATOMIC[28], TCC_EA_ATOMIC_LEVEL[28], TCC_ATOMIC[29], TCC_CYCLE[29], TCC_EA_ATOMIC[29], TCC_EA_ATOMIC_LEVEL[29], TCC_ATOMIC[30], TCC_CYCLE[30], TCC_EA_ATOMIC[30], TCC_EA_ATOMIC_LEVEL[30], TCC_ATOMIC[31], TCC_CYCLE[31], TCC_EA_ATOMIC[31], TCC_EA_ATOMIC_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162020_4116562/input0_results_240321_162020
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_13.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_14.txt
|-> [rocprof] RPL: on '240321_162021' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_14.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162021_4116749'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162021_4116749/input0_results_240321_162021'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162021_4116749/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ[0], TCC_EA_RDREQ_32B[0], TCC_EA_RDREQ_DRAM_CREDIT_STALL[0], TCC_EA_RDREQ_GMI_CREDIT_STALL[0], TCC_EA_RDREQ[1], TCC_EA_RDREQ_32B[1], TCC_EA_RDREQ_DRAM_CREDIT_STALL[1], TCC_EA_RDREQ_GMI_CREDIT_STALL[1], TCC_EA_RDREQ[2], TCC_EA_RDREQ_32B[2], TCC_EA_RDREQ_DRAM_CREDIT_STALL[2], TCC_EA_RDREQ_GMI_CREDIT_STALL[2], TCC_EA_RDREQ[3], TCC_EA_RDREQ_32B[3], TCC_EA_RDREQ_DRAM_CREDIT_STALL[3], TCC_EA_RDREQ_GMI_CREDIT_STALL[3], TCC_EA_RDREQ[4], TCC_EA_RDREQ_32B[4], TCC_EA_RDREQ_DRAM_CREDIT_STALL[4], TCC_EA_RDREQ_GMI_CREDIT_STALL[4], TCC_EA_RDREQ[5], TCC_EA_RDREQ_32B[5], TCC_EA_RDREQ_DRAM_CREDIT_STALL[5], TCC_EA_RDREQ_GMI_CREDIT_STALL[5], TCC_EA_RDREQ[6], TCC_EA_RDREQ_32B[6], TCC_EA_RDREQ_DRAM_CREDIT_STALL[6], TCC_EA_RDREQ_GMI_CREDIT_STALL[6], TCC_EA_RDREQ[7], TCC_EA_RDREQ_32B[7], TCC_EA_RDREQ_DRAM_CREDIT_STALL[7], TCC_EA_RDREQ_GMI_CREDIT_STALL[7], TCC_EA_RDREQ[8], TCC_EA_RDREQ_32B[8], TCC_EA_RDREQ_DRAM_CREDIT_STALL[8], TCC_EA_RDREQ_GMI_CREDIT_STALL[8], TCC_EA_RDREQ[9], TCC_EA_RDREQ_32B[9], TCC_EA_RDREQ_DRAM_CREDIT_STALL[9], TCC_EA_RDREQ_GMI_CREDIT_STALL[9], TCC_EA_RDREQ[10], TCC_EA_RDREQ_32B[10], TCC_EA_RDREQ_DRAM_CREDIT_STALL[10], TCC_EA_RDREQ_GMI_CREDIT_STALL[10], TCC_EA_RDREQ[11], TCC_EA_RDREQ_32B[11], TCC_EA_RDREQ_DRAM_CREDIT_STALL[11], TCC_EA_RDREQ_GMI_CREDIT_STALL[11], TCC_EA_RDREQ[12], TCC_EA_RDREQ_32B[12], TCC_EA_RDREQ_DRAM_CREDIT_STALL[12], TCC_EA_RDREQ_GMI_CREDIT_STALL[12], TCC_EA_RDREQ[13], TCC_EA_RDREQ_32B[13], TCC_EA_RDREQ_DRAM_CREDIT_STALL[13], TCC_EA_RDREQ_GMI_CREDIT_STALL[13], TCC_EA_RDREQ[14], TCC_EA_RDREQ_32B[14], TCC_EA_RDREQ_DRAM_CREDIT_STALL[14], TCC_EA_RDREQ_GMI_CREDIT_STALL[14], TCC_EA_RDREQ[15], TCC_EA_RDREQ_32B[15], TCC_EA_RDREQ_DRAM_CREDIT_STALL[15], TCC_EA_RDREQ_GMI_CREDIT_STALL[15], TCC_EA_RDREQ[16], TCC_EA_RDREQ_32B[16], TCC_EA_RDREQ_DRAM_CREDIT_STALL[16], TCC_EA_RDREQ_GMI_CREDIT_STALL[16], TCC_EA_RDREQ[17], TCC_EA_RDREQ_32B[17], TCC_EA_RDREQ_DRAM_CREDIT_STALL[17], TCC_EA_RDREQ_GMI_CREDIT_STALL[17], TCC_EA_RDREQ[18], TCC_EA_RDREQ_32B[18], TCC_EA_RDREQ_DRAM_CREDIT_STALL[18], TCC_EA_RDREQ_GMI_CREDIT_STALL[18], TCC_EA_RDREQ[19], TCC_EA_RDREQ_32B[19], TCC_EA_RDREQ_DRAM_CREDIT_STALL[19], TCC_EA_RDREQ_GMI_CREDIT_STALL[19], TCC_EA_RDREQ[20], TCC_EA_RDREQ_32B[20], TCC_EA_RDREQ_DRAM_CREDIT_STALL[20], TCC_EA_RDREQ_GMI_CREDIT_STALL[20], TCC_EA_RDREQ[21], TCC_EA_RDREQ_32B[21], TCC_EA_RDREQ_DRAM_CREDIT_STALL[21], TCC_EA_RDREQ_GMI_CREDIT_STALL[21], TCC_EA_RDREQ[22], TCC_EA_RDREQ_32B[22], TCC_EA_RDREQ_DRAM_CREDIT_STALL[22], TCC_EA_RDREQ_GMI_CREDIT_STALL[22], TCC_EA_RDREQ[23], TCC_EA_RDREQ_32B[23], TCC_EA_RDREQ_DRAM_CREDIT_STALL[23], TCC_EA_RDREQ_GMI_CREDIT_STALL[23], TCC_EA_RDREQ[24], TCC_EA_RDREQ_32B[24], TCC_EA_RDREQ_DRAM_CREDIT_STALL[24], TCC_EA_RDREQ_GMI_CREDIT_STALL[24], TCC_EA_RDREQ[25], TCC_EA_RDREQ_32B[25], TCC_EA_RDREQ_DRAM_CREDIT_STALL[25], TCC_EA_RDREQ_GMI_CREDIT_STALL[25], TCC_EA_RDREQ[26], TCC_EA_RDREQ_32B[26], TCC_EA_RDREQ_DRAM_CREDIT_STALL[26], TCC_EA_RDREQ_GMI_CREDIT_STALL[26], TCC_EA_RDREQ[27], TCC_EA_RDREQ_32B[27], TCC_EA_RDREQ_DRAM_CREDIT_STALL[27], TCC_EA_RDREQ_GMI_CREDIT_STALL[27], TCC_EA_RDREQ[28], TCC_EA_RDREQ_32B[28], TCC_EA_RDREQ_DRAM_CREDIT_STALL[28], TCC_EA_RDREQ_GMI_CREDIT_STALL[28], TCC_EA_RDREQ[29], TCC_EA_RDREQ_32B[29], TCC_EA_RDREQ_DRAM_CREDIT_STALL[29], TCC_EA_RDREQ_GMI_CREDIT_STALL[29], TCC_EA_RDREQ[30], TCC_EA_RDREQ_32B[30], TCC_EA_RDREQ_DRAM_CREDIT_STALL[30], TCC_EA_RDREQ_GMI_CREDIT_STALL[30], TCC_EA_RDREQ[31], TCC_EA_RDREQ_32B[31], TCC_EA_RDREQ_DRAM_CREDIT_STALL[31], TCC_EA_RDREQ_GMI_CREDIT_STALL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162021_4116749/input0_results_240321_162021
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_14.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_15.txt
|-> [rocprof] RPL: on '240321_162022' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_15.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162022_4116935'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162022_4116935/input0_results_240321_162022'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162022_4116935/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ_IO_CREDIT_STALL[0], TCC_EA_RDREQ_LEVEL[0], TCC_EA_WRREQ[0], TCC_EA_WRREQ_64B[0], TCC_EA_RDREQ_IO_CREDIT_STALL[1], TCC_EA_RDREQ_LEVEL[1], TCC_EA_WRREQ[1], TCC_EA_WRREQ_64B[1], TCC_EA_RDREQ_IO_CREDIT_STALL[2], TCC_EA_RDREQ_LEVEL[2], TCC_EA_WRREQ[2], TCC_EA_WRREQ_64B[2], TCC_EA_RDREQ_IO_CREDIT_STALL[3], TCC_EA_RDREQ_LEVEL[3], TCC_EA_WRREQ[3], TCC_EA_WRREQ_64B[3], TCC_EA_RDREQ_IO_CREDIT_STALL[4], TCC_EA_RDREQ_LEVEL[4], TCC_EA_WRREQ[4], TCC_EA_WRREQ_64B[4], TCC_EA_RDREQ_IO_CREDIT_STALL[5], TCC_EA_RDREQ_LEVEL[5], TCC_EA_WRREQ[5], TCC_EA_WRREQ_64B[5], TCC_EA_RDREQ_IO_CREDIT_STALL[6], TCC_EA_RDREQ_LEVEL[6], TCC_EA_WRREQ[6], TCC_EA_WRREQ_64B[6], TCC_EA_RDREQ_IO_CREDIT_STALL[7], TCC_EA_RDREQ_LEVEL[7], TCC_EA_WRREQ[7], TCC_EA_WRREQ_64B[7], TCC_EA_RDREQ_IO_CREDIT_STALL[8], TCC_EA_RDREQ_LEVEL[8], TCC_EA_WRREQ[8], TCC_EA_WRREQ_64B[8], TCC_EA_RDREQ_IO_CREDIT_STALL[9], TCC_EA_RDREQ_LEVEL[9], TCC_EA_WRREQ[9], TCC_EA_WRREQ_64B[9], TCC_EA_RDREQ_IO_CREDIT_STALL[10], TCC_EA_RDREQ_LEVEL[10], TCC_EA_WRREQ[10], TCC_EA_WRREQ_64B[10], TCC_EA_RDREQ_IO_CREDIT_STALL[11], TCC_EA_RDREQ_LEVEL[11], TCC_EA_WRREQ[11], TCC_EA_WRREQ_64B[11], TCC_EA_RDREQ_IO_CREDIT_STALL[12], TCC_EA_RDREQ_LEVEL[12], TCC_EA_WRREQ[12], TCC_EA_WRREQ_64B[12], TCC_EA_RDREQ_IO_CREDIT_STALL[13], TCC_EA_RDREQ_LEVEL[13], TCC_EA_WRREQ[13], TCC_EA_WRREQ_64B[13], TCC_EA_RDREQ_IO_CREDIT_STALL[14], TCC_EA_RDREQ_LEVEL[14], TCC_EA_WRREQ[14], TCC_EA_WRREQ_64B[14], TCC_EA_RDREQ_IO_CREDIT_STALL[15], TCC_EA_RDREQ_LEVEL[15], TCC_EA_WRREQ[15], TCC_EA_WRREQ_64B[15], TCC_EA_RDREQ_IO_CREDIT_STALL[16], TCC_EA_RDREQ_LEVEL[16], TCC_EA_WRREQ[16], TCC_EA_WRREQ_64B[16], TCC_EA_RDREQ_IO_CREDIT_STALL[17], TCC_EA_RDREQ_LEVEL[17], TCC_EA_WRREQ[17], TCC_EA_WRREQ_64B[17], TCC_EA_RDREQ_IO_CREDIT_STALL[18], TCC_EA_RDREQ_LEVEL[18], TCC_EA_WRREQ[18], TCC_EA_WRREQ_64B[18], TCC_EA_RDREQ_IO_CREDIT_STALL[19], TCC_EA_RDREQ_LEVEL[19], TCC_EA_WRREQ[19], TCC_EA_WRREQ_64B[19], TCC_EA_RDREQ_IO_CREDIT_STALL[20], TCC_EA_RDREQ_LEVEL[20], TCC_EA_WRREQ[20], TCC_EA_WRREQ_64B[20], TCC_EA_RDREQ_IO_CREDIT_STALL[21], TCC_EA_RDREQ_LEVEL[21], TCC_EA_WRREQ[21], TCC_EA_WRREQ_64B[21], TCC_EA_RDREQ_IO_CREDIT_STALL[22], TCC_EA_RDREQ_LEVEL[22], TCC_EA_WRREQ[22], TCC_EA_WRREQ_64B[22], TCC_EA_RDREQ_IO_CREDIT_STALL[23], TCC_EA_RDREQ_LEVEL[23], TCC_EA_WRREQ[23], TCC_EA_WRREQ_64B[23], TCC_EA_RDREQ_IO_CREDIT_STALL[24], TCC_EA_RDREQ_LEVEL[24], TCC_EA_WRREQ[24], TCC_EA_WRREQ_64B[24], TCC_EA_RDREQ_IO_CREDIT_STALL[25], TCC_EA_RDREQ_LEVEL[25], TCC_EA_WRREQ[25], TCC_EA_WRREQ_64B[25], TCC_EA_RDREQ_IO_CREDIT_STALL[26], TCC_EA_RDREQ_LEVEL[26], TCC_EA_WRREQ[26], TCC_EA_WRREQ_64B[26], TCC_EA_RDREQ_IO_CREDIT_STALL[27], TCC_EA_RDREQ_LEVEL[27], TCC_EA_WRREQ[27], TCC_EA_WRREQ_64B[27], TCC_EA_RDREQ_IO_CREDIT_STALL[28], TCC_EA_RDREQ_LEVEL[28], TCC_EA_WRREQ[28], TCC_EA_WRREQ_64B[28], TCC_EA_RDREQ_IO_CREDIT_STALL[29], TCC_EA_RDREQ_LEVEL[29], TCC_EA_WRREQ[29], TCC_EA_WRREQ_64B[29], TCC_EA_RDREQ_IO_CREDIT_STALL[30], TCC_EA_RDREQ_LEVEL[30], TCC_EA_WRREQ[30], TCC_EA_WRREQ_64B[30], TCC_EA_RDREQ_IO_CREDIT_STALL[31], TCC_EA_RDREQ_LEVEL[31], TCC_EA_WRREQ[31], TCC_EA_WRREQ_64B[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162022_4116935/input0_results_240321_162022
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_15.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_16.txt
|-> [rocprof] RPL: on '240321_162022' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_16.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162022_4117123'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162022_4117123/input0_results_240321_162022'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162022_4117123/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_WRREQ_DRAM_CREDIT_STALL[0], TCC_EA_WRREQ_GMI_CREDIT_STALL[0], TCC_EA_WRREQ_IO_CREDIT_STALL[0], TCC_EA_WRREQ_LEVEL[0], TCC_EA_WRREQ_DRAM_CREDIT_STALL[1], TCC_EA_WRREQ_GMI_CREDIT_STALL[1], TCC_EA_WRREQ_IO_CREDIT_STALL[1], TCC_EA_WRREQ_LEVEL[1], TCC_EA_WRREQ_DRAM_CREDIT_STALL[2], TCC_EA_WRREQ_GMI_CREDIT_STALL[2], TCC_EA_WRREQ_IO_CREDIT_STALL[2], TCC_EA_WRREQ_LEVEL[2], TCC_EA_WRREQ_DRAM_CREDIT_STALL[3], TCC_EA_WRREQ_GMI_CREDIT_STALL[3], TCC_EA_WRREQ_IO_CREDIT_STALL[3], TCC_EA_WRREQ_LEVEL[3], TCC_EA_WRREQ_DRAM_CREDIT_STALL[4], TCC_EA_WRREQ_GMI_CREDIT_STALL[4], TCC_EA_WRREQ_IO_CREDIT_STALL[4], TCC_EA_WRREQ_LEVEL[4], TCC_EA_WRREQ_DRAM_CREDIT_STALL[5], TCC_EA_WRREQ_GMI_CREDIT_STALL[5], TCC_EA_WRREQ_IO_CREDIT_STALL[5], TCC_EA_WRREQ_LEVEL[5], TCC_EA_WRREQ_DRAM_CREDIT_STALL[6], TCC_EA_WRREQ_GMI_CREDIT_STALL[6], TCC_EA_WRREQ_IO_CREDIT_STALL[6], TCC_EA_WRREQ_LEVEL[6], TCC_EA_WRREQ_DRAM_CREDIT_STALL[7], TCC_EA_WRREQ_GMI_CREDIT_STALL[7], TCC_EA_WRREQ_IO_CREDIT_STALL[7], TCC_EA_WRREQ_LEVEL[7], TCC_EA_WRREQ_DRAM_CREDIT_STALL[8], TCC_EA_WRREQ_GMI_CREDIT_STALL[8], TCC_EA_WRREQ_IO_CREDIT_STALL[8], TCC_EA_WRREQ_LEVEL[8], TCC_EA_WRREQ_DRAM_CREDIT_STALL[9], TCC_EA_WRREQ_GMI_CREDIT_STALL[9], TCC_EA_WRREQ_IO_CREDIT_STALL[9], TCC_EA_WRREQ_LEVEL[9], TCC_EA_WRREQ_DRAM_CREDIT_STALL[10], TCC_EA_WRREQ_GMI_CREDIT_STALL[10], TCC_EA_WRREQ_IO_CREDIT_STALL[10], TCC_EA_WRREQ_LEVEL[10], TCC_EA_WRREQ_DRAM_CREDIT_STALL[11], TCC_EA_WRREQ_GMI_CREDIT_STALL[11], TCC_EA_WRREQ_IO_CREDIT_STALL[11], TCC_EA_WRREQ_LEVEL[11], TCC_EA_WRREQ_DRAM_CREDIT_STALL[12], TCC_EA_WRREQ_GMI_CREDIT_STALL[12], TCC_EA_WRREQ_IO_CREDIT_STALL[12], TCC_EA_WRREQ_LEVEL[12], TCC_EA_WRREQ_DRAM_CREDIT_STALL[13], TCC_EA_WRREQ_GMI_CREDIT_STALL[13], TCC_EA_WRREQ_IO_CREDIT_STALL[13], TCC_EA_WRREQ_LEVEL[13], TCC_EA_WRREQ_DRAM_CREDIT_STALL[14], TCC_EA_WRREQ_GMI_CREDIT_STALL[14], TCC_EA_WRREQ_IO_CREDIT_STALL[14], TCC_EA_WRREQ_LEVEL[14], TCC_EA_WRREQ_DRAM_CREDIT_STALL[15], TCC_EA_WRREQ_GMI_CREDIT_STALL[15], TCC_EA_WRREQ_IO_CREDIT_STALL[15], TCC_EA_WRREQ_LEVEL[15], TCC_EA_WRREQ_DRAM_CREDIT_STALL[16], TCC_EA_WRREQ_GMI_CREDIT_STALL[16], TCC_EA_WRREQ_IO_CREDIT_STALL[16], TCC_EA_WRREQ_LEVEL[16], TCC_EA_WRREQ_DRAM_CREDIT_STALL[17], TCC_EA_WRREQ_GMI_CREDIT_STALL[17], TCC_EA_WRREQ_IO_CREDIT_STALL[17], TCC_EA_WRREQ_LEVEL[17], TCC_EA_WRREQ_DRAM_CREDIT_STALL[18], TCC_EA_WRREQ_GMI_CREDIT_STALL[18], TCC_EA_WRREQ_IO_CREDIT_STALL[18], TCC_EA_WRREQ_LEVEL[18], TCC_EA_WRREQ_DRAM_CREDIT_STALL[19], TCC_EA_WRREQ_GMI_CREDIT_STALL[19], TCC_EA_WRREQ_IO_CREDIT_STALL[19], TCC_EA_WRREQ_LEVEL[19], TCC_EA_WRREQ_DRAM_CREDIT_STALL[20], TCC_EA_WRREQ_GMI_CREDIT_STALL[20], TCC_EA_WRREQ_IO_CREDIT_STALL[20], TCC_EA_WRREQ_LEVEL[20], TCC_EA_WRREQ_DRAM_CREDIT_STALL[21], TCC_EA_WRREQ_GMI_CREDIT_STALL[21], TCC_EA_WRREQ_IO_CREDIT_STALL[21], TCC_EA_WRREQ_LEVEL[21], TCC_EA_WRREQ_DRAM_CREDIT_STALL[22], TCC_EA_WRREQ_GMI_CREDIT_STALL[22], TCC_EA_WRREQ_IO_CREDIT_STALL[22], TCC_EA_WRREQ_LEVEL[22], TCC_EA_WRREQ_DRAM_CREDIT_STALL[23], TCC_EA_WRREQ_GMI_CREDIT_STALL[23], TCC_EA_WRREQ_IO_CREDIT_STALL[23], TCC_EA_WRREQ_LEVEL[23], TCC_EA_WRREQ_DRAM_CREDIT_STALL[24], TCC_EA_WRREQ_GMI_CREDIT_STALL[24], TCC_EA_WRREQ_IO_CREDIT_STALL[24], TCC_EA_WRREQ_LEVEL[24], TCC_EA_WRREQ_DRAM_CREDIT_STALL[25], TCC_EA_WRREQ_GMI_CREDIT_STALL[25], TCC_EA_WRREQ_IO_CREDIT_STALL[25], TCC_EA_WRREQ_LEVEL[25], TCC_EA_WRREQ_DRAM_CREDIT_STALL[26], TCC_EA_WRREQ_GMI_CREDIT_STALL[26], TCC_EA_WRREQ_IO_CREDIT_STALL[26], TCC_EA_WRREQ_LEVEL[26], TCC_EA_WRREQ_DRAM_CREDIT_STALL[27], TCC_EA_WRREQ_GMI_CREDIT_STALL[27], TCC_EA_WRREQ_IO_CREDIT_STALL[27], TCC_EA_WRREQ_LEVEL[27], TCC_EA_WRREQ_DRAM_CREDIT_STALL[28], TCC_EA_WRREQ_GMI_CREDIT_STALL[28], TCC_EA_WRREQ_IO_CREDIT_STALL[28], TCC_EA_WRREQ_LEVEL[28], TCC_EA_WRREQ_DRAM_CREDIT_STALL[29], TCC_EA_WRREQ_GMI_CREDIT_STALL[29], TCC_EA_WRREQ_IO_CREDIT_STALL[29], TCC_EA_WRREQ_LEVEL[29], TCC_EA_WRREQ_DRAM_CREDIT_STALL[30], TCC_EA_WRREQ_GMI_CREDIT_STALL[30], TCC_EA_WRREQ_IO_CREDIT_STALL[30], TCC_EA_WRREQ_LEVEL[30], TCC_EA_WRREQ_DRAM_CREDIT_STALL[31], TCC_EA_WRREQ_GMI_CREDIT_STALL[31], TCC_EA_WRREQ_IO_CREDIT_STALL[31], TCC_EA_WRREQ_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162022_4117123/input0_results_240321_162022
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_16.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_17.txt
|-> [rocprof] RPL: on '240321_162023' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_17.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162023_4117308'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162023_4117308/input0_results_240321_162023'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162023_4117308/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_HIT[0], TCC_MISS[0], TCC_READ[0], TCC_REQ[0], TCC_HIT[1], TCC_MISS[1], TCC_READ[1], TCC_REQ[1], TCC_HIT[2], TCC_MISS[2], TCC_READ[2], TCC_REQ[2], TCC_HIT[3], TCC_MISS[3], TCC_READ[3], TCC_REQ[3], TCC_HIT[4], TCC_MISS[4], TCC_READ[4], TCC_REQ[4], TCC_HIT[5], TCC_MISS[5], TCC_READ[5], TCC_REQ[5], TCC_HIT[6], TCC_MISS[6], TCC_READ[6], TCC_REQ[6], TCC_HIT[7], TCC_MISS[7], TCC_READ[7], TCC_REQ[7], TCC_HIT[8], TCC_MISS[8], TCC_READ[8], TCC_REQ[8], TCC_HIT[9], TCC_MISS[9], TCC_READ[9], TCC_REQ[9], TCC_HIT[10], TCC_MISS[10], TCC_READ[10], TCC_REQ[10], TCC_HIT[11], TCC_MISS[11], TCC_READ[11], TCC_REQ[11], TCC_HIT[12], TCC_MISS[12], TCC_READ[12], TCC_REQ[12], TCC_HIT[13], TCC_MISS[13], TCC_READ[13], TCC_REQ[13], TCC_HIT[14], TCC_MISS[14], TCC_READ[14], TCC_REQ[14], TCC_HIT[15], TCC_MISS[15], TCC_READ[15], TCC_REQ[15], TCC_HIT[16], TCC_MISS[16], TCC_READ[16], TCC_REQ[16], TCC_HIT[17], TCC_MISS[17], TCC_READ[17], TCC_REQ[17], TCC_HIT[18], TCC_MISS[18], TCC_READ[18], TCC_REQ[18], TCC_HIT[19], TCC_MISS[19], TCC_READ[19], TCC_REQ[19], TCC_HIT[20], TCC_MISS[20], TCC_READ[20], TCC_REQ[20], TCC_HIT[21], TCC_MISS[21], TCC_READ[21], TCC_REQ[21], TCC_HIT[22], TCC_MISS[22], TCC_READ[22], TCC_REQ[22], TCC_HIT[23], TCC_MISS[23], TCC_READ[23], TCC_REQ[23], TCC_HIT[24], TCC_MISS[24], TCC_READ[24], TCC_REQ[24], TCC_HIT[25], TCC_MISS[25], TCC_READ[25], TCC_REQ[25], TCC_HIT[26], TCC_MISS[26], TCC_READ[26], TCC_REQ[26], TCC_HIT[27], TCC_MISS[27], TCC_READ[27], TCC_REQ[27], TCC_HIT[28], TCC_MISS[28], TCC_READ[28], TCC_REQ[28], TCC_HIT[29], TCC_MISS[29], TCC_READ[29], TCC_REQ[29], TCC_HIT[30], TCC_MISS[30], TCC_READ[30], TCC_REQ[30], TCC_HIT[31], TCC_MISS[31], TCC_READ[31], TCC_REQ[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162023_4117308/input0_results_240321_162023
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_17.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_18.txt
|-> [rocprof] RPL: on '240321_162024' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_18.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162024_4117498'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162024_4117498/input0_results_240321_162024'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162024_4117498/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 96 metrics
|-> [rocprof] TCC_RW_REQ[0], TCC_TOO_MANY_EA_WRREQS_STALL[0], TCC_WRITE[0], TCC_RW_REQ[1], TCC_TOO_MANY_EA_WRREQS_STALL[1], TCC_WRITE[1], TCC_RW_REQ[2], TCC_TOO_MANY_EA_WRREQS_STALL[2], TCC_WRITE[2], TCC_RW_REQ[3], TCC_TOO_MANY_EA_WRREQS_STALL[3], TCC_WRITE[3], TCC_RW_REQ[4], TCC_TOO_MANY_EA_WRREQS_STALL[4], TCC_WRITE[4], TCC_RW_REQ[5], TCC_TOO_MANY_EA_WRREQS_STALL[5], TCC_WRITE[5], TCC_RW_REQ[6], TCC_TOO_MANY_EA_WRREQS_STALL[6], TCC_WRITE[6], TCC_RW_REQ[7], TCC_TOO_MANY_EA_WRREQS_STALL[7], TCC_WRITE[7], TCC_RW_REQ[8], TCC_TOO_MANY_EA_WRREQS_STALL[8], TCC_WRITE[8], TCC_RW_REQ[9], TCC_TOO_MANY_EA_WRREQS_STALL[9], TCC_WRITE[9], TCC_RW_REQ[10], TCC_TOO_MANY_EA_WRREQS_STALL[10], TCC_WRITE[10], TCC_RW_REQ[11], TCC_TOO_MANY_EA_WRREQS_STALL[11], TCC_WRITE[11], TCC_RW_REQ[12], TCC_TOO_MANY_EA_WRREQS_STALL[12], TCC_WRITE[12], TCC_RW_REQ[13], TCC_TOO_MANY_EA_WRREQS_STALL[13], TCC_WRITE[13], TCC_RW_REQ[14], TCC_TOO_MANY_EA_WRREQS_STALL[14], TCC_WRITE[14], TCC_RW_REQ[15], TCC_TOO_MANY_EA_WRREQS_STALL[15], TCC_WRITE[15], TCC_RW_REQ[16], TCC_TOO_MANY_EA_WRREQS_STALL[16], TCC_WRITE[16], TCC_RW_REQ[17], TCC_TOO_MANY_EA_WRREQS_STALL[17], TCC_WRITE[17], TCC_RW_REQ[18], TCC_TOO_MANY_EA_WRREQS_STALL[18], TCC_WRITE[18], TCC_RW_REQ[19], TCC_TOO_MANY_EA_WRREQS_STALL[19], TCC_WRITE[19], TCC_RW_REQ[20], TCC_TOO_MANY_EA_WRREQS_STALL[20], TCC_WRITE[20], TCC_RW_REQ[21], TCC_TOO_MANY_EA_WRREQS_STALL[21], TCC_WRITE[21], TCC_RW_REQ[22], TCC_TOO_MANY_EA_WRREQS_STALL[22], TCC_WRITE[22], TCC_RW_REQ[23], TCC_TOO_MANY_EA_WRREQS_STALL[23], TCC_WRITE[23], TCC_RW_REQ[24], TCC_TOO_MANY_EA_WRREQS_STALL[24], TCC_WRITE[24], TCC_RW_REQ[25], TCC_TOO_MANY_EA_WRREQS_STALL[25], TCC_WRITE[25], TCC_RW_REQ[26], TCC_TOO_MANY_EA_WRREQS_STALL[26], TCC_WRITE[26], TCC_RW_REQ[27], TCC_TOO_MANY_EA_WRREQS_STALL[27], TCC_WRITE[27], TCC_RW_REQ[28], TCC_TOO_MANY_EA_WRREQS_STALL[28], TCC_WRITE[28], TCC_RW_REQ[29], TCC_TOO_MANY_EA_WRREQS_STALL[29], TCC_WRITE[29], TCC_RW_REQ[30], TCC_TOO_MANY_EA_WRREQS_STALL[30], TCC_WRITE[30], TCC_RW_REQ[31], TCC_TOO_MANY_EA_WRREQS_STALL[31], TCC_WRITE[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162024_4117498/input0_results_240321_162024
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_18.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_2.txt
|-> [rocprof] RPL: on '240321_162024' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_2.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162024_4117686'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162024_4117686/input0_results_240321_162024'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162024_4117686/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 26 metrics
|-> [rocprof] SQ_INSTS_VALU_MUL_F32, SQ_INSTS_VALU_FMA_F32, SQ_INSTS_VALU_TRANS_F32, SQ_INSTS_VALU_ADD_F64, SQ_INSTS_VALU_MUL_F64, SQ_INSTS_VALU_FMA_F64, SQ_INSTS_VALU_TRANS_F64, SQ_INSTS_VALU_INT32, TCP_VOLATILE_sum, TCP_TOTAL_ACCESSES_sum, TCP_TOTAL_READ_sum, TCP_TOTAL_WRITE_sum, TA_BUFFER_ATOMIC_WAVEFRONTS_sum, TA_BUFFER_TOTAL_CYCLES_sum, TD_ATOMIC_WAVEFRONT_sum, TD_STORE_WAVEFRONT_sum, SPI_RA_REQ_NO_ALLOC, SPI_RA_REQ_NO_ALLOC_CSN, CPC_CPC_STAT_STALL, CPC_UTCL1_STALL_ON_TRANSLATION, CPF_CPF_STAT_IDLE, CPF_CPF_TCIU_IDLE, TCC_REQ_sum, TCC_STREAMING_REQ_sum, TCC_HIT_sum, TCC_MISS_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162024_4117686/input0_results_240321_162024
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_2.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_3.txt
|-> [rocprof] RPL: on '240321_162025' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_3.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162025_4117888'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162025_4117888/input0_results_240321_162025'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162025_4117888/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 24 metrics
|-> [rocprof] SQ_INSTS_VALU_INT64, SQ_INSTS_SMEM, SQ_INSTS_FLAT, SQ_INSTS_LDS, SQ_INSTS_GDS, SQ_INSTS_EXP_GDS, SQ_INSTS_BRANCH, SQ_INSTS_SENDMSG, TCP_TOTAL_ATOMIC_WITH_RET_sum, TCP_TOTAL_ATOMIC_WITHOUT_RET_sum, TCP_TOTAL_WRITEBACK_INVALIDATES_sum, TCP_TOTAL_CACHE_ACCESSES_sum, TA_BUFFER_COALESCED_READ_CYCLES_sum, TA_BUFFER_COALESCED_WRITE_CYCLES_sum, TD_COALESCABLE_WAVEFRONT_sum, SPI_RA_RES_STALL_CSN, SPI_RA_TMP_STALL_CSN, CPC_CPC_UTCL2IU_BUSY, CPC_CPC_UTCL2IU_IDLE, CPF_CMP_UTCL1_STALL_ON_TRANSLATION, TCC_READ_sum, TCC_WRITE_sum, TCC_ATOMIC_sum, TCC_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162025_4117888/input0_results_240321_162025
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_3.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_4.txt
|-> [rocprof] RPL: on '240321_162025' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_4.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162025_4118072'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162025_4118072/input0_results_240321_162025'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162025_4118072/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 22 metrics
|-> [rocprof] SQ_WAVE_CYCLES, SQ_WAIT_ANY, SQ_WAIT_INST_ANY, SQ_ACTIVE_INST_ANY, SQ_BUSY_CU_CYCLES, SQ_ACTIVE_INST_VMEM, SQ_ACTIVE_INST_LDS, SQ_ACTIVE_INST_VALU, TCP_UTCL1_TRANSLATION_MISS_sum, TCP_UTCL1_TRANSLATION_HIT_sum, TCP_UTCL1_PERMISSION_MISS_sum, TCP_UTCL1_REQUEST_sum, TA_ADDR_STALLED_BY_TC_CYCLES_sum, TA_TOTAL_WAVEFRONTS_sum, SPI_RA_WAVE_SIMD_FULL_CSN, SPI_RA_VGPR_SIMD_FULL_CSN, CPC_CPC_UTCL2IU_STALL, CPC_ME1_BUSY_FOR_PACKET_DECODE, TCC_EA_WRREQ_sum, TCC_EA_WRREQ_64B_sum, TCC_EA_WR_UNCACHED_32B_sum, TCC_EA_WRREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162025_4118072/input0_results_240321_162025
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_4.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_5.txt
|-> [rocprof] RPL: on '240321_162026' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_5.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162026_4118275'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162026_4118275/input0_results_240321_162026'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162026_4118275/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 21 metrics
|-> [rocprof] SQ_ACTIVE_INST_SCA, SQ_ACTIVE_INST_EXP_GDS, SQ_ACTIVE_INST_MISC, SQ_ACTIVE_INST_FLAT, SQ_INST_CYCLES_VMEM_WR, SQ_INST_CYCLES_VMEM_RD, SQ_INST_CYCLES_SMEM, SQ_INST_CYCLES_SALU, TCP_TCP_LATENCY_sum, TCP_TCC_READ_REQ_LATENCY_sum, TCP_TCC_WRITE_REQ_LATENCY_sum, TCP_TCC_READ_REQ_sum, TA_ADDR_STALLED_BY_TD_CYCLES_sum, TA_DATA_STALLED_BY_TC_CYCLES_sum, SPI_RA_SGPR_SIMD_FULL_CSN, SPI_RA_LDS_CU_FULL_CSN, CPC_ME1_DC0_SPI_BUSY, TCC_EA_WRREQ_STALL_sum, TCC_EA_RDREQ_sum, TCC_EA_RDREQ_32B_sum, TCC_EA_RD_UNCACHED_32B_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162026_4118275/input0_results_240321_162026
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_5.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_6.txt
|-> [rocprof] RPL: on '240321_162026' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_6.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162026_4118477'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162026_4118477/input0_results_240321_162026'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162026_4118477/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_THREAD_CYCLES_VALU, SQ_IFETCH, SQ_LDS_BANK_CONFLICT, SQ_LDS_ADDR_CONFLICT, SQ_LDS_UNALIGNED_STALL, SQ_WAVES_EQ_64, SQ_WAVES_LT_64, SQ_WAVES_LT_48, TCP_TCC_WRITE_REQ_sum, TCP_TCC_ATOMIC_WITH_RET_REQ_sum, TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum, TCP_TCC_NC_READ_REQ_sum, TA_FLAT_WAVEFRONTS_sum, TA_FLAT_READ_WAVEFRONTS_sum, SPI_RA_BAR_CU_FULL_CSN, SPI_RA_TGLIM_CU_FULL_CSN, TCC_EA_RDREQ_DRAM_sum, TCC_TAG_STALL_sum, TCC_NORMAL_WRITEBACK_sum, TCC_ALL_TC_OP_WB_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162026_4118477/input0_results_240321_162026
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_6.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_7.txt
|-> [rocprof] RPL: on '240321_162027' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_7.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162027_4118664'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162027_4118664/input0_results_240321_162027'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162027_4118664/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_WAVES_LT_32, SQ_WAVES_LT_16, SQ_ITEMS, SQ_LDS_MEM_VIOLATIONS, SQ_LDS_ATOMIC_RETURN, SQ_LDS_IDX_ACTIVE, SQ_WAVES_RESTORED, SQ_WAVES_SAVED, TCP_TCC_NC_WRITE_REQ_sum, TCP_TCC_NC_ATOMIC_REQ_sum, TCP_TCC_UC_READ_REQ_sum, TCP_TCC_UC_WRITE_REQ_sum, TA_FLAT_WRITE_WAVEFRONTS_sum, TA_FLAT_ATOMIC_WAVEFRONTS_sum, SPI_RA_WVLIM_STALL_CSN, SPI_SWC_CSC_WR, TCC_NORMAL_EVICT_sum, TCC_ALL_TC_OP_INV_EVICT_sum, TCC_TOO_MANY_EA_WRREQS_STALL_sum, TCC_EA_ATOMIC_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162027_4118664/input0_results_240321_162027
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_7.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_8.txt
|-> [rocprof] RPL: on '240321_162027' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_8.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162027_4118848'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162027_4118848/input0_results_240321_162027'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162027_4118848/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 17 metrics
|-> [rocprof] SQ_INSTS_SMEM_NORM, SQ_INSTS_MFMA, SQ_INSTS_VALU_MFMA_I8, SQ_INSTS_VALU_MFMA_F16, SQ_INSTS_VALU_MFMA_BF16, SQ_INSTS_VALU_MFMA_F32, SQ_INSTS_VALU_MFMA_F64, SQ_VALU_MFMA_BUSY_CYCLES, TCP_TCC_UC_ATOMIC_REQ_sum, TCP_TCC_CC_READ_REQ_sum, TCP_TCC_CC_WRITE_REQ_sum, TCP_TCC_CC_ATOMIC_REQ_sum, SPI_VWC_CSC_WR, SPI_RA_BULKY_CU_FULL_CSN, TCC_EA_RDREQ_LEVEL_sum, TCC_EA_WRREQ_LEVEL_sum, TCC_EA_ATOMIC_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162027_4118848/input0_results_240321_162027
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_8.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_9.txt
|-> [rocprof] RPL: on '240321_162028' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/pmc_perf_9.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162028_4119050'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162028_4119050/input0_results_240321_162028'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162028_4119050/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 12 metrics
|-> [rocprof] SQ_INSTS_FLAT_LDS_ONLY, SQ_INSTS_VALU_MFMA_MOPS_I8, SQ_INSTS_VALU_MFMA_MOPS_F16, SQ_INSTS_VALU_MFMA_MOPS_BF16, SQ_INSTS_VALU_MFMA_MOPS_F32, SQ_INSTS_VALU_MFMA_MOPS_F64, SQC_TC_INST_REQ, SQC_TC_DATA_READ_REQ, TCP_TCC_RW_READ_REQ_sum, TCP_TCC_RW_WRITE_REQ_sum, TCP_TCC_RW_ATOMIC_REQ_sum, TCP_PENDING_STALL_CYCLES_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162028_4119050/input0_results_240321_162028
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/pmc_perf_9.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel_substr/MI200/perfmon/timestamps.txt
|-> [rocprof] RPL: on '240321_162028' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel_substr/MI200/perfmon/timestamps.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_162028_4119234'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_162028_4119234/input0_results_240321_162028'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_162028_4119234/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 0 metrics
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_162028_4119234/input0_results_240321_162028
|-> [rocprof] File 'tests/workloads/kernel_substr/MI200/timestamps.csv' is generating
|-> [rocprof]
[roofline] Checking for roofline.csv in tests/workloads/kernel_substr/MI200
[roofline] No roofline data found. Generating...
+4
Voir le fichier
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1 Dispatch_ID Kernel_Name GPU_ID
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
Diff de fichier supprimé car une ou plusieurs lignes sont trop longues
+4 -4
Voir le fichier
@@ -1,5 +1,5 @@
device,HBMBw,HBMBwLow,hbmBwHigh,L2Bw,L2BwLow,L2BwHigh,L1Bw,L1BwLow,L1BwHigh,LDSBw,LDSBwLow,LDSBwHigh,FP32Flops,FP32FlopsLow,FP32FlopsHigh,FP64Flops,FP64FlopsLow,FP64FlopsHigh,MFMABF16Flops,MFMABF16FlopsLow,MFMABF16FlopsHigh,MFMAF16Flops,MFMAF16FlopsLow,MFMAF16FlopsHigh,MFMAF32Flops,MFMAF32FlopsLow,MFMAF32FlopsHigh,MFMAF64Flops,MFMAF64FlopsLow,MFMAF64FlopsHigh,MFMAI8Ops,MFMAFI8OpsLow,MFMAI8OpsHigh
0,1388.4747,1387.9553,1388.9941,5014.9604,5012.063,5017.8579,9222.5117,9221.9268,9223.0967,18050.73,18049.891,18051.57,20930.57,20869.91,20991.23,20265.133,20264.418,20265.848,170420.36,170416.08,170424.64,164859.31,164854.41,164864.22,41417.266,41416.672,41417.859,41477.48,41476.703,41478.258,166330.06,165840.5,166819.62
1,1389.7251,1389.1949,1390.2552,5028.3931,5026.2188,5030.5674,9234.2783,9233.5322,9235.0244,18087.188,18083.588,18090.787,20971.094,20904.305,21037.883,20291.975,20291.318,20292.631,170824.14,170818.92,170829.36,165082.02,165078.61,165085.42,41468.77,41467.949,41469.59,41525.664,41524.758,41526.57,166822.78,166819.69,166825.88
2,1388.7198,1388.2028,1389.2369,5036.1206,5033.917,5038.3242,9261.9297,9261.1611,9262.6982,18106.574,18105.59,18107.559,21049.045,21048.621,21049.469,20356.066,20355.422,20356.711,171138.38,171134.81,171141.94,165494.95,165491.14,165498.77,41581.402,41580.301,41582.504,41641.324,41640.227,41642.422,166971.16,166374.72,167567.59
3,1388.6997,1388.0704,1389.329,5030.8838,5029.3442,5032.4233,9239.749,9239.2314,9240.2666,18025.273,18022.016,18028.531,21030.592,21030.15,21031.033,20311.781,20311.162,20312.4,171031.81,171027.84,171035.78,165235.41,165231.41,165239.41,41440.09,41286.125,41594.055,41580.582,41580.031,41581.133,167008.56,167006.89,167010.23
0,1388.9658,1388.3842,1389.5475,5015.9883,5012.938,5019.0386,9225.1172,9224.5098,9225.7246,17715.531,17712.326,17718.736,20938.605,20878.021,20999.189,20270.34,20269.768,20270.912,170471.98,170468.02,170475.95,164910.77,164906.02,164915.52,41434.18,41433.711,41434.648,41492.617,41491.949,41493.285,166394.14,165902.47,166885.81
1,1389.0469,1388.5142,1389.5796,5028.8301,5026.8472,5030.813,9236.9521,9236.207,9237.6973,18239.07,18236.047,18242.094,20985.062,20918.174,21051.951,20304.107,20303.527,20304.688,170946.45,170942.11,170950.8,165186.3,165182.33,165190.27,41504.238,41503.312,41505.164,41562.047,41561.035,41563.059,166949,166945.52,166952.48
2,1388.8693,1388.3575,1389.381,5035.5464,5033.2891,5037.8037,9260.1914,9259.3496,9261.0332,18218.021,18217.1,18218.943,21053.375,21052.932,21053.818,20353.885,20353.205,20354.564,171154.2,171149.33,171159.08,165526.98,165523.47,165530.5,41583.637,41582.98,41584.293,41644.199,41643.281,41645.117,166959.98,166367,167552.97
3,1389.0635,1388.4847,1389.6422,5033.917,5032.3442,5035.4897,9239.7314,9239.0586,9240.4043,18333.445,18332.5,18334.391,21036.664,21036.422,21036.906,20312.801,20312.271,20313.33,171053,171047.16,171058.84,165173.69,165169.31,165178.06,41435.715,41282.48,41588.949,41581.074,41579.879,41582.27,166969.88,166965.3,166974.45
1 device HBMBw HBMBwLow hbmBwHigh L2Bw L2BwLow L2BwHigh L1Bw L1BwLow L1BwHigh LDSBw LDSBwLow LDSBwHigh FP32Flops FP32FlopsLow FP32FlopsHigh FP64Flops FP64FlopsLow FP64FlopsHigh MFMABF16Flops MFMABF16FlopsLow MFMABF16FlopsHigh MFMAF16Flops MFMAF16FlopsLow MFMAF16FlopsHigh MFMAF32Flops MFMAF32FlopsLow MFMAF32FlopsHigh MFMAF64Flops MFMAF64FlopsLow MFMAF64FlopsHigh MFMAI8Ops MFMAFI8OpsLow MFMAI8OpsHigh
2 0 1388.4747 1388.9658 1387.9553 1388.3842 1388.9941 1389.5475 5014.9604 5015.9883 5012.063 5012.938 5017.8579 5019.0386 9222.5117 9225.1172 9221.9268 9224.5098 9223.0967 9225.7246 18050.73 17715.531 18049.891 17712.326 18051.57 17718.736 20930.57 20938.605 20869.91 20878.021 20991.23 20999.189 20265.133 20270.34 20264.418 20269.768 20265.848 20270.912 170420.36 170471.98 170416.08 170468.02 170424.64 170475.95 164859.31 164910.77 164854.41 164906.02 164864.22 164915.52 41417.266 41434.18 41416.672 41433.711 41417.859 41434.648 41477.48 41492.617 41476.703 41491.949 41478.258 41493.285 166330.06 166394.14 165840.5 165902.47 166819.62 166885.81
3 1 1389.7251 1389.0469 1389.1949 1388.5142 1390.2552 1389.5796 5028.3931 5028.8301 5026.2188 5026.8472 5030.5674 5030.813 9234.2783 9236.9521 9233.5322 9236.207 9235.0244 9237.6973 18087.188 18239.07 18083.588 18236.047 18090.787 18242.094 20971.094 20985.062 20904.305 20918.174 21037.883 21051.951 20291.975 20304.107 20291.318 20303.527 20292.631 20304.688 170824.14 170946.45 170818.92 170942.11 170829.36 170950.8 165082.02 165186.3 165078.61 165182.33 165085.42 165190.27 41468.77 41504.238 41467.949 41503.312 41469.59 41505.164 41525.664 41562.047 41524.758 41561.035 41526.57 41563.059 166822.78 166949 166819.69 166945.52 166825.88 166952.48
4 2 1388.7198 1388.8693 1388.2028 1388.3575 1389.2369 1389.381 5036.1206 5035.5464 5033.917 5033.2891 5038.3242 5037.8037 9261.9297 9260.1914 9261.1611 9259.3496 9262.6982 9261.0332 18106.574 18218.021 18105.59 18217.1 18107.559 18218.943 21049.045 21053.375 21048.621 21052.932 21049.469 21053.818 20356.066 20353.885 20355.422 20353.205 20356.711 20354.564 171138.38 171154.2 171134.81 171149.33 171141.94 171159.08 165494.95 165526.98 165491.14 165523.47 165498.77 165530.5 41581.402 41583.637 41580.301 41582.98 41582.504 41584.293 41641.324 41644.199 41640.227 41643.281 41642.422 41645.117 166971.16 166959.98 166374.72 166367 167567.59 167552.97
5 3 1388.6997 1389.0635 1388.0704 1388.4847 1389.329 1389.6422 5030.8838 5033.917 5029.3442 5032.3442 5032.4233 5035.4897 9239.749 9239.7314 9239.2314 9239.0586 9240.2666 9240.4043 18025.273 18333.445 18022.016 18332.5 18028.531 18334.391 21030.592 21036.664 21030.15 21036.422 21031.033 21036.906 20311.781 20312.801 20311.162 20312.271 20312.4 20313.33 171031.81 171053 171027.84 171047.16 171035.78 171058.84 165235.41 165173.69 165231.41 165169.31 165239.41 165178.06 41440.09 41435.715 41286.125 41282.48 41594.055 41588.949 41580.582 41581.074 41580.031 41579.879 41581.133 41582.27 167008.56 166969.88 167006.89 166965.3 167010.23 166974.45
+1 -1
Voir le fichier
@@ -1,2 +1,2 @@
workload_name,command,ip_blocks,timestamp,version,hostname,cpu_model,sbios,linux_distro,linux_kernel_version,amd_gpu_kernel_version,cpu_memory,gpu_memory,rocm_version,vbios,compute_partition,memory_partition,gpu_model,gpu_arch,gpu_l1,gpu_l2,cu_per_gpu,simd_per_cu,se_per_gpu,wave_size,workgroup_max_size,max_waves_per_cu,max_sclk,max_mclk,cur_sclk,cur_mclk,total_l2_chan,lds_banks_per_cu,sqc_per_gpu,pipes_per_gpu,hbm_bw,num_xcd
kernel_substr,./sample/vcopy -n 1048576 -b 256 -i 3,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF|roofline,Thu 07 Mar 2024 02:15:28 PM (CST),2,t007-002.hpcfund,AMD EPYC 7V13 64-Core Processor,American Megatrends Inc.0602,Rocky Linux 9.1 (Blue Onyx),5.14.0-162.18.1.el9_1.x86_64,,527650760,,6.0.2-115,113-D67301-059,NA,NA,MI200,gfx90a,16,8192,104,4,8,64,1024,32,1700,1600,1700,1600,32,32,56,4,1638.4,1
kernel_substr,./tests/vcopy -n 1048576 -b 256 -i 3,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF|roofline,Thu 21 Mar 2024 04:20:15 PM (CDT),2,t007-002.hpcfund,AMD EPYC 7V13 64-Core Processor,American Megatrends Inc.0602,Rocky Linux 9.1 (Blue Onyx),5.14.0-162.18.1.el9_1.x86_64,,527650760,,6.0.2-115,113-D67301-059,NA,NA,MI200,gfx90a,16,8192,104,4,8,64,1024,32,1700,1600,1700,1600,32,32,56,4,1638.4,1
1 workload_name command ip_blocks timestamp version hostname cpu_model sbios linux_distro linux_kernel_version amd_gpu_kernel_version cpu_memory gpu_memory rocm_version vbios compute_partition memory_partition gpu_model gpu_arch gpu_l1 gpu_l2 cu_per_gpu simd_per_cu se_per_gpu wave_size workgroup_max_size max_waves_per_cu max_sclk max_mclk cur_sclk cur_mclk total_l2_chan lds_banks_per_cu sqc_per_gpu pipes_per_gpu hbm_bw num_xcd
2 kernel_substr ./sample/vcopy -n 1048576 -b 256 -i 3 ./tests/vcopy -n 1048576 -b 256 -i 3 SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF|roofline Thu 07 Mar 2024 02:15:28 PM (CST) Thu 21 Mar 2024 04:20:15 PM (CDT) 2 t007-002.hpcfund AMD EPYC 7V13 64-Core Processor American Megatrends Inc.0602 Rocky Linux 9.1 (Blue Onyx) 5.14.0-162.18.1.el9_1.x86_64 527650760 6.0.2-115 113-D67301-059 NA NA MI200 gfx90a 16 8192 104 4 8 64 1024 32 1700 1600 1700 1600 32 32 56 4 1638.4 1
+3 -3
Voir le fichier
@@ -1,4 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,295872,295872,1048576,256,0,0,8,0,16,64,0x0,0x7f38a4160ec0,198274458301241,198274458325591,198274458345911,198274458362076
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,295872,295872,1048576,256,0,0,8,0,16,64,0x0,0x7f38a4160ec0,198274458356796,198274458365591,198274458381111,198274458440294
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,295872,295872,1048576,256,0,0,8,0,16,64,0x0,0x7f38a4160ec0,198274458391411,198274458443671,198274458459991,198274458461033
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4119394,4119394,1048576,256,0,0,8,0,16,64,0x0,0x7fac132a4ec0,1411760919510468,1411760919537410,1411760919557090,1411760919572024
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4119394,4119394,1048576,256,0,0,8,0,16,64,0x0,0x7fac132a4ec0,1411760919567997,1411760919576130,1411760919591970,1411760919645724
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4119394,4119394,1048576,256,0,0,8,0,16,64,0x0,0x7fac132a4ec0,1411760919602192,1411760919648450,1411760919664930,1411760919666122
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 295872 4119394 295872 4119394 1048576 256 0 0 8 0 16 64 0x0 0x7f38a4160ec0 0x7fac132a4ec0 198274458301241 1411760919510468 198274458325591 1411760919537410 198274458345911 1411760919557090 198274458362076 1411760919572024
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 295872 4119394 295872 4119394 1048576 256 0 0 8 0 16 64 0x0 0x7f38a4160ec0 0x7fac132a4ec0 198274458356796 1411760919567997 198274458365591 1411760919576130 198274458381111 1411760919591970 198274458440294 1411760919645724
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 295872 4119394 295872 4119394 1048576 256 0 0 8 0 16 64 0x0 0x7f38a4160ec0 0x7fac132a4ec0 198274458391411 1411760919602192 198274458443671 1411760919648450 198274458459991 1411760919664930 198274458461033 1411760919666122