Pytest add mi200 to analyze workloads (#334)

* Updated links in documentation. (#328)

Updated to reflect new GitHub organization.
Fixed broken links to GitHub pages.

Signed-off-by: David Galiffi <David.Galiffi@amd.com>

* update branch for 2.x documentation builds

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* update checkout action and use concurrency instead of cancel-workflow-action

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* test addition of user option for container launch

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* remove --user option for container, try chown instead

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* fixing yaml syntax

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* reorder job step - start with checkout

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* restore missing run directive

Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>

* Update workloads to include log.txt
Add missing MI200 workloads

Signed-off-by: Jose Santos <josantos@amd.com>

* Signed-off-by: Jose Santos <josantos@amd.com>
Add vcopy workload for tests

* Change exit codes for caught failures

Signed-off-by: Jose Santos <josantos@amd.com>

* reformat

Signed-off-by: Jose Santos <josantos@amd.com>

* Add pytest-xdist for pytest -n

Signed-off-by: Jose Santos <josantos@amd.com>

---------

Signed-off-by: David Galiffi <David.Galiffi@amd.com>
Signed-off-by: Karl W. Schulz <karl.schulz@amd.com>
Signed-off-by: Jose Santos <josantos@amd.com>
Co-authored-by: David Galiffi <David.Galiffi@amd.com>
Co-authored-by: Karl W. Schulz <karl.schulz@amd.com>
This commit is contained in:
JoseSantosAMD
2024-03-25 10:20:31 -05:00
zatwierdzone przez Cole Ramos
rodzic 482fd6f2ca
commit da506ad9b5
1122 zmienionych plików z 37938 dodań i 3563 usunięć
@@ -1 +1,4 @@
Index,KernelName
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295056,1295056,1048576,256,0,0,8,8,16,64,0x0,0x7fcafc83cec0,47758,47758,16384,65536,14395,1831576,1414609382362391,1414621055425142,1414621055449622,1414609390882981
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295056,1295056,1048576,256,0,0,8,8,16,64,0x0,0x7fcafc83cec0,46493,46493,16384,65536,8138,1048588,1414609390906726,1414621055537142,1414621055556022,1414609391222231
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295056,1295056,1048576,256,0,0,8,8,16,64,0x0,0x7fcafc83cec0,44994,44994,16384,65536,8169,1048584,1414609391254421,1414621055576182,1414621055595542,1414609391431225
1 Index Dispatch_ID KernelName Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295056 1295056 1048576 256 0 0 8 8 16 64 0x0 0x7fcafc83cec0 47758 47758 16384 65536 14395 1831576 1414609382362391 1414621055425142 1414621055449622 1414609390882981
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295056 1295056 1048576 256 0 0 8 8 16 64 0x0 0x7fcafc83cec0 46493 46493 16384 65536 8138 1048588 1414609390906726 1414621055537142 1414621055556022 1414609391222231
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295056 1295056 1048576 256 0 0 8 8 16 64 0x0 0x7fcafc83cec0 44994 44994 16384 65536 8169 1048584 1414609391254421 1414621055576182 1414621055595542 1414609391431225
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295257,1295257,1048576,256,0,0,8,8,16,64,0x0,0x7ff0afd60ec0,0,0,0,1414609872531785,1414621055425142,1414621055449622,1414609880166996
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295257,1295257,1048576,256,0,0,8,8,16,64,0x0,0x7ff0afd60ec0,0,0,0,1414609880186282,1414621055537142,1414621055556022,1414609880519821
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295257,1295257,1048576,256,0,0,8,8,16,64,0x0,0x7ff0afd60ec0,0,0,0,1414609880548024,1414621055576182,1414621055595542,1414609880734476
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295257 1295257 1048576 256 0 0 8 8 16 64 0x0 0x7ff0afd60ec0 0 0 0 1414609872531785 1414621055425142 1414621055449622 1414609880166996
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295257 1295257 1048576 256 0 0 8 8 16 64 0x0 0x7ff0afd60ec0 0 0 0 1414609880186282 1414621055537142 1414621055556022 1414609880519821
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295257 1295257 1048576 256 0 0 8 8 16 64 0x0 0x7ff0afd60ec0 0 0 0 1414609880548024 1414621055576182 1414621055595542 1414609880734476
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295459,1295459,1048576,256,0,0,8,8,16,64,0x0,0x7fbaa9898ec0,65536,192926,24614264,1414610346774400,1414621055425142,1414621055449622,1414610354390535
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295459,1295459,1048576,256,0,0,8,8,16,64,0x0,0x7fbaa9898ec0,65536,170544,21889328,1414610354409831,1414621055537142,1414621055556022,1414610354715538
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295459,1295459,1048576,256,0,0,8,8,16,64,0x0,0x7fbaa9898ec0,65536,173974,22234656,1414610354743811,1414621055576182,1414621055595542,1414610354923470
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295459 1295459 1048576 256 0 0 8 8 16 64 0x0 0x7fbaa9898ec0 65536 192926 24614264 1414610346774400 1414621055425142 1414621055449622 1414610354390535
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295459 1295459 1048576 256 0 0 8 8 16 64 0x0 0x7fbaa9898ec0 65536 170544 21889328 1414610354409831 1414621055537142 1414621055556022 1414610354715538
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295459 1295459 1048576 256 0 0 8 8 16 64 0x0 0x7fbaa9898ec0 65536 173974 22234656 1414610354743811 1414621055576182 1414621055595542 1414610354923470
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295661,1295661,1048576,256,0,0,8,8,16,64,0x0,0x7fcdb12dcec0,32768,664178,85014096,1414610810961751,1414621055425142,1414621055449622,1414610818650923
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295661,1295661,1048576,256,0,0,8,8,16,64,0x0,0x7fcdb12dcec0,32768,662767,84821212,1414610818672203,1414621055537142,1414621055556022,1414610818960106
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295661,1295661,1048576,256,0,0,8,8,16,64,0x0,0x7fcdb12dcec0,32768,655809,83963416,1414610818988480,1414621055576182,1414621055595542,1414610819205870
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295661 1295661 1048576 256 0 0 8 8 16 64 0x0 0x7fcdb12dcec0 32768 664178 85014096 1414610810961751 1414621055425142 1414621055449622 1414610818650923
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295661 1295661 1048576 256 0 0 8 8 16 64 0x0 0x7fcdb12dcec0 32768 662767 84821212 1414610818672203 1414621055537142 1414621055556022 1414610818960106
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295661 1295661 1048576 256 0 0 8 8 16 64 0x0 0x7fcdb12dcec0 32768 655809 83963416 1414610818988480 1414621055576182 1414621055595542 1414610819205870
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295847,1295847,1048576,256,0,0,8,8,16,64,0x0,0x7f5c3f6e4ec0,48355,48355,17730,386848,16384,24958385,233718,0,100367700,1414611287019490,1414621055425142,1414621055449622,1414611294707580
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295847,1295847,1048576,256,0,0,8,8,16,64,0x0,0x7f5c3f6e4ec0,43528,43528,13690,348232,16384,25762917,238429,0,103589552,1414611294736925,1414621055537142,1414621055556022,1414611295058863
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295847,1295847,1048576,256,0,0,8,8,16,64,0x0,0x7f5c3f6e4ec0,43639,43639,13971,349120,16384,25009170,233467,0,100574040,1414611295096984,1414621055576182,1414621055595542,1414611295279148
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295847 1295847 1048576 256 0 0 8 8 16 64 0x0 0x7f5c3f6e4ec0 48355 48355 17730 386848 16384 24958385 233718 0 100367700 1414611287019490 1414621055425142 1414621055449622 1414611294707580
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295847 1295847 1048576 256 0 0 8 8 16 64 0x0 0x7f5c3f6e4ec0 43528 43528 13690 348232 16384 25762917 238429 0 103589552 1414611294736925 1414621055537142 1414621055556022 1414611295058863
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295847 1295847 1048576 256 0 0 8 8 16 64 0x0 0x7f5c3f6e4ec0 43639 43639 13971 349120 16384 25009170 233467 0 100574040 1414611295096984 1414621055576182 1414621055595542 1414611295279148
+702
Wyświetl plik
@@ -0,0 +1,702 @@
Omniperf version: 2.0.0-RC1
Profiler choice: rocprofv1
Path: /home1/josantos/omniperf/tests/workloads/kernel/MI100
Target: MI100
Command: ./tests/vcopy -n 1048576 -b 256 -i 3
Kernel Selection: ['vecCopy']
Dispatch Selection: None
IP Blocks: All
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Collecting Performance Counters
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_IFETCH_LEVEL.txt
|-> [rocprof] RPL: on '240321_170755' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_IFETCH_LEVEL.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170755_1294896'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170755_1294896/input0_results_240321_170755'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170755_1294896/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 6 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, SQ_WAVES, SQ_IFETCH, SQ_IFETCH_LEVEL, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170755_1294896/input0_results_240321_170755
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_IFETCH_LEVEL.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_LDS.txt
|-> [rocprof] RPL: on '240321_170755' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_LDS.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170755_1295097'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170755_1295097/input0_results_240321_170755'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170755_1295097/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_LDS, SQ_INST_LEVEL_LDS, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170755_1295097/input0_results_240321_170755
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_INST_LEVEL_LDS.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_SMEM.txt
|-> [rocprof] RPL: on '240321_170756' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_SMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170756_1295299'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170756_1295299/input0_results_240321_170756'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170756_1295299/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_SMEM, SQ_INST_LEVEL_SMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170756_1295299/input0_results_240321_170756
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_INST_LEVEL_SMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_VMEM.txt
|-> [rocprof] RPL: on '240321_170756' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_VMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170756_1295501'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170756_1295501/input0_results_240321_170756'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170756_1295501/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_VMEM, SQ_INST_LEVEL_VMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170756_1295501/input0_results_240321_170756
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_INST_LEVEL_VMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_LEVEL_WAVES.txt
|-> [rocprof] RPL: on '240321_170757' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_LEVEL_WAVES.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170757_1295687'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170757_1295687/input0_results_240321_170757'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170757_1295687/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 9 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, CPC_ME1_BUSY_FOR_PACKET_DECODE, SQ_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQ_BUSY_CYCLES, SQ_LEVEL_WAVES, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170757_1295687/input0_results_240321_170757
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_LEVEL_WAVES.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_0.txt
|-> [rocprof] RPL: on '240321_170757' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_0.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170757_1295888'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170757_1295888/input0_results_240321_170757'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170757_1295888/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 28 metrics
|-> [rocprof] SQ_CYCLES, SQ_BUSY_CYCLES, SQ_BUSY_CU_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQC_TC_INST_REQ, SQC_TC_DATA_READ_REQ, SQC_TC_DATA_WRITE_REQ, GRBM_COUNT, GRBM_GUI_ACTIVE, TCP_GATE_EN1_sum, TCP_GATE_EN2_sum, TCP_TD_TCP_STALL_CYCLES_sum, TCP_TCR_TCP_STALL_CYCLES_sum, TA_TA_BUSY_sum, TA_BUFFER_WAVEFRONTS_sum, TD_TD_BUSY_sum, TD_TC_STALL_sum, SPI_CSN_WINDOW_VALID, SPI_CSN_BUSY, CPC_CPC_STAT_BUSY, CPC_CPC_STAT_IDLE, CPF_CPF_STAT_BUSY, CPF_CPF_STAT_STALL, TCC_CYCLE_sum, TCC_BUSY_sum, TCC_PROBE_sum, TCC_PROBE_ALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170757_1295888/input0_results_240321_170757
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_0.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_1.txt
|-> [rocprof] RPL: on '240321_170758' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_1.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170758_1296091'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170758_1296091/input0_results_240321_170758'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170758_1296091/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 27 metrics
|-> [rocprof] SQC_TC_DATA_ATOMIC_REQ, SQC_TC_STALL, SQC_TC_REQ, SQC_DCACHE_REQ_READ_16, SQC_ICACHE_REQ, SQC_ICACHE_HITS, SQC_ICACHE_MISSES, SQC_ICACHE_MISSES_DUPLICATE, GRBM_SPI_BUSY, TCP_READ_TAGCONFLICT_STALL_CYCLES_sum, TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum, TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum, TCP_TA_TCP_STATE_READ_sum, TA_BUFFER_READ_WAVEFRONTS_sum, TA_BUFFER_WRITE_WAVEFRONTS_sum, TD_COALESCABLE_WAVEFRONT_sum, TD_LOAD_WAVEFRONT_sum, SPI_CSN_NUM_THREADGROUPS, SPI_CSN_WAVE, CPC_CPC_TCIU_BUSY, CPC_CPC_TCIU_IDLE, CPF_CPF_TCIU_BUSY, CPF_CPF_TCIU_STALL, TCC_NC_REQ_sum, TCC_UC_REQ_sum, TCC_CC_REQ_sum, TCC_RW_REQ_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170758_1296091/input0_results_240321_170758
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_1.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_10.txt
|-> [rocprof] RPL: on '240321_170758' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_10.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170758_1296294'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170758_1296294/input0_results_240321_170758'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170758_1296294/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 1 metrics
|-> [rocprof] TCC_EA_ATOMIC_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170758_1296294/input0_results_240321_170758
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_10.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_11.txt
|-> [rocprof] RPL: on '240321_170759' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_11.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170759_1296495'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170759_1296495/input0_results_240321_170759'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170759_1296495/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_ATOMIC[0], TCC_CYCLE[0], TCC_EA_ATOMIC[0], TCC_EA_ATOMIC_LEVEL[0], TCC_ATOMIC[1], TCC_CYCLE[1], TCC_EA_ATOMIC[1], TCC_EA_ATOMIC_LEVEL[1], TCC_ATOMIC[2], TCC_CYCLE[2], TCC_EA_ATOMIC[2], TCC_EA_ATOMIC_LEVEL[2], TCC_ATOMIC[3], TCC_CYCLE[3], TCC_EA_ATOMIC[3], TCC_EA_ATOMIC_LEVEL[3], TCC_ATOMIC[4], TCC_CYCLE[4], TCC_EA_ATOMIC[4], TCC_EA_ATOMIC_LEVEL[4], TCC_ATOMIC[5], TCC_CYCLE[5], TCC_EA_ATOMIC[5], TCC_EA_ATOMIC_LEVEL[5], TCC_ATOMIC[6], TCC_CYCLE[6], TCC_EA_ATOMIC[6], TCC_EA_ATOMIC_LEVEL[6], TCC_ATOMIC[7], TCC_CYCLE[7], TCC_EA_ATOMIC[7], TCC_EA_ATOMIC_LEVEL[7], TCC_ATOMIC[8], TCC_CYCLE[8], TCC_EA_ATOMIC[8], TCC_EA_ATOMIC_LEVEL[8], TCC_ATOMIC[9], TCC_CYCLE[9], TCC_EA_ATOMIC[9], TCC_EA_ATOMIC_LEVEL[9], TCC_ATOMIC[10], TCC_CYCLE[10], TCC_EA_ATOMIC[10], TCC_EA_ATOMIC_LEVEL[10], TCC_ATOMIC[11], TCC_CYCLE[11], TCC_EA_ATOMIC[11], TCC_EA_ATOMIC_LEVEL[11], TCC_ATOMIC[12], TCC_CYCLE[12], TCC_EA_ATOMIC[12], TCC_EA_ATOMIC_LEVEL[12], TCC_ATOMIC[13], TCC_CYCLE[13], TCC_EA_ATOMIC[13], TCC_EA_ATOMIC_LEVEL[13], TCC_ATOMIC[14], TCC_CYCLE[14], TCC_EA_ATOMIC[14], TCC_EA_ATOMIC_LEVEL[14], TCC_ATOMIC[15], TCC_CYCLE[15], TCC_EA_ATOMIC[15], TCC_EA_ATOMIC_LEVEL[15], TCC_ATOMIC[16], TCC_CYCLE[16], TCC_EA_ATOMIC[16], TCC_EA_ATOMIC_LEVEL[16], TCC_ATOMIC[17], TCC_CYCLE[17], TCC_EA_ATOMIC[17], TCC_EA_ATOMIC_LEVEL[17], TCC_ATOMIC[18], TCC_CYCLE[18], TCC_EA_ATOMIC[18], TCC_EA_ATOMIC_LEVEL[18], TCC_ATOMIC[19], TCC_CYCLE[19], TCC_EA_ATOMIC[19], TCC_EA_ATOMIC_LEVEL[19], TCC_ATOMIC[20], TCC_CYCLE[20], TCC_EA_ATOMIC[20], TCC_EA_ATOMIC_LEVEL[20], TCC_ATOMIC[21], TCC_CYCLE[21], TCC_EA_ATOMIC[21], TCC_EA_ATOMIC_LEVEL[21], TCC_ATOMIC[22], TCC_CYCLE[22], TCC_EA_ATOMIC[22], TCC_EA_ATOMIC_LEVEL[22], TCC_ATOMIC[23], TCC_CYCLE[23], TCC_EA_ATOMIC[23], TCC_EA_ATOMIC_LEVEL[23], TCC_ATOMIC[24], TCC_CYCLE[24], TCC_EA_ATOMIC[24], TCC_EA_ATOMIC_LEVEL[24], TCC_ATOMIC[25], TCC_CYCLE[25], TCC_EA_ATOMIC[25], TCC_EA_ATOMIC_LEVEL[25], TCC_ATOMIC[26], TCC_CYCLE[26], TCC_EA_ATOMIC[26], TCC_EA_ATOMIC_LEVEL[26], TCC_ATOMIC[27], TCC_CYCLE[27], TCC_EA_ATOMIC[27], TCC_EA_ATOMIC_LEVEL[27], TCC_ATOMIC[28], TCC_CYCLE[28], TCC_EA_ATOMIC[28], TCC_EA_ATOMIC_LEVEL[28], TCC_ATOMIC[29], TCC_CYCLE[29], TCC_EA_ATOMIC[29], TCC_EA_ATOMIC_LEVEL[29], TCC_ATOMIC[30], TCC_CYCLE[30], TCC_EA_ATOMIC[30], TCC_EA_ATOMIC_LEVEL[30], TCC_ATOMIC[31], TCC_CYCLE[31], TCC_EA_ATOMIC[31], TCC_EA_ATOMIC_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170759_1296495/input0_results_240321_170759
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_11.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_12.txt
|-> [rocprof] RPL: on '240321_170759' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_12.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170759_1296697'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170759_1296697/input0_results_240321_170759'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170759_1296697/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ[0], TCC_EA_RDREQ_32B[0], TCC_EA_RDREQ_DRAM_CREDIT_STALL[0], TCC_EA_RDREQ_GMI_CREDIT_STALL[0], TCC_EA_RDREQ[1], TCC_EA_RDREQ_32B[1], TCC_EA_RDREQ_DRAM_CREDIT_STALL[1], TCC_EA_RDREQ_GMI_CREDIT_STALL[1], TCC_EA_RDREQ[2], TCC_EA_RDREQ_32B[2], TCC_EA_RDREQ_DRAM_CREDIT_STALL[2], TCC_EA_RDREQ_GMI_CREDIT_STALL[2], TCC_EA_RDREQ[3], TCC_EA_RDREQ_32B[3], TCC_EA_RDREQ_DRAM_CREDIT_STALL[3], TCC_EA_RDREQ_GMI_CREDIT_STALL[3], TCC_EA_RDREQ[4], TCC_EA_RDREQ_32B[4], TCC_EA_RDREQ_DRAM_CREDIT_STALL[4], TCC_EA_RDREQ_GMI_CREDIT_STALL[4], TCC_EA_RDREQ[5], TCC_EA_RDREQ_32B[5], TCC_EA_RDREQ_DRAM_CREDIT_STALL[5], TCC_EA_RDREQ_GMI_CREDIT_STALL[5], TCC_EA_RDREQ[6], TCC_EA_RDREQ_32B[6], TCC_EA_RDREQ_DRAM_CREDIT_STALL[6], TCC_EA_RDREQ_GMI_CREDIT_STALL[6], TCC_EA_RDREQ[7], TCC_EA_RDREQ_32B[7], TCC_EA_RDREQ_DRAM_CREDIT_STALL[7], TCC_EA_RDREQ_GMI_CREDIT_STALL[7], TCC_EA_RDREQ[8], TCC_EA_RDREQ_32B[8], TCC_EA_RDREQ_DRAM_CREDIT_STALL[8], TCC_EA_RDREQ_GMI_CREDIT_STALL[8], TCC_EA_RDREQ[9], TCC_EA_RDREQ_32B[9], TCC_EA_RDREQ_DRAM_CREDIT_STALL[9], TCC_EA_RDREQ_GMI_CREDIT_STALL[9], TCC_EA_RDREQ[10], TCC_EA_RDREQ_32B[10], TCC_EA_RDREQ_DRAM_CREDIT_STALL[10], TCC_EA_RDREQ_GMI_CREDIT_STALL[10], TCC_EA_RDREQ[11], TCC_EA_RDREQ_32B[11], TCC_EA_RDREQ_DRAM_CREDIT_STALL[11], TCC_EA_RDREQ_GMI_CREDIT_STALL[11], TCC_EA_RDREQ[12], TCC_EA_RDREQ_32B[12], TCC_EA_RDREQ_DRAM_CREDIT_STALL[12], TCC_EA_RDREQ_GMI_CREDIT_STALL[12], TCC_EA_RDREQ[13], TCC_EA_RDREQ_32B[13], TCC_EA_RDREQ_DRAM_CREDIT_STALL[13], TCC_EA_RDREQ_GMI_CREDIT_STALL[13], TCC_EA_RDREQ[14], TCC_EA_RDREQ_32B[14], TCC_EA_RDREQ_DRAM_CREDIT_STALL[14], TCC_EA_RDREQ_GMI_CREDIT_STALL[14], TCC_EA_RDREQ[15], TCC_EA_RDREQ_32B[15], TCC_EA_RDREQ_DRAM_CREDIT_STALL[15], TCC_EA_RDREQ_GMI_CREDIT_STALL[15], TCC_EA_RDREQ[16], TCC_EA_RDREQ_32B[16], TCC_EA_RDREQ_DRAM_CREDIT_STALL[16], TCC_EA_RDREQ_GMI_CREDIT_STALL[16], TCC_EA_RDREQ[17], TCC_EA_RDREQ_32B[17], TCC_EA_RDREQ_DRAM_CREDIT_STALL[17], TCC_EA_RDREQ_GMI_CREDIT_STALL[17], TCC_EA_RDREQ[18], TCC_EA_RDREQ_32B[18], TCC_EA_RDREQ_DRAM_CREDIT_STALL[18], TCC_EA_RDREQ_GMI_CREDIT_STALL[18], TCC_EA_RDREQ[19], TCC_EA_RDREQ_32B[19], TCC_EA_RDREQ_DRAM_CREDIT_STALL[19], TCC_EA_RDREQ_GMI_CREDIT_STALL[19], TCC_EA_RDREQ[20], TCC_EA_RDREQ_32B[20], TCC_EA_RDREQ_DRAM_CREDIT_STALL[20], TCC_EA_RDREQ_GMI_CREDIT_STALL[20], TCC_EA_RDREQ[21], TCC_EA_RDREQ_32B[21], TCC_EA_RDREQ_DRAM_CREDIT_STALL[21], TCC_EA_RDREQ_GMI_CREDIT_STALL[21], TCC_EA_RDREQ[22], TCC_EA_RDREQ_32B[22], TCC_EA_RDREQ_DRAM_CREDIT_STALL[22], TCC_EA_RDREQ_GMI_CREDIT_STALL[22], TCC_EA_RDREQ[23], TCC_EA_RDREQ_32B[23], TCC_EA_RDREQ_DRAM_CREDIT_STALL[23], TCC_EA_RDREQ_GMI_CREDIT_STALL[23], TCC_EA_RDREQ[24], TCC_EA_RDREQ_32B[24], TCC_EA_RDREQ_DRAM_CREDIT_STALL[24], TCC_EA_RDREQ_GMI_CREDIT_STALL[24], TCC_EA_RDREQ[25], TCC_EA_RDREQ_32B[25], TCC_EA_RDREQ_DRAM_CREDIT_STALL[25], TCC_EA_RDREQ_GMI_CREDIT_STALL[25], TCC_EA_RDREQ[26], TCC_EA_RDREQ_32B[26], TCC_EA_RDREQ_DRAM_CREDIT_STALL[26], TCC_EA_RDREQ_GMI_CREDIT_STALL[26], TCC_EA_RDREQ[27], TCC_EA_RDREQ_32B[27], TCC_EA_RDREQ_DRAM_CREDIT_STALL[27], TCC_EA_RDREQ_GMI_CREDIT_STALL[27], TCC_EA_RDREQ[28], TCC_EA_RDREQ_32B[28], TCC_EA_RDREQ_DRAM_CREDIT_STALL[28], TCC_EA_RDREQ_GMI_CREDIT_STALL[28], TCC_EA_RDREQ[29], TCC_EA_RDREQ_32B[29], TCC_EA_RDREQ_DRAM_CREDIT_STALL[29], TCC_EA_RDREQ_GMI_CREDIT_STALL[29], TCC_EA_RDREQ[30], TCC_EA_RDREQ_32B[30], TCC_EA_RDREQ_DRAM_CREDIT_STALL[30], TCC_EA_RDREQ_GMI_CREDIT_STALL[30], TCC_EA_RDREQ[31], TCC_EA_RDREQ_32B[31], TCC_EA_RDREQ_DRAM_CREDIT_STALL[31], TCC_EA_RDREQ_GMI_CREDIT_STALL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170759_1296697/input0_results_240321_170759
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_12.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_13.txt
|-> [rocprof] RPL: on '240321_170800' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_13.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170800_1296901'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170800_1296901/input0_results_240321_170800'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170800_1296901/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ_IO_CREDIT_STALL[0], TCC_EA_RDREQ_LEVEL[0], TCC_EA_WRREQ[0], TCC_EA_WRREQ_64B[0], TCC_EA_RDREQ_IO_CREDIT_STALL[1], TCC_EA_RDREQ_LEVEL[1], TCC_EA_WRREQ[1], TCC_EA_WRREQ_64B[1], TCC_EA_RDREQ_IO_CREDIT_STALL[2], TCC_EA_RDREQ_LEVEL[2], TCC_EA_WRREQ[2], TCC_EA_WRREQ_64B[2], TCC_EA_RDREQ_IO_CREDIT_STALL[3], TCC_EA_RDREQ_LEVEL[3], TCC_EA_WRREQ[3], TCC_EA_WRREQ_64B[3], TCC_EA_RDREQ_IO_CREDIT_STALL[4], TCC_EA_RDREQ_LEVEL[4], TCC_EA_WRREQ[4], TCC_EA_WRREQ_64B[4], TCC_EA_RDREQ_IO_CREDIT_STALL[5], TCC_EA_RDREQ_LEVEL[5], TCC_EA_WRREQ[5], TCC_EA_WRREQ_64B[5], TCC_EA_RDREQ_IO_CREDIT_STALL[6], TCC_EA_RDREQ_LEVEL[6], TCC_EA_WRREQ[6], TCC_EA_WRREQ_64B[6], TCC_EA_RDREQ_IO_CREDIT_STALL[7], TCC_EA_RDREQ_LEVEL[7], TCC_EA_WRREQ[7], TCC_EA_WRREQ_64B[7], TCC_EA_RDREQ_IO_CREDIT_STALL[8], TCC_EA_RDREQ_LEVEL[8], TCC_EA_WRREQ[8], TCC_EA_WRREQ_64B[8], TCC_EA_RDREQ_IO_CREDIT_STALL[9], TCC_EA_RDREQ_LEVEL[9], TCC_EA_WRREQ[9], TCC_EA_WRREQ_64B[9], TCC_EA_RDREQ_IO_CREDIT_STALL[10], TCC_EA_RDREQ_LEVEL[10], TCC_EA_WRREQ[10], TCC_EA_WRREQ_64B[10], TCC_EA_RDREQ_IO_CREDIT_STALL[11], TCC_EA_RDREQ_LEVEL[11], TCC_EA_WRREQ[11], TCC_EA_WRREQ_64B[11], TCC_EA_RDREQ_IO_CREDIT_STALL[12], TCC_EA_RDREQ_LEVEL[12], TCC_EA_WRREQ[12], TCC_EA_WRREQ_64B[12], TCC_EA_RDREQ_IO_CREDIT_STALL[13], TCC_EA_RDREQ_LEVEL[13], TCC_EA_WRREQ[13], TCC_EA_WRREQ_64B[13], TCC_EA_RDREQ_IO_CREDIT_STALL[14], TCC_EA_RDREQ_LEVEL[14], TCC_EA_WRREQ[14], TCC_EA_WRREQ_64B[14], TCC_EA_RDREQ_IO_CREDIT_STALL[15], TCC_EA_RDREQ_LEVEL[15], TCC_EA_WRREQ[15], TCC_EA_WRREQ_64B[15], TCC_EA_RDREQ_IO_CREDIT_STALL[16], TCC_EA_RDREQ_LEVEL[16], TCC_EA_WRREQ[16], TCC_EA_WRREQ_64B[16], TCC_EA_RDREQ_IO_CREDIT_STALL[17], TCC_EA_RDREQ_LEVEL[17], TCC_EA_WRREQ[17], TCC_EA_WRREQ_64B[17], TCC_EA_RDREQ_IO_CREDIT_STALL[18], TCC_EA_RDREQ_LEVEL[18], TCC_EA_WRREQ[18], TCC_EA_WRREQ_64B[18], TCC_EA_RDREQ_IO_CREDIT_STALL[19], TCC_EA_RDREQ_LEVEL[19], TCC_EA_WRREQ[19], TCC_EA_WRREQ_64B[19], TCC_EA_RDREQ_IO_CREDIT_STALL[20], TCC_EA_RDREQ_LEVEL[20], TCC_EA_WRREQ[20], TCC_EA_WRREQ_64B[20], TCC_EA_RDREQ_IO_CREDIT_STALL[21], TCC_EA_RDREQ_LEVEL[21], TCC_EA_WRREQ[21], TCC_EA_WRREQ_64B[21], TCC_EA_RDREQ_IO_CREDIT_STALL[22], TCC_EA_RDREQ_LEVEL[22], TCC_EA_WRREQ[22], TCC_EA_WRREQ_64B[22], TCC_EA_RDREQ_IO_CREDIT_STALL[23], TCC_EA_RDREQ_LEVEL[23], TCC_EA_WRREQ[23], TCC_EA_WRREQ_64B[23], TCC_EA_RDREQ_IO_CREDIT_STALL[24], TCC_EA_RDREQ_LEVEL[24], TCC_EA_WRREQ[24], TCC_EA_WRREQ_64B[24], TCC_EA_RDREQ_IO_CREDIT_STALL[25], TCC_EA_RDREQ_LEVEL[25], TCC_EA_WRREQ[25], TCC_EA_WRREQ_64B[25], TCC_EA_RDREQ_IO_CREDIT_STALL[26], TCC_EA_RDREQ_LEVEL[26], TCC_EA_WRREQ[26], TCC_EA_WRREQ_64B[26], TCC_EA_RDREQ_IO_CREDIT_STALL[27], TCC_EA_RDREQ_LEVEL[27], TCC_EA_WRREQ[27], TCC_EA_WRREQ_64B[27], TCC_EA_RDREQ_IO_CREDIT_STALL[28], TCC_EA_RDREQ_LEVEL[28], TCC_EA_WRREQ[28], TCC_EA_WRREQ_64B[28], TCC_EA_RDREQ_IO_CREDIT_STALL[29], TCC_EA_RDREQ_LEVEL[29], TCC_EA_WRREQ[29], TCC_EA_WRREQ_64B[29], TCC_EA_RDREQ_IO_CREDIT_STALL[30], TCC_EA_RDREQ_LEVEL[30], TCC_EA_WRREQ[30], TCC_EA_WRREQ_64B[30], TCC_EA_RDREQ_IO_CREDIT_STALL[31], TCC_EA_RDREQ_LEVEL[31], TCC_EA_WRREQ[31], TCC_EA_WRREQ_64B[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170800_1296901/input0_results_240321_170800
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_13.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_14.txt
|-> [rocprof] RPL: on '240321_170801' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_14.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170801_1297099'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170801_1297099/input0_results_240321_170801'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170801_1297099/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_WRREQ_DRAM_CREDIT_STALL[0], TCC_EA_WRREQ_GMI_CREDIT_STALL[0], TCC_EA_WRREQ_IO_CREDIT_STALL[0], TCC_EA_WRREQ_LEVEL[0], TCC_EA_WRREQ_DRAM_CREDIT_STALL[1], TCC_EA_WRREQ_GMI_CREDIT_STALL[1], TCC_EA_WRREQ_IO_CREDIT_STALL[1], TCC_EA_WRREQ_LEVEL[1], TCC_EA_WRREQ_DRAM_CREDIT_STALL[2], TCC_EA_WRREQ_GMI_CREDIT_STALL[2], TCC_EA_WRREQ_IO_CREDIT_STALL[2], TCC_EA_WRREQ_LEVEL[2], TCC_EA_WRREQ_DRAM_CREDIT_STALL[3], TCC_EA_WRREQ_GMI_CREDIT_STALL[3], TCC_EA_WRREQ_IO_CREDIT_STALL[3], TCC_EA_WRREQ_LEVEL[3], TCC_EA_WRREQ_DRAM_CREDIT_STALL[4], TCC_EA_WRREQ_GMI_CREDIT_STALL[4], TCC_EA_WRREQ_IO_CREDIT_STALL[4], TCC_EA_WRREQ_LEVEL[4], TCC_EA_WRREQ_DRAM_CREDIT_STALL[5], TCC_EA_WRREQ_GMI_CREDIT_STALL[5], TCC_EA_WRREQ_IO_CREDIT_STALL[5], TCC_EA_WRREQ_LEVEL[5], TCC_EA_WRREQ_DRAM_CREDIT_STALL[6], TCC_EA_WRREQ_GMI_CREDIT_STALL[6], TCC_EA_WRREQ_IO_CREDIT_STALL[6], TCC_EA_WRREQ_LEVEL[6], TCC_EA_WRREQ_DRAM_CREDIT_STALL[7], TCC_EA_WRREQ_GMI_CREDIT_STALL[7], TCC_EA_WRREQ_IO_CREDIT_STALL[7], TCC_EA_WRREQ_LEVEL[7], TCC_EA_WRREQ_DRAM_CREDIT_STALL[8], TCC_EA_WRREQ_GMI_CREDIT_STALL[8], TCC_EA_WRREQ_IO_CREDIT_STALL[8], TCC_EA_WRREQ_LEVEL[8], TCC_EA_WRREQ_DRAM_CREDIT_STALL[9], TCC_EA_WRREQ_GMI_CREDIT_STALL[9], TCC_EA_WRREQ_IO_CREDIT_STALL[9], TCC_EA_WRREQ_LEVEL[9], TCC_EA_WRREQ_DRAM_CREDIT_STALL[10], TCC_EA_WRREQ_GMI_CREDIT_STALL[10], TCC_EA_WRREQ_IO_CREDIT_STALL[10], TCC_EA_WRREQ_LEVEL[10], TCC_EA_WRREQ_DRAM_CREDIT_STALL[11], TCC_EA_WRREQ_GMI_CREDIT_STALL[11], TCC_EA_WRREQ_IO_CREDIT_STALL[11], TCC_EA_WRREQ_LEVEL[11], TCC_EA_WRREQ_DRAM_CREDIT_STALL[12], TCC_EA_WRREQ_GMI_CREDIT_STALL[12], TCC_EA_WRREQ_IO_CREDIT_STALL[12], TCC_EA_WRREQ_LEVEL[12], TCC_EA_WRREQ_DRAM_CREDIT_STALL[13], TCC_EA_WRREQ_GMI_CREDIT_STALL[13], TCC_EA_WRREQ_IO_CREDIT_STALL[13], TCC_EA_WRREQ_LEVEL[13], TCC_EA_WRREQ_DRAM_CREDIT_STALL[14], TCC_EA_WRREQ_GMI_CREDIT_STALL[14], TCC_EA_WRREQ_IO_CREDIT_STALL[14], TCC_EA_WRREQ_LEVEL[14], TCC_EA_WRREQ_DRAM_CREDIT_STALL[15], TCC_EA_WRREQ_GMI_CREDIT_STALL[15], TCC_EA_WRREQ_IO_CREDIT_STALL[15], TCC_EA_WRREQ_LEVEL[15], TCC_EA_WRREQ_DRAM_CREDIT_STALL[16], TCC_EA_WRREQ_GMI_CREDIT_STALL[16], TCC_EA_WRREQ_IO_CREDIT_STALL[16], TCC_EA_WRREQ_LEVEL[16], TCC_EA_WRREQ_DRAM_CREDIT_STALL[17], TCC_EA_WRREQ_GMI_CREDIT_STALL[17], TCC_EA_WRREQ_IO_CREDIT_STALL[17], TCC_EA_WRREQ_LEVEL[17], TCC_EA_WRREQ_DRAM_CREDIT_STALL[18], TCC_EA_WRREQ_GMI_CREDIT_STALL[18], TCC_EA_WRREQ_IO_CREDIT_STALL[18], TCC_EA_WRREQ_LEVEL[18], TCC_EA_WRREQ_DRAM_CREDIT_STALL[19], TCC_EA_WRREQ_GMI_CREDIT_STALL[19], TCC_EA_WRREQ_IO_CREDIT_STALL[19], TCC_EA_WRREQ_LEVEL[19], TCC_EA_WRREQ_DRAM_CREDIT_STALL[20], TCC_EA_WRREQ_GMI_CREDIT_STALL[20], TCC_EA_WRREQ_IO_CREDIT_STALL[20], TCC_EA_WRREQ_LEVEL[20], TCC_EA_WRREQ_DRAM_CREDIT_STALL[21], TCC_EA_WRREQ_GMI_CREDIT_STALL[21], TCC_EA_WRREQ_IO_CREDIT_STALL[21], TCC_EA_WRREQ_LEVEL[21], TCC_EA_WRREQ_DRAM_CREDIT_STALL[22], TCC_EA_WRREQ_GMI_CREDIT_STALL[22], TCC_EA_WRREQ_IO_CREDIT_STALL[22], TCC_EA_WRREQ_LEVEL[22], TCC_EA_WRREQ_DRAM_CREDIT_STALL[23], TCC_EA_WRREQ_GMI_CREDIT_STALL[23], TCC_EA_WRREQ_IO_CREDIT_STALL[23], TCC_EA_WRREQ_LEVEL[23], TCC_EA_WRREQ_DRAM_CREDIT_STALL[24], TCC_EA_WRREQ_GMI_CREDIT_STALL[24], TCC_EA_WRREQ_IO_CREDIT_STALL[24], TCC_EA_WRREQ_LEVEL[24], TCC_EA_WRREQ_DRAM_CREDIT_STALL[25], TCC_EA_WRREQ_GMI_CREDIT_STALL[25], TCC_EA_WRREQ_IO_CREDIT_STALL[25], TCC_EA_WRREQ_LEVEL[25], TCC_EA_WRREQ_DRAM_CREDIT_STALL[26], TCC_EA_WRREQ_GMI_CREDIT_STALL[26], TCC_EA_WRREQ_IO_CREDIT_STALL[26], TCC_EA_WRREQ_LEVEL[26], TCC_EA_WRREQ_DRAM_CREDIT_STALL[27], TCC_EA_WRREQ_GMI_CREDIT_STALL[27], TCC_EA_WRREQ_IO_CREDIT_STALL[27], TCC_EA_WRREQ_LEVEL[27], TCC_EA_WRREQ_DRAM_CREDIT_STALL[28], TCC_EA_WRREQ_GMI_CREDIT_STALL[28], TCC_EA_WRREQ_IO_CREDIT_STALL[28], TCC_EA_WRREQ_LEVEL[28], TCC_EA_WRREQ_DRAM_CREDIT_STALL[29], TCC_EA_WRREQ_GMI_CREDIT_STALL[29], TCC_EA_WRREQ_IO_CREDIT_STALL[29], TCC_EA_WRREQ_LEVEL[29], TCC_EA_WRREQ_DRAM_CREDIT_STALL[30], TCC_EA_WRREQ_GMI_CREDIT_STALL[30], TCC_EA_WRREQ_IO_CREDIT_STALL[30], TCC_EA_WRREQ_LEVEL[30], TCC_EA_WRREQ_DRAM_CREDIT_STALL[31], TCC_EA_WRREQ_GMI_CREDIT_STALL[31], TCC_EA_WRREQ_IO_CREDIT_STALL[31], TCC_EA_WRREQ_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170801_1297099/input0_results_240321_170801
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_14.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_15.txt
|-> [rocprof] RPL: on '240321_170801' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_15.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170801_1297303'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170801_1297303/input0_results_240321_170801'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170801_1297303/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_HIT[0], TCC_MISS[0], TCC_READ[0], TCC_REQ[0], TCC_HIT[1], TCC_MISS[1], TCC_READ[1], TCC_REQ[1], TCC_HIT[2], TCC_MISS[2], TCC_READ[2], TCC_REQ[2], TCC_HIT[3], TCC_MISS[3], TCC_READ[3], TCC_REQ[3], TCC_HIT[4], TCC_MISS[4], TCC_READ[4], TCC_REQ[4], TCC_HIT[5], TCC_MISS[5], TCC_READ[5], TCC_REQ[5], TCC_HIT[6], TCC_MISS[6], TCC_READ[6], TCC_REQ[6], TCC_HIT[7], TCC_MISS[7], TCC_READ[7], TCC_REQ[7], TCC_HIT[8], TCC_MISS[8], TCC_READ[8], TCC_REQ[8], TCC_HIT[9], TCC_MISS[9], TCC_READ[9], TCC_REQ[9], TCC_HIT[10], TCC_MISS[10], TCC_READ[10], TCC_REQ[10], TCC_HIT[11], TCC_MISS[11], TCC_READ[11], TCC_REQ[11], TCC_HIT[12], TCC_MISS[12], TCC_READ[12], TCC_REQ[12], TCC_HIT[13], TCC_MISS[13], TCC_READ[13], TCC_REQ[13], TCC_HIT[14], TCC_MISS[14], TCC_READ[14], TCC_REQ[14], TCC_HIT[15], TCC_MISS[15], TCC_READ[15], TCC_REQ[15], TCC_HIT[16], TCC_MISS[16], TCC_READ[16], TCC_REQ[16], TCC_HIT[17], TCC_MISS[17], TCC_READ[17], TCC_REQ[17], TCC_HIT[18], TCC_MISS[18], TCC_READ[18], TCC_REQ[18], TCC_HIT[19], TCC_MISS[19], TCC_READ[19], TCC_REQ[19], TCC_HIT[20], TCC_MISS[20], TCC_READ[20], TCC_REQ[20], TCC_HIT[21], TCC_MISS[21], TCC_READ[21], TCC_REQ[21], TCC_HIT[22], TCC_MISS[22], TCC_READ[22], TCC_REQ[22], TCC_HIT[23], TCC_MISS[23], TCC_READ[23], TCC_REQ[23], TCC_HIT[24], TCC_MISS[24], TCC_READ[24], TCC_REQ[24], TCC_HIT[25], TCC_MISS[25], TCC_READ[25], TCC_REQ[25], TCC_HIT[26], TCC_MISS[26], TCC_READ[26], TCC_REQ[26], TCC_HIT[27], TCC_MISS[27], TCC_READ[27], TCC_REQ[27], TCC_HIT[28], TCC_MISS[28], TCC_READ[28], TCC_REQ[28], TCC_HIT[29], TCC_MISS[29], TCC_READ[29], TCC_REQ[29], TCC_HIT[30], TCC_MISS[30], TCC_READ[30], TCC_REQ[30], TCC_HIT[31], TCC_MISS[31], TCC_READ[31], TCC_REQ[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170801_1297303/input0_results_240321_170801
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_15.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_16.txt
|-> [rocprof] RPL: on '240321_170802' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_16.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170802_1297506'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170802_1297506/input0_results_240321_170802'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170802_1297506/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 96 metrics
|-> [rocprof] TCC_RW_REQ[0], TCC_TOO_MANY_EA_WRREQS_STALL[0], TCC_WRITE[0], TCC_RW_REQ[1], TCC_TOO_MANY_EA_WRREQS_STALL[1], TCC_WRITE[1], TCC_RW_REQ[2], TCC_TOO_MANY_EA_WRREQS_STALL[2], TCC_WRITE[2], TCC_RW_REQ[3], TCC_TOO_MANY_EA_WRREQS_STALL[3], TCC_WRITE[3], TCC_RW_REQ[4], TCC_TOO_MANY_EA_WRREQS_STALL[4], TCC_WRITE[4], TCC_RW_REQ[5], TCC_TOO_MANY_EA_WRREQS_STALL[5], TCC_WRITE[5], TCC_RW_REQ[6], TCC_TOO_MANY_EA_WRREQS_STALL[6], TCC_WRITE[6], TCC_RW_REQ[7], TCC_TOO_MANY_EA_WRREQS_STALL[7], TCC_WRITE[7], TCC_RW_REQ[8], TCC_TOO_MANY_EA_WRREQS_STALL[8], TCC_WRITE[8], TCC_RW_REQ[9], TCC_TOO_MANY_EA_WRREQS_STALL[9], TCC_WRITE[9], TCC_RW_REQ[10], TCC_TOO_MANY_EA_WRREQS_STALL[10], TCC_WRITE[10], TCC_RW_REQ[11], TCC_TOO_MANY_EA_WRREQS_STALL[11], TCC_WRITE[11], TCC_RW_REQ[12], TCC_TOO_MANY_EA_WRREQS_STALL[12], TCC_WRITE[12], TCC_RW_REQ[13], TCC_TOO_MANY_EA_WRREQS_STALL[13], TCC_WRITE[13], TCC_RW_REQ[14], TCC_TOO_MANY_EA_WRREQS_STALL[14], TCC_WRITE[14], TCC_RW_REQ[15], TCC_TOO_MANY_EA_WRREQS_STALL[15], TCC_WRITE[15], TCC_RW_REQ[16], TCC_TOO_MANY_EA_WRREQS_STALL[16], TCC_WRITE[16], TCC_RW_REQ[17], TCC_TOO_MANY_EA_WRREQS_STALL[17], TCC_WRITE[17], TCC_RW_REQ[18], TCC_TOO_MANY_EA_WRREQS_STALL[18], TCC_WRITE[18], TCC_RW_REQ[19], TCC_TOO_MANY_EA_WRREQS_STALL[19], TCC_WRITE[19], TCC_RW_REQ[20], TCC_TOO_MANY_EA_WRREQS_STALL[20], TCC_WRITE[20], TCC_RW_REQ[21], TCC_TOO_MANY_EA_WRREQS_STALL[21], TCC_WRITE[21], TCC_RW_REQ[22], TCC_TOO_MANY_EA_WRREQS_STALL[22], TCC_WRITE[22], TCC_RW_REQ[23], TCC_TOO_MANY_EA_WRREQS_STALL[23], TCC_WRITE[23], TCC_RW_REQ[24], TCC_TOO_MANY_EA_WRREQS_STALL[24], TCC_WRITE[24], TCC_RW_REQ[25], TCC_TOO_MANY_EA_WRREQS_STALL[25], TCC_WRITE[25], TCC_RW_REQ[26], TCC_TOO_MANY_EA_WRREQS_STALL[26], TCC_WRITE[26], TCC_RW_REQ[27], TCC_TOO_MANY_EA_WRREQS_STALL[27], TCC_WRITE[27], TCC_RW_REQ[28], TCC_TOO_MANY_EA_WRREQS_STALL[28], TCC_WRITE[28], TCC_RW_REQ[29], TCC_TOO_MANY_EA_WRREQS_STALL[29], TCC_WRITE[29], TCC_RW_REQ[30], TCC_TOO_MANY_EA_WRREQS_STALL[30], TCC_WRITE[30], TCC_RW_REQ[31], TCC_TOO_MANY_EA_WRREQS_STALL[31], TCC_WRITE[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170802_1297506/input0_results_240321_170802
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_16.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_2.txt
|-> [rocprof] RPL: on '240321_170803' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_2.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170803_1297703'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170803_1297703/input0_results_240321_170803'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170803_1297703/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 26 metrics
|-> [rocprof] SQC_DCACHE_INPUT_VALID_READYB, SQC_DCACHE_ATOMIC, SQC_DCACHE_REQ_READ_8, SQC_DCACHE_REQ, SQC_DCACHE_HITS, SQC_DCACHE_MISSES, SQC_DCACHE_MISSES_DUPLICATE, SQC_DCACHE_REQ_READ_1, TCP_VOLATILE_sum, TCP_TOTAL_ACCESSES_sum, TCP_TOTAL_READ_sum, TCP_TOTAL_WRITE_sum, TA_BUFFER_ATOMIC_WAVEFRONTS_sum, TA_BUFFER_TOTAL_CYCLES_sum, TD_ATOMIC_WAVEFRONT_sum, TD_STORE_WAVEFRONT_sum, SPI_RA_REQ_NO_ALLOC, SPI_RA_REQ_NO_ALLOC_CSN, CPC_CPC_STAT_STALL, CPC_UTCL1_STALL_ON_TRANSLATION, CPF_CPF_STAT_IDLE, CPF_CPF_TCIU_IDLE, TCC_REQ_sum, TCC_STREAMING_REQ_sum, TCC_HIT_sum, TCC_MISS_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170803_1297703/input0_results_240321_170803
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_2.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_3.txt
|-> [rocprof] RPL: on '240321_170803' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_3.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170803_1297890'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170803_1297890/input0_results_240321_170803'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170803_1297890/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 23 metrics
|-> [rocprof] SQC_DCACHE_REQ_READ_2, SQC_DCACHE_REQ_READ_4, SQ_INSTS_VMEM_WR, SQ_INSTS_VMEM_RD, SQ_INSTS_VMEM, SQ_INSTS_SALU, SQ_INSTS_VSKIPPED, SQ_INSTS_SMEM, TCP_TOTAL_ATOMIC_WITH_RET_sum, TCP_TOTAL_ATOMIC_WITHOUT_RET_sum, TCP_TOTAL_WRITEBACK_INVALIDATES_sum, TCP_TOTAL_CACHE_ACCESSES_sum, TA_BUFFER_COALESCED_READ_CYCLES_sum, TA_BUFFER_COALESCED_WRITE_CYCLES_sum, SPI_RA_RES_STALL_CSN, SPI_RA_TMP_STALL_CSN, CPC_CPC_UTCL2IU_BUSY, CPC_CPC_UTCL2IU_IDLE, CPF_CMP_UTCL1_STALL_ON_TRANSLATION, TCC_READ_sum, TCC_WRITE_sum, TCC_ATOMIC_sum, TCC_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170803_1297890/input0_results_240321_170803
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_3.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_4.txt
|-> [rocprof] RPL: on '240321_170804' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_4.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170804_1298091'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170804_1298091/input0_results_240321_170804'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170804_1298091/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 22 metrics
|-> [rocprof] SQ_INSTS_FLAT, SQ_INSTS_LDS, SQ_INSTS_GDS, SQ_INSTS_EXP_GDS, SQ_INSTS_BRANCH, SQ_INSTS_SENDMSG, SQ_INSTS, SQ_WAIT_ANY, TCP_UTCL1_TRANSLATION_MISS_sum, TCP_UTCL1_TRANSLATION_HIT_sum, TCP_UTCL1_PERMISSION_MISS_sum, TCP_UTCL1_REQUEST_sum, TA_ADDR_STALLED_BY_TC_CYCLES_sum, TA_TOTAL_WAVEFRONTS_sum, SPI_RA_WAVE_SIMD_FULL_CSN, SPI_RA_VGPR_SIMD_FULL_CSN, CPC_CPC_UTCL2IU_STALL, CPC_ME1_BUSY_FOR_PACKET_DECODE, TCC_EA_WRREQ_sum, TCC_EA_WRREQ_64B_sum, TCC_EA_WR_UNCACHED_32B_sum, TCC_EA_WRREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170804_1298091/input0_results_240321_170804
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_4.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_5.txt
|-> [rocprof] RPL: on '240321_170804' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_5.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170804_1298276'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170804_1298276/input0_results_240321_170804'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170804_1298276/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 21 metrics
|-> [rocprof] SQ_WAIT_INST_ANY, SQ_ACTIVE_INST_ANY, SQ_INSTS_VALU, SQ_ACTIVE_INST_VMEM, SQ_ACTIVE_INST_LDS, SQ_ACTIVE_INST_VALU, SQ_ACTIVE_INST_SCA, SQ_ACTIVE_INST_EXP_GDS, TCP_TCP_LATENCY_sum, TCP_TCC_READ_REQ_LATENCY_sum, TCP_TCC_WRITE_REQ_LATENCY_sum, TCP_TCC_READ_REQ_sum, TA_ADDR_STALLED_BY_TD_CYCLES_sum, TA_DATA_STALLED_BY_TC_CYCLES_sum, SPI_RA_SGPR_SIMD_FULL_CSN, SPI_RA_LDS_CU_FULL_CSN, CPC_ME1_DC0_SPI_BUSY, TCC_EA_WRREQ_STALL_sum, TCC_EA_WRREQ_IO_CREDIT_STALL_sum, TCC_EA_WRREQ_GMI_CREDIT_STALL_sum, TCC_EA_WRREQ_DRAM_CREDIT_STALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170804_1298276/input0_results_240321_170804
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_5.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_6.txt
|-> [rocprof] RPL: on '240321_170805' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_6.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170805_1298477'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170805_1298477/input0_results_240321_170805'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170805_1298477/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_ACTIVE_INST_MISC, SQ_ACTIVE_INST_FLAT, SQ_INST_CYCLES_VMEM_WR, SQ_INST_CYCLES_VMEM_RD, SQ_INST_CYCLES_SMEM, SQ_INST_CYCLES_SALU, SQ_THREAD_CYCLES_VALU, SQ_IFETCH, TCP_TCC_WRITE_REQ_sum, TCP_TCC_ATOMIC_WITH_RET_REQ_sum, TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum, TCP_TCC_NC_READ_REQ_sum, TA_FLAT_WAVEFRONTS_sum, TA_FLAT_READ_WAVEFRONTS_sum, SPI_RA_BAR_CU_FULL_CSN, SPI_RA_TGLIM_CU_FULL_CSN, TCC_EA_RDREQ_sum, TCC_EA_RDREQ_32B_sum, TCC_EA_RD_UNCACHED_32B_sum, TCC_EA_RDREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170805_1298477/input0_results_240321_170805
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_6.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_7.txt
|-> [rocprof] RPL: on '240321_170805' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_7.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170805_1298679'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170805_1298679/input0_results_240321_170805'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170805_1298679/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_LDS_BANK_CONFLICT, SQ_LDS_ADDR_CONFLICT, SQ_LDS_UNALIGNED_STALL, SQ_WAVES_EQ_64, SQ_WAVES_LT_64, SQ_WAVES_LT_48, SQ_WAVES_LT_32, SQ_WAVES_LT_16, TCP_TCC_NC_WRITE_REQ_sum, TCP_TCC_NC_ATOMIC_REQ_sum, TCP_TCC_UC_READ_REQ_sum, TCP_TCC_UC_WRITE_REQ_sum, TA_FLAT_WRITE_WAVEFRONTS_sum, TA_FLAT_ATOMIC_WAVEFRONTS_sum, SPI_RA_WVLIM_STALL_CSN, SPI_SWC_CSC_WR, TCC_EA_RDREQ_IO_CREDIT_STALL_sum, TCC_EA_RDREQ_GMI_CREDIT_STALL_sum, TCC_EA_RDREQ_DRAM_CREDIT_STALL_sum, TCC_TAG_STALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170805_1298679/input0_results_240321_170805
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_7.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_8.txt
|-> [rocprof] RPL: on '240321_170806' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_8.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170806_1298868'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170806_1298868/input0_results_240321_170806'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170806_1298868/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 17 metrics
|-> [rocprof] SQ_ITEMS, SQ_LDS_MEM_VIOLATIONS, SQ_LDS_ATOMIC_RETURN, SQ_LDS_IDX_ACTIVE, SQ_WAVES_RESTORED, SQ_WAVES_SAVED, SQ_INSTS_SMEM_NORM, TCP_TCC_UC_ATOMIC_REQ_sum, TCP_TCC_CC_READ_REQ_sum, TCP_TCC_CC_WRITE_REQ_sum, TCP_TCC_CC_ATOMIC_REQ_sum, SPI_VWC_CSC_WR, SPI_RA_BULKY_CU_FULL_CSN, TCC_NORMAL_WRITEBACK_sum, TCC_ALL_TC_OP_WB_WRITEBACK_sum, TCC_NORMAL_EVICT_sum, TCC_ALL_TC_OP_INV_EVICT_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170806_1298868/input0_results_240321_170806
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_8.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_9.txt
|-> [rocprof] RPL: on '240321_170806' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_9.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170806_1299070'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170806_1299070/input0_results_240321_170806'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170806_1299070/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] TCP_TCC_RW_READ_REQ_sum, TCP_TCC_RW_WRITE_REQ_sum, TCP_TCC_RW_ATOMIC_REQ_sum, TCP_PENDING_STALL_CYCLES_sum, TCC_TOO_MANY_EA_WRREQS_STALL_sum, TCC_EA_ATOMIC_sum, TCC_EA_RDREQ_LEVEL_sum, TCC_EA_WRREQ_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170806_1299070/input0_results_240321_170806
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_9.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/timestamps.txt
|-> [rocprof] RPL: on '240321_170807' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/timestamps.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170807_1299273'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170807_1299273/input0_results_240321_170807'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170807_1299273/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 0 metrics
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170807_1299273/input0_results_240321_170807
|-> [rocprof] File 'tests/workloads/kernel/MI100/timestamps.csv' is generating
|-> [rocprof]
@@ -2,4 +2,4 @@ pmc: GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PRE
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVE
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_CYCLES SQ_BUSY_CYCLES SQ_BUSY_CU_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQC_TC_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQC_TC_DATA_ATOMIC_REQ SQC_TC_STALL SQC_TC_REQ SQC_DCACHE_REQ_READ_16 SQC_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_EA_ATOMIC_LEVEL_sum
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_ATOMIC[0] TCC_CYCLE[0] TCC_EA_ATOMIC[0] TCC_EA_ATOMIC_LEVEL[0] TCC_ATO
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_EA_RDREQ[0] TCC_EA_RDREQ_32B[0] TCC_EA_RDREQ_DRAM_CREDIT_STALL[0] TCC_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_EA_RDREQ_IO_CREDIT_STALL[0] TCC_EA_RDREQ_LEVEL[0] TCC_EA_WRREQ[0] TCC_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_EA_WRREQ_DRAM_CREDIT_STALL[0] TCC_EA_WRREQ_GMI_CREDIT_STALL[0] TCC_EA_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_HIT[0] TCC_MISS[0] TCC_READ[0] TCC_REQ[0] TCC_HIT[1] TCC_MISS[1] TCC_R
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_RW_REQ[0] TCC_TOO_MANY_EA_WRREQS_STALL[0] TCC_WRITE[0] TCC_RW_REQ[1] T
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQC_DCACHE_INPUT_VALID_READYB SQC_DCACHE_ATOMIC SQC_DCACHE_REQ_READ_8 SQC_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQC_DCACHE_REQ_READ_2 SQC_DCACHE_REQ_READ_4 SQ_INSTS_VMEM_WR SQ_INSTS_VMEM
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_FLAT SQ_INSTS_LDS SQ_INSTS_GDS SQ_INSTS_EXP_GDS SQ_INSTS_BRANCH S
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_WAIT_INST_ANY SQ_ACTIVE_INST_ANY SQ_INSTS_VALU SQ_ACTIVE_INST_VMEM SQ_A
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_ACTIVE_INST_MISC SQ_ACTIVE_INST_FLAT SQ_INST_CYCLES_VMEM_WR SQ_INST_CYC
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_LDS_BANK_CONFLICT SQ_LDS_ADDR_CONFLICT SQ_LDS_UNALIGNED_STALL SQ_WAVES_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_ITEMS SQ_LDS_MEM_VIOLATIONS SQ_LDS_ATOMIC_RETURN SQ_LDS_IDX_ACTIVE SQ_W
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCP_TCC_RW_READ_REQ_sum TCP_TCC_RW_WRITE_REQ_sum TCP_TCC_RW_ATOMIC_REQ_sum
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc:
gpu:
range:
kernel:
kernel: vecCopy
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1 Dispatch_ID Kernel_Name GPU_ID
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
File diff suppressed because one or more lines are too long
@@ -0,0 +1,2 @@
workload_name,command,ip_blocks,timestamp,version,hostname,cpu_model,sbios,linux_distro,linux_kernel_version,amd_gpu_kernel_version,cpu_memory,gpu_memory,rocm_version,vbios,compute_partition,memory_partition,gpu_model,gpu_arch,gpu_l1,gpu_l2,cu_per_gpu,simd_per_cu,se_per_gpu,wave_size,workgroup_max_size,max_waves_per_cu,max_sclk,max_mclk,cur_sclk,cur_mclk,total_l2_chan,lds_banks_per_cu,sqc_per_gpu,pipes_per_gpu,hbm_bw,num_xcd
kernel,./tests/vcopy -n 1048576 -b 256 -i 3,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF,Thu 21 Mar 2024 05:07:54 PM (CDT),2,t007-001.hpcfund,AMD EPYC 7V13 64-Core Processor,American Megatrends Inc.0602,Rocky Linux 9.1 (Blue Onyx),5.14.0-162.18.1.el9_1.x86_64,,527651008,,6.0.2-115,113-D3431401-100,NA,NA,MI100,gfx908,16,8192,120,4,8,64,1024,40,1502,1200,1502,1200,32,32,64,4,1228.8,1
1 workload_name command ip_blocks timestamp version hostname cpu_model sbios linux_distro linux_kernel_version amd_gpu_kernel_version cpu_memory gpu_memory rocm_version vbios compute_partition memory_partition gpu_model gpu_arch gpu_l1 gpu_l2 cu_per_gpu simd_per_cu se_per_gpu wave_size workgroup_max_size max_waves_per_cu max_sclk max_mclk cur_sclk cur_mclk total_l2_chan lds_banks_per_cu sqc_per_gpu pipes_per_gpu hbm_bw num_xcd
2 kernel ./tests/vcopy -n 1048576 -b 256 -i 3 SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF Thu 21 Mar 2024 05:07:54 PM (CDT) 2 t007-001.hpcfund AMD EPYC 7V13 64-Core Processor American Megatrends Inc.0602 Rocky Linux 9.1 (Blue Onyx) 5.14.0-162.18.1.el9_1.x86_64 527651008 6.0.2-115 113-D3431401-100 NA NA MI100 gfx908 16 8192 120 4 8 64 1024 40 1502 1200 1502 1200 32 32 64 4 1228.8 1
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1299433,1299433,1048576,256,0,0,8,8,16,64,0x0,0x7fc619c60ec0,1414621055400032,1414621055425142,1414621055449622,1414621055460846
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1299433,1299433,1048576,256,0,0,8,8,16,64,0x0,0x7fc619c60ec0,1414621055462319,1414621055537142,1414621055556022,1414621055557449
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1299433,1299433,1048576,256,0,0,8,8,16,64,0x0,0x7fc619c60ec0,1414621055567848,1414621055576182,1414621055595542,1414621055596913
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1299433 1299433 1048576 256 0 0 8 8 16 64 0x0 0x7fc619c60ec0 1414621055400032 1414621055425142 1414621055449622 1414621055460846
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1299433 1299433 1048576 256 0 0 8 8 16 64 0x0 0x7fc619c60ec0 1414621055462319 1414621055537142 1414621055556022 1414621055557449
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1299433 1299433 1048576 256 0 0 8 8 16 64 0x0 0x7fc619c60ec0 1414621055567848 1414621055576182 1414621055595542 1414621055596913
@@ -1 +1,4 @@
Index,KernelName
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4190428,4190428,1048576,256,0,0,8,0,16,64,0x0,0x7f94512c0ec0,27378,27378,16384,65536,13161,1467220,1414432756404789,1414445656659785,1414445656680105,1414432772257328
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4190428,4190428,1048576,256,0,0,8,0,16,64,0x0,0x7f94512c0ec0,40627,40627,16384,65536,9359,1048624,1414432772278037,1414445656699625,1414445656715145,1414432772686309
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4190428,4190428,1048576,256,0,0,8,0,16,64,0x0,0x7f94512c0ec0,40747,40747,16384,65536,9106,1048600,1414432772715755,1414445656774345,1414445656790665,1414432772875517
1 Index Dispatch_ID KernelName Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4190428 4190428 1048576 256 0 0 8 0 16 64 0x0 0x7f94512c0ec0 27378 27378 16384 65536 13161 1467220 1414432756404789 1414445656659785 1414445656680105 1414432772257328
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4190428 4190428 1048576 256 0 0 8 0 16 64 0x0 0x7f94512c0ec0 40627 40627 16384 65536 9359 1048624 1414432772278037 1414445656699625 1414445656715145 1414432772686309
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4190428 4190428 1048576 256 0 0 8 0 16 64 0x0 0x7f94512c0ec0 40747 40747 16384 65536 9106 1048600 1414432772715755 1414445656774345 1414445656790665 1414432772875517
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4190642,4190642,1048576,256,0,0,8,0,16,64,0x0,0x7f6c840ecec0,0,0,0,1414433244984736,1414445656659785,1414445656680105,1414433261546075
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4190642,4190642,1048576,256,0,0,8,0,16,64,0x0,0x7f6c840ecec0,0,0,0,1414433261564100,1414445656699625,1414445656715145,1414433261863666
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4190642,4190642,1048576,256,0,0,8,0,16,64,0x0,0x7f6c840ecec0,0,0,0,1414433261890016,1414445656774345,1414445656790665,1414433262043226
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4190642 4190642 1048576 256 0 0 8 0 16 64 0x0 0x7f6c840ecec0 0 0 0 1414433244984736 1414445656659785 1414445656680105 1414433261546075
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4190642 4190642 1048576 256 0 0 8 0 16 64 0x0 0x7f6c840ecec0 0 0 0 1414433261564100 1414445656699625 1414445656715145 1414433261863666
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4190642 4190642 1048576 256 0 0 8 0 16 64 0x0 0x7f6c840ecec0 0 0 0 1414433261890016 1414445656774345 1414445656790665 1414433262043226
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4190847,4190847,1048576,256,0,0,8,0,16,64,0x0,0x7f19e81d0ec0,65536,187730,20996312,1414433732354835,1414445656659785,1414445656680105,1414433748542708
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4190847,4190847,1048576,256,0,0,8,0,16,64,0x0,0x7f19e81d0ec0,65536,281018,31378144,1414433748560412,1414445656699625,1414445656715145,1414433748861932
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4190847,4190847,1048576,256,0,0,8,0,16,64,0x0,0x7f19e81d0ec0,65536,258024,28932200,1414433748888122,1414445656774345,1414445656790665,1414433749041842
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4190847 4190847 1048576 256 0 0 8 0 16 64 0x0 0x7f19e81d0ec0 65536 187730 20996312 1414433732354835 1414445656659785 1414445656680105 1414433748542708
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4190847 4190847 1048576 256 0 0 8 0 16 64 0x0 0x7f19e81d0ec0 65536 281018 31378144 1414433748560412 1414445656699625 1414445656715145 1414433748861932
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4190847 4190847 1048576 256 0 0 8 0 16 64 0x0 0x7f19e81d0ec0 65536 258024 28932200 1414433748888122 1414445656774345 1414445656790665 1414433749041842
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4191050,4191050,1048576,256,0,0,8,0,16,64,0x0,0x7fcf14168ec0,32768,306417,34321340,1414434204510803,1414445656659785,1414445656680105,1414434220573068
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4191050,4191050,1048576,256,0,0,8,0,16,64,0x0,0x7fcf14168ec0,32768,580620,65031856,1414434220596282,1414445656699625,1414445656715145,1414434220884157
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4191050,4191050,1048576,256,0,0,8,0,16,64,0x0,0x7fcf14168ec0,32768,586983,65717864,1414434220910687,1414445656774345,1414445656790665,1414434221039700
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4191050 4191050 1048576 256 0 0 8 0 16 64 0x0 0x7fcf14168ec0 32768 306417 34321340 1414434204510803 1414445656659785 1414445656680105 1414434220573068
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4191050 4191050 1048576 256 0 0 8 0 16 64 0x0 0x7fcf14168ec0 32768 580620 65031856 1414434220596282 1414445656699625 1414445656715145 1414434220884157
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4191050 4191050 1048576 256 0 0 8 0 16 64 0x0 0x7fcf14168ec0 32768 586983 65717864 1414434220910687 1414445656774345 1414445656790665 1414434221039700
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4191237,4191237,1048576,256,0,0,8,0,16,64,0x0,0x7ffbe6b9cec0,27648,27648,11103,221192,16384,10666434,128024,0,43180252,1414434690769320,1414445656659785,1414445656680105,1414434706762735
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4191237,4191237,1048576,256,0,0,8,0,16,64,0x0,0x7ffbe6b9cec0,41673,41673,14419,333392,16384,19327058,223251,0,77801560,1414434706790227,1414445656699625,1414445656715145,1414434707222514
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4191237,4191237,1048576,256,0,0,8,0,16,64,0x0,0x7ffbe6b9cec0,41027,41027,13871,328224,16384,18889841,217781,0,76050388,1414434707257661,1414445656774345,1414445656790665,1414434707434395
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4191237 4191237 1048576 256 0 0 8 0 16 64 0x0 0x7ffbe6b9cec0 27648 27648 11103 221192 16384 10666434 128024 0 43180252 1414434690769320 1414445656659785 1414445656680105 1414434706762735
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4191237 4191237 1048576 256 0 0 8 0 16 64 0x0 0x7ffbe6b9cec0 41673 41673 14419 333392 16384 19327058 223251 0 77801560 1414434706790227 1414445656699625 1414445656715145 1414434707222514
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4191237 4191237 1048576 256 0 0 8 0 16 64 0x0 0x7ffbe6b9cec0 41027 41027 13871 328224 16384 18889841 217781 0 76050388 1414434707257661 1414445656774345 1414445656790665 1414434707434395
+764
Wyświetl plik
@@ -0,0 +1,764 @@
Omniperf version: 2.0.0-RC1
Profiler choice: rocprofv1
Path: /home1/josantos/omniperf/tests/workloads/kernel/MI200
Target: MI200
Command: ./tests/vcopy -n 1048576 -b 256 -i 3
Kernel Selection: ['vecCopy']
Dispatch Selection: None
IP Blocks: All
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Collecting Performance Counters
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_IFETCH_LEVEL.txt
|-> [rocprof] RPL: on '240321_170500' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_IFETCH_LEVEL.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170500_4190268'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170500_4190268/input0_results_240321_170500'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170500_4190268/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 6 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, SQ_WAVES, SQ_IFETCH, SQ_IFETCH_LEVEL, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170500_4190268/input0_results_240321_170500
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_IFETCH_LEVEL.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_LDS.txt
|-> [rocprof] RPL: on '240321_170501' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_LDS.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170501_4190482'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170501_4190482/input0_results_240321_170501'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170501_4190482/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_LDS, SQ_INST_LEVEL_LDS, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170501_4190482/input0_results_240321_170501
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_INST_LEVEL_LDS.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_SMEM.txt
|-> [rocprof] RPL: on '240321_170501' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_SMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170501_4190684'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170501_4190684/input0_results_240321_170501'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170501_4190684/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_SMEM, SQ_INST_LEVEL_SMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170501_4190684/input0_results_240321_170501
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_INST_LEVEL_SMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_VMEM.txt
|-> [rocprof] RPL: on '240321_170502' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_VMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170502_4190890'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170502_4190890/input0_results_240321_170502'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170502_4190890/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_VMEM, SQ_INST_LEVEL_VMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170502_4190890/input0_results_240321_170502
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_INST_LEVEL_VMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_LEVEL_WAVES.txt
|-> [rocprof] RPL: on '240321_170502' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_LEVEL_WAVES.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170502_4191077'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170502_4191077/input0_results_240321_170502'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170502_4191077/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 9 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, CPC_ME1_BUSY_FOR_PACKET_DECODE, SQ_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQ_BUSY_CYCLES, SQ_LEVEL_WAVES, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170502_4191077/input0_results_240321_170502
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_LEVEL_WAVES.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_0.txt
|-> [rocprof] RPL: on '240321_170502' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_0.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170502_4191279'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170502_4191279/input0_results_240321_170502'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170502_4191279/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 28 metrics
|-> [rocprof] SQ_CYCLES, SQ_BUSY_CYCLES, SQ_WAVES, SQ_INSTS_VALU_CVT, SQ_INSTS_VMEM_WR, SQ_INSTS_VMEM_RD, SQ_INSTS_VMEM, SQ_INSTS_SALU, GRBM_COUNT, GRBM_GUI_ACTIVE, TCP_GATE_EN1_sum, TCP_GATE_EN2_sum, TCP_TD_TCP_STALL_CYCLES_sum, TCP_TCR_TCP_STALL_CYCLES_sum, TA_TA_BUSY_sum, TA_BUFFER_WAVEFRONTS_sum, TD_TD_BUSY_sum, TD_TC_STALL_sum, SPI_CSN_WINDOW_VALID, SPI_CSN_BUSY, CPC_CPC_STAT_BUSY, CPC_CPC_STAT_IDLE, CPF_CPF_STAT_BUSY, CPF_CPF_STAT_STALL, TCC_CYCLE_sum, TCC_BUSY_sum, TCC_PROBE_sum, TCC_PROBE_ALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170502_4191279/input0_results_240321_170502
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_0.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_1.txt
|-> [rocprof] RPL: on '240321_170503' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_1.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170503_4191481'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170503_4191481/input0_results_240321_170503'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170503_4191481/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 27 metrics
|-> [rocprof] SQ_INSTS_VSKIPPED, SQ_INSTS, SQ_INSTS_VALU, SQ_INSTS_VALU_ADD_F16, SQ_INSTS_VALU_MUL_F16, SQ_INSTS_VALU_FMA_F16, SQ_INSTS_VALU_TRANS_F16, SQ_INSTS_VALU_ADD_F32, GRBM_SPI_BUSY, TCP_READ_TAGCONFLICT_STALL_CYCLES_sum, TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum, TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum, TCP_TA_TCP_STATE_READ_sum, TA_BUFFER_READ_WAVEFRONTS_sum, TA_BUFFER_WRITE_WAVEFRONTS_sum, TD_SPI_STALL_sum, TD_LOAD_WAVEFRONT_sum, SPI_CSN_NUM_THREADGROUPS, SPI_CSN_WAVE, CPC_CPC_TCIU_BUSY, CPC_CPC_TCIU_IDLE, CPF_CPF_TCIU_BUSY, CPF_CPF_TCIU_STALL, TCC_NC_REQ_sum, TCC_UC_REQ_sum, TCC_CC_REQ_sum, TCC_RW_REQ_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170503_4191481/input0_results_240321_170503
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_1.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_10.txt
|-> [rocprof] RPL: on '240321_170503' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_10.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170503_4191685'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170503_4191685/input0_results_240321_170503'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170503_4191685/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] SQC_TC_DATA_WRITE_REQ, SQC_TC_DATA_ATOMIC_REQ, SQC_TC_STALL, SQC_TC_REQ, SQC_DCACHE_REQ_READ_16, SQC_ICACHE_REQ, SQC_ICACHE_HITS, SQC_ICACHE_MISSES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170503_4191685/input0_results_240321_170503
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_10.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_11.txt
|-> [rocprof] RPL: on '240321_170504' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_11.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170504_4191871'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170504_4191871/input0_results_240321_170504'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170504_4191871/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] SQC_ICACHE_MISSES_DUPLICATE, SQC_DCACHE_INPUT_VALID_READYB, SQC_DCACHE_ATOMIC, SQC_DCACHE_REQ_READ_8, SQC_DCACHE_REQ, SQC_DCACHE_HITS, SQC_DCACHE_MISSES, SQC_DCACHE_MISSES_DUPLICATE
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170504_4191871/input0_results_240321_170504
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_11.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_12.txt
|-> [rocprof] RPL: on '240321_170504' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_12.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170504_4192073'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170504_4192073/input0_results_240321_170504'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170504_4192073/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQC_DCACHE_REQ_READ_1, SQC_DCACHE_REQ_READ_2, SQC_DCACHE_REQ_READ_4
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170504_4192073/input0_results_240321_170504
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_12.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_13.txt
|-> [rocprof] RPL: on '240321_170505' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_13.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170505_4192275'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170505_4192275/input0_results_240321_170505'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170505_4192275/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_ATOMIC[0], TCC_CYCLE[0], TCC_EA_ATOMIC[0], TCC_EA_ATOMIC_LEVEL[0], TCC_ATOMIC[1], TCC_CYCLE[1], TCC_EA_ATOMIC[1], TCC_EA_ATOMIC_LEVEL[1], TCC_ATOMIC[2], TCC_CYCLE[2], TCC_EA_ATOMIC[2], TCC_EA_ATOMIC_LEVEL[2], TCC_ATOMIC[3], TCC_CYCLE[3], TCC_EA_ATOMIC[3], TCC_EA_ATOMIC_LEVEL[3], TCC_ATOMIC[4], TCC_CYCLE[4], TCC_EA_ATOMIC[4], TCC_EA_ATOMIC_LEVEL[4], TCC_ATOMIC[5], TCC_CYCLE[5], TCC_EA_ATOMIC[5], TCC_EA_ATOMIC_LEVEL[5], TCC_ATOMIC[6], TCC_CYCLE[6], TCC_EA_ATOMIC[6], TCC_EA_ATOMIC_LEVEL[6], TCC_ATOMIC[7], TCC_CYCLE[7], TCC_EA_ATOMIC[7], TCC_EA_ATOMIC_LEVEL[7], TCC_ATOMIC[8], TCC_CYCLE[8], TCC_EA_ATOMIC[8], TCC_EA_ATOMIC_LEVEL[8], TCC_ATOMIC[9], TCC_CYCLE[9], TCC_EA_ATOMIC[9], TCC_EA_ATOMIC_LEVEL[9], TCC_ATOMIC[10], TCC_CYCLE[10], TCC_EA_ATOMIC[10], TCC_EA_ATOMIC_LEVEL[10], TCC_ATOMIC[11], TCC_CYCLE[11], TCC_EA_ATOMIC[11], TCC_EA_ATOMIC_LEVEL[11], TCC_ATOMIC[12], TCC_CYCLE[12], TCC_EA_ATOMIC[12], TCC_EA_ATOMIC_LEVEL[12], TCC_ATOMIC[13], TCC_CYCLE[13], TCC_EA_ATOMIC[13], TCC_EA_ATOMIC_LEVEL[13], TCC_ATOMIC[14], TCC_CYCLE[14], TCC_EA_ATOMIC[14], TCC_EA_ATOMIC_LEVEL[14], TCC_ATOMIC[15], TCC_CYCLE[15], TCC_EA_ATOMIC[15], TCC_EA_ATOMIC_LEVEL[15], TCC_ATOMIC[16], TCC_CYCLE[16], TCC_EA_ATOMIC[16], TCC_EA_ATOMIC_LEVEL[16], TCC_ATOMIC[17], TCC_CYCLE[17], TCC_EA_ATOMIC[17], TCC_EA_ATOMIC_LEVEL[17], TCC_ATOMIC[18], TCC_CYCLE[18], TCC_EA_ATOMIC[18], TCC_EA_ATOMIC_LEVEL[18], TCC_ATOMIC[19], TCC_CYCLE[19], TCC_EA_ATOMIC[19], TCC_EA_ATOMIC_LEVEL[19], TCC_ATOMIC[20], TCC_CYCLE[20], TCC_EA_ATOMIC[20], TCC_EA_ATOMIC_LEVEL[20], TCC_ATOMIC[21], TCC_CYCLE[21], TCC_EA_ATOMIC[21], TCC_EA_ATOMIC_LEVEL[21], TCC_ATOMIC[22], TCC_CYCLE[22], TCC_EA_ATOMIC[22], TCC_EA_ATOMIC_LEVEL[22], TCC_ATOMIC[23], TCC_CYCLE[23], TCC_EA_ATOMIC[23], TCC_EA_ATOMIC_LEVEL[23], TCC_ATOMIC[24], TCC_CYCLE[24], TCC_EA_ATOMIC[24], TCC_EA_ATOMIC_LEVEL[24], TCC_ATOMIC[25], TCC_CYCLE[25], TCC_EA_ATOMIC[25], TCC_EA_ATOMIC_LEVEL[25], TCC_ATOMIC[26], TCC_CYCLE[26], TCC_EA_ATOMIC[26], TCC_EA_ATOMIC_LEVEL[26], TCC_ATOMIC[27], TCC_CYCLE[27], TCC_EA_ATOMIC[27], TCC_EA_ATOMIC_LEVEL[27], TCC_ATOMIC[28], TCC_CYCLE[28], TCC_EA_ATOMIC[28], TCC_EA_ATOMIC_LEVEL[28], TCC_ATOMIC[29], TCC_CYCLE[29], TCC_EA_ATOMIC[29], TCC_EA_ATOMIC_LEVEL[29], TCC_ATOMIC[30], TCC_CYCLE[30], TCC_EA_ATOMIC[30], TCC_EA_ATOMIC_LEVEL[30], TCC_ATOMIC[31], TCC_CYCLE[31], TCC_EA_ATOMIC[31], TCC_EA_ATOMIC_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170505_4192275/input0_results_240321_170505
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_13.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_14.txt
|-> [rocprof] RPL: on '240321_170506' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_14.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170506_4192465'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170506_4192465/input0_results_240321_170506'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170506_4192465/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ[0], TCC_EA_RDREQ_32B[0], TCC_EA_RDREQ_DRAM_CREDIT_STALL[0], TCC_EA_RDREQ_GMI_CREDIT_STALL[0], TCC_EA_RDREQ[1], TCC_EA_RDREQ_32B[1], TCC_EA_RDREQ_DRAM_CREDIT_STALL[1], TCC_EA_RDREQ_GMI_CREDIT_STALL[1], TCC_EA_RDREQ[2], TCC_EA_RDREQ_32B[2], TCC_EA_RDREQ_DRAM_CREDIT_STALL[2], TCC_EA_RDREQ_GMI_CREDIT_STALL[2], TCC_EA_RDREQ[3], TCC_EA_RDREQ_32B[3], TCC_EA_RDREQ_DRAM_CREDIT_STALL[3], TCC_EA_RDREQ_GMI_CREDIT_STALL[3], TCC_EA_RDREQ[4], TCC_EA_RDREQ_32B[4], TCC_EA_RDREQ_DRAM_CREDIT_STALL[4], TCC_EA_RDREQ_GMI_CREDIT_STALL[4], TCC_EA_RDREQ[5], TCC_EA_RDREQ_32B[5], TCC_EA_RDREQ_DRAM_CREDIT_STALL[5], TCC_EA_RDREQ_GMI_CREDIT_STALL[5], TCC_EA_RDREQ[6], TCC_EA_RDREQ_32B[6], TCC_EA_RDREQ_DRAM_CREDIT_STALL[6], TCC_EA_RDREQ_GMI_CREDIT_STALL[6], TCC_EA_RDREQ[7], TCC_EA_RDREQ_32B[7], TCC_EA_RDREQ_DRAM_CREDIT_STALL[7], TCC_EA_RDREQ_GMI_CREDIT_STALL[7], TCC_EA_RDREQ[8], TCC_EA_RDREQ_32B[8], TCC_EA_RDREQ_DRAM_CREDIT_STALL[8], TCC_EA_RDREQ_GMI_CREDIT_STALL[8], TCC_EA_RDREQ[9], TCC_EA_RDREQ_32B[9], TCC_EA_RDREQ_DRAM_CREDIT_STALL[9], TCC_EA_RDREQ_GMI_CREDIT_STALL[9], TCC_EA_RDREQ[10], TCC_EA_RDREQ_32B[10], TCC_EA_RDREQ_DRAM_CREDIT_STALL[10], TCC_EA_RDREQ_GMI_CREDIT_STALL[10], TCC_EA_RDREQ[11], TCC_EA_RDREQ_32B[11], TCC_EA_RDREQ_DRAM_CREDIT_STALL[11], TCC_EA_RDREQ_GMI_CREDIT_STALL[11], TCC_EA_RDREQ[12], TCC_EA_RDREQ_32B[12], TCC_EA_RDREQ_DRAM_CREDIT_STALL[12], TCC_EA_RDREQ_GMI_CREDIT_STALL[12], TCC_EA_RDREQ[13], TCC_EA_RDREQ_32B[13], TCC_EA_RDREQ_DRAM_CREDIT_STALL[13], TCC_EA_RDREQ_GMI_CREDIT_STALL[13], TCC_EA_RDREQ[14], TCC_EA_RDREQ_32B[14], TCC_EA_RDREQ_DRAM_CREDIT_STALL[14], TCC_EA_RDREQ_GMI_CREDIT_STALL[14], TCC_EA_RDREQ[15], TCC_EA_RDREQ_32B[15], TCC_EA_RDREQ_DRAM_CREDIT_STALL[15], TCC_EA_RDREQ_GMI_CREDIT_STALL[15], TCC_EA_RDREQ[16], TCC_EA_RDREQ_32B[16], TCC_EA_RDREQ_DRAM_CREDIT_STALL[16], TCC_EA_RDREQ_GMI_CREDIT_STALL[16], TCC_EA_RDREQ[17], TCC_EA_RDREQ_32B[17], TCC_EA_RDREQ_DRAM_CREDIT_STALL[17], TCC_EA_RDREQ_GMI_CREDIT_STALL[17], TCC_EA_RDREQ[18], TCC_EA_RDREQ_32B[18], TCC_EA_RDREQ_DRAM_CREDIT_STALL[18], TCC_EA_RDREQ_GMI_CREDIT_STALL[18], TCC_EA_RDREQ[19], TCC_EA_RDREQ_32B[19], TCC_EA_RDREQ_DRAM_CREDIT_STALL[19], TCC_EA_RDREQ_GMI_CREDIT_STALL[19], TCC_EA_RDREQ[20], TCC_EA_RDREQ_32B[20], TCC_EA_RDREQ_DRAM_CREDIT_STALL[20], TCC_EA_RDREQ_GMI_CREDIT_STALL[20], TCC_EA_RDREQ[21], TCC_EA_RDREQ_32B[21], TCC_EA_RDREQ_DRAM_CREDIT_STALL[21], TCC_EA_RDREQ_GMI_CREDIT_STALL[21], TCC_EA_RDREQ[22], TCC_EA_RDREQ_32B[22], TCC_EA_RDREQ_DRAM_CREDIT_STALL[22], TCC_EA_RDREQ_GMI_CREDIT_STALL[22], TCC_EA_RDREQ[23], TCC_EA_RDREQ_32B[23], TCC_EA_RDREQ_DRAM_CREDIT_STALL[23], TCC_EA_RDREQ_GMI_CREDIT_STALL[23], TCC_EA_RDREQ[24], TCC_EA_RDREQ_32B[24], TCC_EA_RDREQ_DRAM_CREDIT_STALL[24], TCC_EA_RDREQ_GMI_CREDIT_STALL[24], TCC_EA_RDREQ[25], TCC_EA_RDREQ_32B[25], TCC_EA_RDREQ_DRAM_CREDIT_STALL[25], TCC_EA_RDREQ_GMI_CREDIT_STALL[25], TCC_EA_RDREQ[26], TCC_EA_RDREQ_32B[26], TCC_EA_RDREQ_DRAM_CREDIT_STALL[26], TCC_EA_RDREQ_GMI_CREDIT_STALL[26], TCC_EA_RDREQ[27], TCC_EA_RDREQ_32B[27], TCC_EA_RDREQ_DRAM_CREDIT_STALL[27], TCC_EA_RDREQ_GMI_CREDIT_STALL[27], TCC_EA_RDREQ[28], TCC_EA_RDREQ_32B[28], TCC_EA_RDREQ_DRAM_CREDIT_STALL[28], TCC_EA_RDREQ_GMI_CREDIT_STALL[28], TCC_EA_RDREQ[29], TCC_EA_RDREQ_32B[29], TCC_EA_RDREQ_DRAM_CREDIT_STALL[29], TCC_EA_RDREQ_GMI_CREDIT_STALL[29], TCC_EA_RDREQ[30], TCC_EA_RDREQ_32B[30], TCC_EA_RDREQ_DRAM_CREDIT_STALL[30], TCC_EA_RDREQ_GMI_CREDIT_STALL[30], TCC_EA_RDREQ[31], TCC_EA_RDREQ_32B[31], TCC_EA_RDREQ_DRAM_CREDIT_STALL[31], TCC_EA_RDREQ_GMI_CREDIT_STALL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170506_4192465/input0_results_240321_170506
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_14.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_15.txt
|-> [rocprof] RPL: on '240321_170506' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_15.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170506_4192664'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170506_4192664/input0_results_240321_170506'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170506_4192664/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ_IO_CREDIT_STALL[0], TCC_EA_RDREQ_LEVEL[0], TCC_EA_WRREQ[0], TCC_EA_WRREQ_64B[0], TCC_EA_RDREQ_IO_CREDIT_STALL[1], TCC_EA_RDREQ_LEVEL[1], TCC_EA_WRREQ[1], TCC_EA_WRREQ_64B[1], TCC_EA_RDREQ_IO_CREDIT_STALL[2], TCC_EA_RDREQ_LEVEL[2], TCC_EA_WRREQ[2], TCC_EA_WRREQ_64B[2], TCC_EA_RDREQ_IO_CREDIT_STALL[3], TCC_EA_RDREQ_LEVEL[3], TCC_EA_WRREQ[3], TCC_EA_WRREQ_64B[3], TCC_EA_RDREQ_IO_CREDIT_STALL[4], TCC_EA_RDREQ_LEVEL[4], TCC_EA_WRREQ[4], TCC_EA_WRREQ_64B[4], TCC_EA_RDREQ_IO_CREDIT_STALL[5], TCC_EA_RDREQ_LEVEL[5], TCC_EA_WRREQ[5], TCC_EA_WRREQ_64B[5], TCC_EA_RDREQ_IO_CREDIT_STALL[6], TCC_EA_RDREQ_LEVEL[6], TCC_EA_WRREQ[6], TCC_EA_WRREQ_64B[6], TCC_EA_RDREQ_IO_CREDIT_STALL[7], TCC_EA_RDREQ_LEVEL[7], TCC_EA_WRREQ[7], TCC_EA_WRREQ_64B[7], TCC_EA_RDREQ_IO_CREDIT_STALL[8], TCC_EA_RDREQ_LEVEL[8], TCC_EA_WRREQ[8], TCC_EA_WRREQ_64B[8], TCC_EA_RDREQ_IO_CREDIT_STALL[9], TCC_EA_RDREQ_LEVEL[9], TCC_EA_WRREQ[9], TCC_EA_WRREQ_64B[9], TCC_EA_RDREQ_IO_CREDIT_STALL[10], TCC_EA_RDREQ_LEVEL[10], TCC_EA_WRREQ[10], TCC_EA_WRREQ_64B[10], TCC_EA_RDREQ_IO_CREDIT_STALL[11], TCC_EA_RDREQ_LEVEL[11], TCC_EA_WRREQ[11], TCC_EA_WRREQ_64B[11], TCC_EA_RDREQ_IO_CREDIT_STALL[12], TCC_EA_RDREQ_LEVEL[12], TCC_EA_WRREQ[12], TCC_EA_WRREQ_64B[12], TCC_EA_RDREQ_IO_CREDIT_STALL[13], TCC_EA_RDREQ_LEVEL[13], TCC_EA_WRREQ[13], TCC_EA_WRREQ_64B[13], TCC_EA_RDREQ_IO_CREDIT_STALL[14], TCC_EA_RDREQ_LEVEL[14], TCC_EA_WRREQ[14], TCC_EA_WRREQ_64B[14], TCC_EA_RDREQ_IO_CREDIT_STALL[15], TCC_EA_RDREQ_LEVEL[15], TCC_EA_WRREQ[15], TCC_EA_WRREQ_64B[15], TCC_EA_RDREQ_IO_CREDIT_STALL[16], TCC_EA_RDREQ_LEVEL[16], TCC_EA_WRREQ[16], TCC_EA_WRREQ_64B[16], TCC_EA_RDREQ_IO_CREDIT_STALL[17], TCC_EA_RDREQ_LEVEL[17], TCC_EA_WRREQ[17], TCC_EA_WRREQ_64B[17], TCC_EA_RDREQ_IO_CREDIT_STALL[18], TCC_EA_RDREQ_LEVEL[18], TCC_EA_WRREQ[18], TCC_EA_WRREQ_64B[18], TCC_EA_RDREQ_IO_CREDIT_STALL[19], TCC_EA_RDREQ_LEVEL[19], TCC_EA_WRREQ[19], TCC_EA_WRREQ_64B[19], TCC_EA_RDREQ_IO_CREDIT_STALL[20], TCC_EA_RDREQ_LEVEL[20], TCC_EA_WRREQ[20], TCC_EA_WRREQ_64B[20], TCC_EA_RDREQ_IO_CREDIT_STALL[21], TCC_EA_RDREQ_LEVEL[21], TCC_EA_WRREQ[21], TCC_EA_WRREQ_64B[21], TCC_EA_RDREQ_IO_CREDIT_STALL[22], TCC_EA_RDREQ_LEVEL[22], TCC_EA_WRREQ[22], TCC_EA_WRREQ_64B[22], TCC_EA_RDREQ_IO_CREDIT_STALL[23], TCC_EA_RDREQ_LEVEL[23], TCC_EA_WRREQ[23], TCC_EA_WRREQ_64B[23], TCC_EA_RDREQ_IO_CREDIT_STALL[24], TCC_EA_RDREQ_LEVEL[24], TCC_EA_WRREQ[24], TCC_EA_WRREQ_64B[24], TCC_EA_RDREQ_IO_CREDIT_STALL[25], TCC_EA_RDREQ_LEVEL[25], TCC_EA_WRREQ[25], TCC_EA_WRREQ_64B[25], TCC_EA_RDREQ_IO_CREDIT_STALL[26], TCC_EA_RDREQ_LEVEL[26], TCC_EA_WRREQ[26], TCC_EA_WRREQ_64B[26], TCC_EA_RDREQ_IO_CREDIT_STALL[27], TCC_EA_RDREQ_LEVEL[27], TCC_EA_WRREQ[27], TCC_EA_WRREQ_64B[27], TCC_EA_RDREQ_IO_CREDIT_STALL[28], TCC_EA_RDREQ_LEVEL[28], TCC_EA_WRREQ[28], TCC_EA_WRREQ_64B[28], TCC_EA_RDREQ_IO_CREDIT_STALL[29], TCC_EA_RDREQ_LEVEL[29], TCC_EA_WRREQ[29], TCC_EA_WRREQ_64B[29], TCC_EA_RDREQ_IO_CREDIT_STALL[30], TCC_EA_RDREQ_LEVEL[30], TCC_EA_WRREQ[30], TCC_EA_WRREQ_64B[30], TCC_EA_RDREQ_IO_CREDIT_STALL[31], TCC_EA_RDREQ_LEVEL[31], TCC_EA_WRREQ[31], TCC_EA_WRREQ_64B[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170506_4192664/input0_results_240321_170506
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_15.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_16.txt
|-> [rocprof] RPL: on '240321_170507' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_16.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170507_4192866'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170507_4192866/input0_results_240321_170507'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170507_4192866/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_WRREQ_DRAM_CREDIT_STALL[0], TCC_EA_WRREQ_GMI_CREDIT_STALL[0], TCC_EA_WRREQ_IO_CREDIT_STALL[0], TCC_EA_WRREQ_LEVEL[0], TCC_EA_WRREQ_DRAM_CREDIT_STALL[1], TCC_EA_WRREQ_GMI_CREDIT_STALL[1], TCC_EA_WRREQ_IO_CREDIT_STALL[1], TCC_EA_WRREQ_LEVEL[1], TCC_EA_WRREQ_DRAM_CREDIT_STALL[2], TCC_EA_WRREQ_GMI_CREDIT_STALL[2], TCC_EA_WRREQ_IO_CREDIT_STALL[2], TCC_EA_WRREQ_LEVEL[2], TCC_EA_WRREQ_DRAM_CREDIT_STALL[3], TCC_EA_WRREQ_GMI_CREDIT_STALL[3], TCC_EA_WRREQ_IO_CREDIT_STALL[3], TCC_EA_WRREQ_LEVEL[3], TCC_EA_WRREQ_DRAM_CREDIT_STALL[4], TCC_EA_WRREQ_GMI_CREDIT_STALL[4], TCC_EA_WRREQ_IO_CREDIT_STALL[4], TCC_EA_WRREQ_LEVEL[4], TCC_EA_WRREQ_DRAM_CREDIT_STALL[5], TCC_EA_WRREQ_GMI_CREDIT_STALL[5], TCC_EA_WRREQ_IO_CREDIT_STALL[5], TCC_EA_WRREQ_LEVEL[5], TCC_EA_WRREQ_DRAM_CREDIT_STALL[6], TCC_EA_WRREQ_GMI_CREDIT_STALL[6], TCC_EA_WRREQ_IO_CREDIT_STALL[6], TCC_EA_WRREQ_LEVEL[6], TCC_EA_WRREQ_DRAM_CREDIT_STALL[7], TCC_EA_WRREQ_GMI_CREDIT_STALL[7], TCC_EA_WRREQ_IO_CREDIT_STALL[7], TCC_EA_WRREQ_LEVEL[7], TCC_EA_WRREQ_DRAM_CREDIT_STALL[8], TCC_EA_WRREQ_GMI_CREDIT_STALL[8], TCC_EA_WRREQ_IO_CREDIT_STALL[8], TCC_EA_WRREQ_LEVEL[8], TCC_EA_WRREQ_DRAM_CREDIT_STALL[9], TCC_EA_WRREQ_GMI_CREDIT_STALL[9], TCC_EA_WRREQ_IO_CREDIT_STALL[9], TCC_EA_WRREQ_LEVEL[9], TCC_EA_WRREQ_DRAM_CREDIT_STALL[10], TCC_EA_WRREQ_GMI_CREDIT_STALL[10], TCC_EA_WRREQ_IO_CREDIT_STALL[10], TCC_EA_WRREQ_LEVEL[10], TCC_EA_WRREQ_DRAM_CREDIT_STALL[11], TCC_EA_WRREQ_GMI_CREDIT_STALL[11], TCC_EA_WRREQ_IO_CREDIT_STALL[11], TCC_EA_WRREQ_LEVEL[11], TCC_EA_WRREQ_DRAM_CREDIT_STALL[12], TCC_EA_WRREQ_GMI_CREDIT_STALL[12], TCC_EA_WRREQ_IO_CREDIT_STALL[12], TCC_EA_WRREQ_LEVEL[12], TCC_EA_WRREQ_DRAM_CREDIT_STALL[13], TCC_EA_WRREQ_GMI_CREDIT_STALL[13], TCC_EA_WRREQ_IO_CREDIT_STALL[13], TCC_EA_WRREQ_LEVEL[13], TCC_EA_WRREQ_DRAM_CREDIT_STALL[14], TCC_EA_WRREQ_GMI_CREDIT_STALL[14], TCC_EA_WRREQ_IO_CREDIT_STALL[14], TCC_EA_WRREQ_LEVEL[14], TCC_EA_WRREQ_DRAM_CREDIT_STALL[15], TCC_EA_WRREQ_GMI_CREDIT_STALL[15], TCC_EA_WRREQ_IO_CREDIT_STALL[15], TCC_EA_WRREQ_LEVEL[15], TCC_EA_WRREQ_DRAM_CREDIT_STALL[16], TCC_EA_WRREQ_GMI_CREDIT_STALL[16], TCC_EA_WRREQ_IO_CREDIT_STALL[16], TCC_EA_WRREQ_LEVEL[16], TCC_EA_WRREQ_DRAM_CREDIT_STALL[17], TCC_EA_WRREQ_GMI_CREDIT_STALL[17], TCC_EA_WRREQ_IO_CREDIT_STALL[17], TCC_EA_WRREQ_LEVEL[17], TCC_EA_WRREQ_DRAM_CREDIT_STALL[18], TCC_EA_WRREQ_GMI_CREDIT_STALL[18], TCC_EA_WRREQ_IO_CREDIT_STALL[18], TCC_EA_WRREQ_LEVEL[18], TCC_EA_WRREQ_DRAM_CREDIT_STALL[19], TCC_EA_WRREQ_GMI_CREDIT_STALL[19], TCC_EA_WRREQ_IO_CREDIT_STALL[19], TCC_EA_WRREQ_LEVEL[19], TCC_EA_WRREQ_DRAM_CREDIT_STALL[20], TCC_EA_WRREQ_GMI_CREDIT_STALL[20], TCC_EA_WRREQ_IO_CREDIT_STALL[20], TCC_EA_WRREQ_LEVEL[20], TCC_EA_WRREQ_DRAM_CREDIT_STALL[21], TCC_EA_WRREQ_GMI_CREDIT_STALL[21], TCC_EA_WRREQ_IO_CREDIT_STALL[21], TCC_EA_WRREQ_LEVEL[21], TCC_EA_WRREQ_DRAM_CREDIT_STALL[22], TCC_EA_WRREQ_GMI_CREDIT_STALL[22], TCC_EA_WRREQ_IO_CREDIT_STALL[22], TCC_EA_WRREQ_LEVEL[22], TCC_EA_WRREQ_DRAM_CREDIT_STALL[23], TCC_EA_WRREQ_GMI_CREDIT_STALL[23], TCC_EA_WRREQ_IO_CREDIT_STALL[23], TCC_EA_WRREQ_LEVEL[23], TCC_EA_WRREQ_DRAM_CREDIT_STALL[24], TCC_EA_WRREQ_GMI_CREDIT_STALL[24], TCC_EA_WRREQ_IO_CREDIT_STALL[24], TCC_EA_WRREQ_LEVEL[24], TCC_EA_WRREQ_DRAM_CREDIT_STALL[25], TCC_EA_WRREQ_GMI_CREDIT_STALL[25], TCC_EA_WRREQ_IO_CREDIT_STALL[25], TCC_EA_WRREQ_LEVEL[25], TCC_EA_WRREQ_DRAM_CREDIT_STALL[26], TCC_EA_WRREQ_GMI_CREDIT_STALL[26], TCC_EA_WRREQ_IO_CREDIT_STALL[26], TCC_EA_WRREQ_LEVEL[26], TCC_EA_WRREQ_DRAM_CREDIT_STALL[27], TCC_EA_WRREQ_GMI_CREDIT_STALL[27], TCC_EA_WRREQ_IO_CREDIT_STALL[27], TCC_EA_WRREQ_LEVEL[27], TCC_EA_WRREQ_DRAM_CREDIT_STALL[28], TCC_EA_WRREQ_GMI_CREDIT_STALL[28], TCC_EA_WRREQ_IO_CREDIT_STALL[28], TCC_EA_WRREQ_LEVEL[28], TCC_EA_WRREQ_DRAM_CREDIT_STALL[29], TCC_EA_WRREQ_GMI_CREDIT_STALL[29], TCC_EA_WRREQ_IO_CREDIT_STALL[29], TCC_EA_WRREQ_LEVEL[29], TCC_EA_WRREQ_DRAM_CREDIT_STALL[30], TCC_EA_WRREQ_GMI_CREDIT_STALL[30], TCC_EA_WRREQ_IO_CREDIT_STALL[30], TCC_EA_WRREQ_LEVEL[30], TCC_EA_WRREQ_DRAM_CREDIT_STALL[31], TCC_EA_WRREQ_GMI_CREDIT_STALL[31], TCC_EA_WRREQ_IO_CREDIT_STALL[31], TCC_EA_WRREQ_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170507_4192866/input0_results_240321_170507
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_16.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_17.txt
|-> [rocprof] RPL: on '240321_170508' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_17.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170508_4193068'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170508_4193068/input0_results_240321_170508'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170508_4193068/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_HIT[0], TCC_MISS[0], TCC_READ[0], TCC_REQ[0], TCC_HIT[1], TCC_MISS[1], TCC_READ[1], TCC_REQ[1], TCC_HIT[2], TCC_MISS[2], TCC_READ[2], TCC_REQ[2], TCC_HIT[3], TCC_MISS[3], TCC_READ[3], TCC_REQ[3], TCC_HIT[4], TCC_MISS[4], TCC_READ[4], TCC_REQ[4], TCC_HIT[5], TCC_MISS[5], TCC_READ[5], TCC_REQ[5], TCC_HIT[6], TCC_MISS[6], TCC_READ[6], TCC_REQ[6], TCC_HIT[7], TCC_MISS[7], TCC_READ[7], TCC_REQ[7], TCC_HIT[8], TCC_MISS[8], TCC_READ[8], TCC_REQ[8], TCC_HIT[9], TCC_MISS[9], TCC_READ[9], TCC_REQ[9], TCC_HIT[10], TCC_MISS[10], TCC_READ[10], TCC_REQ[10], TCC_HIT[11], TCC_MISS[11], TCC_READ[11], TCC_REQ[11], TCC_HIT[12], TCC_MISS[12], TCC_READ[12], TCC_REQ[12], TCC_HIT[13], TCC_MISS[13], TCC_READ[13], TCC_REQ[13], TCC_HIT[14], TCC_MISS[14], TCC_READ[14], TCC_REQ[14], TCC_HIT[15], TCC_MISS[15], TCC_READ[15], TCC_REQ[15], TCC_HIT[16], TCC_MISS[16], TCC_READ[16], TCC_REQ[16], TCC_HIT[17], TCC_MISS[17], TCC_READ[17], TCC_REQ[17], TCC_HIT[18], TCC_MISS[18], TCC_READ[18], TCC_REQ[18], TCC_HIT[19], TCC_MISS[19], TCC_READ[19], TCC_REQ[19], TCC_HIT[20], TCC_MISS[20], TCC_READ[20], TCC_REQ[20], TCC_HIT[21], TCC_MISS[21], TCC_READ[21], TCC_REQ[21], TCC_HIT[22], TCC_MISS[22], TCC_READ[22], TCC_REQ[22], TCC_HIT[23], TCC_MISS[23], TCC_READ[23], TCC_REQ[23], TCC_HIT[24], TCC_MISS[24], TCC_READ[24], TCC_REQ[24], TCC_HIT[25], TCC_MISS[25], TCC_READ[25], TCC_REQ[25], TCC_HIT[26], TCC_MISS[26], TCC_READ[26], TCC_REQ[26], TCC_HIT[27], TCC_MISS[27], TCC_READ[27], TCC_REQ[27], TCC_HIT[28], TCC_MISS[28], TCC_READ[28], TCC_REQ[28], TCC_HIT[29], TCC_MISS[29], TCC_READ[29], TCC_REQ[29], TCC_HIT[30], TCC_MISS[30], TCC_READ[30], TCC_REQ[30], TCC_HIT[31], TCC_MISS[31], TCC_READ[31], TCC_REQ[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170508_4193068/input0_results_240321_170508
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_17.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_18.txt
|-> [rocprof] RPL: on '240321_170508' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_18.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170508_4193269'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170508_4193269/input0_results_240321_170508'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170508_4193269/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 96 metrics
|-> [rocprof] TCC_RW_REQ[0], TCC_TOO_MANY_EA_WRREQS_STALL[0], TCC_WRITE[0], TCC_RW_REQ[1], TCC_TOO_MANY_EA_WRREQS_STALL[1], TCC_WRITE[1], TCC_RW_REQ[2], TCC_TOO_MANY_EA_WRREQS_STALL[2], TCC_WRITE[2], TCC_RW_REQ[3], TCC_TOO_MANY_EA_WRREQS_STALL[3], TCC_WRITE[3], TCC_RW_REQ[4], TCC_TOO_MANY_EA_WRREQS_STALL[4], TCC_WRITE[4], TCC_RW_REQ[5], TCC_TOO_MANY_EA_WRREQS_STALL[5], TCC_WRITE[5], TCC_RW_REQ[6], TCC_TOO_MANY_EA_WRREQS_STALL[6], TCC_WRITE[6], TCC_RW_REQ[7], TCC_TOO_MANY_EA_WRREQS_STALL[7], TCC_WRITE[7], TCC_RW_REQ[8], TCC_TOO_MANY_EA_WRREQS_STALL[8], TCC_WRITE[8], TCC_RW_REQ[9], TCC_TOO_MANY_EA_WRREQS_STALL[9], TCC_WRITE[9], TCC_RW_REQ[10], TCC_TOO_MANY_EA_WRREQS_STALL[10], TCC_WRITE[10], TCC_RW_REQ[11], TCC_TOO_MANY_EA_WRREQS_STALL[11], TCC_WRITE[11], TCC_RW_REQ[12], TCC_TOO_MANY_EA_WRREQS_STALL[12], TCC_WRITE[12], TCC_RW_REQ[13], TCC_TOO_MANY_EA_WRREQS_STALL[13], TCC_WRITE[13], TCC_RW_REQ[14], TCC_TOO_MANY_EA_WRREQS_STALL[14], TCC_WRITE[14], TCC_RW_REQ[15], TCC_TOO_MANY_EA_WRREQS_STALL[15], TCC_WRITE[15], TCC_RW_REQ[16], TCC_TOO_MANY_EA_WRREQS_STALL[16], TCC_WRITE[16], TCC_RW_REQ[17], TCC_TOO_MANY_EA_WRREQS_STALL[17], TCC_WRITE[17], TCC_RW_REQ[18], TCC_TOO_MANY_EA_WRREQS_STALL[18], TCC_WRITE[18], TCC_RW_REQ[19], TCC_TOO_MANY_EA_WRREQS_STALL[19], TCC_WRITE[19], TCC_RW_REQ[20], TCC_TOO_MANY_EA_WRREQS_STALL[20], TCC_WRITE[20], TCC_RW_REQ[21], TCC_TOO_MANY_EA_WRREQS_STALL[21], TCC_WRITE[21], TCC_RW_REQ[22], TCC_TOO_MANY_EA_WRREQS_STALL[22], TCC_WRITE[22], TCC_RW_REQ[23], TCC_TOO_MANY_EA_WRREQS_STALL[23], TCC_WRITE[23], TCC_RW_REQ[24], TCC_TOO_MANY_EA_WRREQS_STALL[24], TCC_WRITE[24], TCC_RW_REQ[25], TCC_TOO_MANY_EA_WRREQS_STALL[25], TCC_WRITE[25], TCC_RW_REQ[26], TCC_TOO_MANY_EA_WRREQS_STALL[26], TCC_WRITE[26], TCC_RW_REQ[27], TCC_TOO_MANY_EA_WRREQS_STALL[27], TCC_WRITE[27], TCC_RW_REQ[28], TCC_TOO_MANY_EA_WRREQS_STALL[28], TCC_WRITE[28], TCC_RW_REQ[29], TCC_TOO_MANY_EA_WRREQS_STALL[29], TCC_WRITE[29], TCC_RW_REQ[30], TCC_TOO_MANY_EA_WRREQS_STALL[30], TCC_WRITE[30], TCC_RW_REQ[31], TCC_TOO_MANY_EA_WRREQS_STALL[31], TCC_WRITE[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170508_4193269/input0_results_240321_170508
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_18.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_2.txt
|-> [rocprof] RPL: on '240321_170509' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_2.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170509_4193471'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170509_4193471/input0_results_240321_170509'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170509_4193471/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 26 metrics
|-> [rocprof] SQ_INSTS_VALU_MUL_F32, SQ_INSTS_VALU_FMA_F32, SQ_INSTS_VALU_TRANS_F32, SQ_INSTS_VALU_ADD_F64, SQ_INSTS_VALU_MUL_F64, SQ_INSTS_VALU_FMA_F64, SQ_INSTS_VALU_TRANS_F64, SQ_INSTS_VALU_INT32, TCP_VOLATILE_sum, TCP_TOTAL_ACCESSES_sum, TCP_TOTAL_READ_sum, TCP_TOTAL_WRITE_sum, TA_BUFFER_ATOMIC_WAVEFRONTS_sum, TA_BUFFER_TOTAL_CYCLES_sum, TD_ATOMIC_WAVEFRONT_sum, TD_STORE_WAVEFRONT_sum, SPI_RA_REQ_NO_ALLOC, SPI_RA_REQ_NO_ALLOC_CSN, CPC_CPC_STAT_STALL, CPC_UTCL1_STALL_ON_TRANSLATION, CPF_CPF_STAT_IDLE, CPF_CPF_TCIU_IDLE, TCC_REQ_sum, TCC_STREAMING_REQ_sum, TCC_HIT_sum, TCC_MISS_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170509_4193471/input0_results_240321_170509
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_2.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_3.txt
|-> [rocprof] RPL: on '240321_170510' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_3.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170510_4193673'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170510_4193673/input0_results_240321_170510'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170510_4193673/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 24 metrics
|-> [rocprof] SQ_INSTS_VALU_INT64, SQ_INSTS_SMEM, SQ_INSTS_FLAT, SQ_INSTS_LDS, SQ_INSTS_GDS, SQ_INSTS_EXP_GDS, SQ_INSTS_BRANCH, SQ_INSTS_SENDMSG, TCP_TOTAL_ATOMIC_WITH_RET_sum, TCP_TOTAL_ATOMIC_WITHOUT_RET_sum, TCP_TOTAL_WRITEBACK_INVALIDATES_sum, TCP_TOTAL_CACHE_ACCESSES_sum, TA_BUFFER_COALESCED_READ_CYCLES_sum, TA_BUFFER_COALESCED_WRITE_CYCLES_sum, TD_COALESCABLE_WAVEFRONT_sum, SPI_RA_RES_STALL_CSN, SPI_RA_TMP_STALL_CSN, CPC_CPC_UTCL2IU_BUSY, CPC_CPC_UTCL2IU_IDLE, CPF_CMP_UTCL1_STALL_ON_TRANSLATION, TCC_READ_sum, TCC_WRITE_sum, TCC_ATOMIC_sum, TCC_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170510_4193673/input0_results_240321_170510
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_3.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_4.txt
|-> [rocprof] RPL: on '240321_170510' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_4.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170510_4193875'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170510_4193875/input0_results_240321_170510'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170510_4193875/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 22 metrics
|-> [rocprof] SQ_WAVE_CYCLES, SQ_WAIT_ANY, SQ_WAIT_INST_ANY, SQ_ACTIVE_INST_ANY, SQ_BUSY_CU_CYCLES, SQ_ACTIVE_INST_VMEM, SQ_ACTIVE_INST_LDS, SQ_ACTIVE_INST_VALU, TCP_UTCL1_TRANSLATION_MISS_sum, TCP_UTCL1_TRANSLATION_HIT_sum, TCP_UTCL1_PERMISSION_MISS_sum, TCP_UTCL1_REQUEST_sum, TA_ADDR_STALLED_BY_TC_CYCLES_sum, TA_TOTAL_WAVEFRONTS_sum, SPI_RA_WAVE_SIMD_FULL_CSN, SPI_RA_VGPR_SIMD_FULL_CSN, CPC_CPC_UTCL2IU_STALL, CPC_ME1_BUSY_FOR_PACKET_DECODE, TCC_EA_WRREQ_sum, TCC_EA_WRREQ_64B_sum, TCC_EA_WR_UNCACHED_32B_sum, TCC_EA_WRREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170510_4193875/input0_results_240321_170510
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_4.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_5.txt
|-> [rocprof] RPL: on '240321_170510' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_5.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170510_4194077'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170510_4194077/input0_results_240321_170510'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170510_4194077/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 21 metrics
|-> [rocprof] SQ_ACTIVE_INST_SCA, SQ_ACTIVE_INST_EXP_GDS, SQ_ACTIVE_INST_MISC, SQ_ACTIVE_INST_FLAT, SQ_INST_CYCLES_VMEM_WR, SQ_INST_CYCLES_VMEM_RD, SQ_INST_CYCLES_SMEM, SQ_INST_CYCLES_SALU, TCP_TCP_LATENCY_sum, TCP_TCC_READ_REQ_LATENCY_sum, TCP_TCC_WRITE_REQ_LATENCY_sum, TCP_TCC_READ_REQ_sum, TA_ADDR_STALLED_BY_TD_CYCLES_sum, TA_DATA_STALLED_BY_TC_CYCLES_sum, SPI_RA_SGPR_SIMD_FULL_CSN, SPI_RA_LDS_CU_FULL_CSN, CPC_ME1_DC0_SPI_BUSY, TCC_EA_WRREQ_STALL_sum, TCC_EA_RDREQ_sum, TCC_EA_RDREQ_32B_sum, TCC_EA_RD_UNCACHED_32B_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170510_4194077/input0_results_240321_170510
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_5.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_6.txt
|-> [rocprof] RPL: on '240321_170511' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_6.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170511_4194261'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170511_4194261/input0_results_240321_170511'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170511_4194261/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_THREAD_CYCLES_VALU, SQ_IFETCH, SQ_LDS_BANK_CONFLICT, SQ_LDS_ADDR_CONFLICT, SQ_LDS_UNALIGNED_STALL, SQ_WAVES_EQ_64, SQ_WAVES_LT_64, SQ_WAVES_LT_48, TCP_TCC_WRITE_REQ_sum, TCP_TCC_ATOMIC_WITH_RET_REQ_sum, TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum, TCP_TCC_NC_READ_REQ_sum, TA_FLAT_WAVEFRONTS_sum, TA_FLAT_READ_WAVEFRONTS_sum, SPI_RA_BAR_CU_FULL_CSN, SPI_RA_TGLIM_CU_FULL_CSN, TCC_EA_RDREQ_DRAM_sum, TCC_TAG_STALL_sum, TCC_NORMAL_WRITEBACK_sum, TCC_ALL_TC_OP_WB_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170511_4194261/input0_results_240321_170511
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_6.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_7.txt
|-> [rocprof] RPL: on '240321_170511' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_7.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170511_672'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170511_672/input0_results_240321_170511'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170511_672/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_WAVES_LT_32, SQ_WAVES_LT_16, SQ_ITEMS, SQ_LDS_MEM_VIOLATIONS, SQ_LDS_ATOMIC_RETURN, SQ_LDS_IDX_ACTIVE, SQ_WAVES_RESTORED, SQ_WAVES_SAVED, TCP_TCC_NC_WRITE_REQ_sum, TCP_TCC_NC_ATOMIC_REQ_sum, TCP_TCC_UC_READ_REQ_sum, TCP_TCC_UC_WRITE_REQ_sum, TA_FLAT_WRITE_WAVEFRONTS_sum, TA_FLAT_ATOMIC_WAVEFRONTS_sum, SPI_RA_WVLIM_STALL_CSN, SPI_SWC_CSC_WR, TCC_NORMAL_EVICT_sum, TCC_ALL_TC_OP_INV_EVICT_sum, TCC_TOO_MANY_EA_WRREQS_STALL_sum, TCC_EA_ATOMIC_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170511_672/input0_results_240321_170511
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_7.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_8.txt
|-> [rocprof] RPL: on '240321_170512' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_8.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170512_894'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170512_894/input0_results_240321_170512'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170512_894/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 17 metrics
|-> [rocprof] SQ_INSTS_SMEM_NORM, SQ_INSTS_MFMA, SQ_INSTS_VALU_MFMA_I8, SQ_INSTS_VALU_MFMA_F16, SQ_INSTS_VALU_MFMA_BF16, SQ_INSTS_VALU_MFMA_F32, SQ_INSTS_VALU_MFMA_F64, SQ_VALU_MFMA_BUSY_CYCLES, TCP_TCC_UC_ATOMIC_REQ_sum, TCP_TCC_CC_READ_REQ_sum, TCP_TCC_CC_WRITE_REQ_sum, TCP_TCC_CC_ATOMIC_REQ_sum, SPI_VWC_CSC_WR, SPI_RA_BULKY_CU_FULL_CSN, TCC_EA_RDREQ_LEVEL_sum, TCC_EA_WRREQ_LEVEL_sum, TCC_EA_ATOMIC_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170512_894/input0_results_240321_170512
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_8.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_9.txt
|-> [rocprof] RPL: on '240321_170512' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_9.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170512_1097'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170512_1097/input0_results_240321_170512'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170512_1097/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 12 metrics
|-> [rocprof] SQ_INSTS_FLAT_LDS_ONLY, SQ_INSTS_VALU_MFMA_MOPS_I8, SQ_INSTS_VALU_MFMA_MOPS_F16, SQ_INSTS_VALU_MFMA_MOPS_BF16, SQ_INSTS_VALU_MFMA_MOPS_F32, SQ_INSTS_VALU_MFMA_MOPS_F64, SQC_TC_INST_REQ, SQC_TC_DATA_READ_REQ, TCP_TCC_RW_READ_REQ_sum, TCP_TCC_RW_WRITE_REQ_sum, TCP_TCC_RW_ATOMIC_REQ_sum, TCP_PENDING_STALL_CYCLES_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170512_1097/input0_results_240321_170512
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_9.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/timestamps.txt
|-> [rocprof] RPL: on '240321_170513' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/timestamps.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170513_1301'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170513_1301/input0_results_240321_170513'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170513_1301/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 0 metrics
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170513_1301/input0_results_240321_170513
|-> [rocprof] File 'tests/workloads/kernel/MI200/timestamps.csv' is generating
|-> [rocprof]
[roofline] Checking for roofline.csv in tests/workloads/kernel/MI200
[roofline] No roofline data found. Generating...
@@ -2,4 +2,4 @@ pmc: GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PRE
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVE
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_CYCLES SQ_BUSY_CYCLES SQ_WAVES SQ_INSTS_VALU_CVT SQ_INSTS_VMEM_WR SQ_IN
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_VSKIPPED SQ_INSTS SQ_INSTS_VALU SQ_INSTS_VALU_ADD_F16 SQ_INSTS_VA
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQC_TC_DATA_WRITE_REQ SQC_TC_DATA_ATOMIC_REQ SQC_TC_STALL SQC_TC_REQ SQC_D
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQC_ICACHE_MISSES_DUPLICATE SQC_DCACHE_INPUT_VALID_READYB SQC_DCACHE_ATOMI
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQC_DCACHE_REQ_READ_1 SQC_DCACHE_REQ_READ_2 SQC_DCACHE_REQ_READ_4
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_ATOMIC[0] TCC_CYCLE[0] TCC_EA_ATOMIC[0] TCC_EA_ATOMIC_LEVEL[0] TCC_ATO
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_EA_RDREQ[0] TCC_EA_RDREQ_32B[0] TCC_EA_RDREQ_DRAM_CREDIT_STALL[0] TCC_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_EA_RDREQ_IO_CREDIT_STALL[0] TCC_EA_RDREQ_LEVEL[0] TCC_EA_WRREQ[0] TCC_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_EA_WRREQ_DRAM_CREDIT_STALL[0] TCC_EA_WRREQ_GMI_CREDIT_STALL[0] TCC_EA_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_HIT[0] TCC_MISS[0] TCC_READ[0] TCC_REQ[0] TCC_HIT[1] TCC_MISS[1] TCC_R
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: TCC_RW_REQ[0] TCC_TOO_MANY_EA_WRREQS_STALL[0] TCC_WRITE[0] TCC_RW_REQ[1] T
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_VALU_MUL_F32 SQ_INSTS_VALU_FMA_F32 SQ_INSTS_VALU_TRANS_F32 SQ_INS
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_VALU_INT64 SQ_INSTS_SMEM SQ_INSTS_FLAT SQ_INSTS_LDS SQ_INSTS_GDS
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_WAVE_CYCLES SQ_WAIT_ANY SQ_WAIT_INST_ANY SQ_ACTIVE_INST_ANY SQ_BUSY_CU_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_ACTIVE_INST_SCA SQ_ACTIVE_INST_EXP_GDS SQ_ACTIVE_INST_MISC SQ_ACTIVE_IN
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_THREAD_CYCLES_VALU SQ_IFETCH SQ_LDS_BANK_CONFLICT SQ_LDS_ADDR_CONFLICT
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_WAVES_LT_32 SQ_WAVES_LT_16 SQ_ITEMS SQ_LDS_MEM_VIOLATIONS SQ_LDS_ATOMIC
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_SMEM_NORM SQ_INSTS_MFMA SQ_INSTS_VALU_MFMA_I8 SQ_INSTS_VALU_MFMA_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc: SQ_INSTS_FLAT_LDS_ONLY SQ_INSTS_VALU_MFMA_MOPS_I8 SQ_INSTS_VALU_MFMA_MOPS_
gpu:
range:
kernel:
kernel: vecCopy
@@ -2,4 +2,4 @@ pmc:
gpu:
range:
kernel:
kernel: vecCopy
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1 Dispatch_ID Kernel_Name GPU_ID
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
File diff suppressed because one or more lines are too long
@@ -0,0 +1,5 @@
device,HBMBw,HBMBwLow,hbmBwHigh,L2Bw,L2BwLow,L2BwHigh,L1Bw,L1BwLow,L1BwHigh,LDSBw,LDSBwLow,LDSBwHigh,FP32Flops,FP32FlopsLow,FP32FlopsHigh,FP64Flops,FP64FlopsLow,FP64FlopsHigh,MFMABF16Flops,MFMABF16FlopsLow,MFMABF16FlopsHigh,MFMAF16Flops,MFMAF16FlopsLow,MFMAF16FlopsHigh,MFMAF32Flops,MFMAF32FlopsLow,MFMAF32FlopsHigh,MFMAF64Flops,MFMAF64FlopsLow,MFMAF64FlopsHigh,MFMAI8Ops,MFMAFI8OpsLow,MFMAI8OpsHigh
0,1388.8003,1388.2018,1389.3988,5018.0034,5014.9653,5021.0415,9227.3828,9226.832,9227.9336,17715.787,17712.695,17718.879,20942.057,20881.729,21002.385,20273.387,20272.754,20274.02,170490.59,170486.83,170494.36,164905.91,164902.36,164909.45,41435.109,41434.566,41435.652,41490.23,41489.523,41490.938,166390.95,165895.59,166886.31
1,1389.1022,1388.6135,1389.5908,5028.1416,5026.2451,5030.0381,9235.0059,9234.3955,9235.6162,18231.234,18229.113,18233.355,20975.01,20907.771,21042.248,20293.785,20293.287,20294.283,170865.64,170860.17,170871.11,165134.81,165131.06,165138.56,41491.012,41490.277,41491.746,41546.773,41546.109,41547.438,166891.17,166888.8,166893.55
2,1388.7727,1388.2306,1389.3148,5035.6616,5033.3149,5038.0083,9261.2324,9260.5088,9261.9561,18217.076,18216.133,18218.02,21049.338,21048.953,21049.723,20351.391,20350.875,20351.906,171111.95,171108.86,171115.05,165494.22,165491.36,165497.08,41585.855,41584.934,41586.777,41643.219,41642.5,41643.938,166984.53,166391.91,167577.16
3,1389.2549,1388.5906,1389.9192,5032.6069,5031.0239,5034.1899,9239.5078,9238.6982,9240.3174,18329.676,18328.783,18330.568,21036.42,21036.109,21036.73,20313.297,20312.844,20313.75,171033,171029.09,171036.91,165165.02,165160.69,165169.34,41426.988,41272.484,41581.492,41573.762,41572.508,41575.016,166972.52,166968.34,166976.69
1 device HBMBw HBMBwLow hbmBwHigh L2Bw L2BwLow L2BwHigh L1Bw L1BwLow L1BwHigh LDSBw LDSBwLow LDSBwHigh FP32Flops FP32FlopsLow FP32FlopsHigh FP64Flops FP64FlopsLow FP64FlopsHigh MFMABF16Flops MFMABF16FlopsLow MFMABF16FlopsHigh MFMAF16Flops MFMAF16FlopsLow MFMAF16FlopsHigh MFMAF32Flops MFMAF32FlopsLow MFMAF32FlopsHigh MFMAF64Flops MFMAF64FlopsLow MFMAF64FlopsHigh MFMAI8Ops MFMAFI8OpsLow MFMAI8OpsHigh
2 0 1388.8003 1388.2018 1389.3988 5018.0034 5014.9653 5021.0415 9227.3828 9226.832 9227.9336 17715.787 17712.695 17718.879 20942.057 20881.729 21002.385 20273.387 20272.754 20274.02 170490.59 170486.83 170494.36 164905.91 164902.36 164909.45 41435.109 41434.566 41435.652 41490.23 41489.523 41490.938 166390.95 165895.59 166886.31
3 1 1389.1022 1388.6135 1389.5908 5028.1416 5026.2451 5030.0381 9235.0059 9234.3955 9235.6162 18231.234 18229.113 18233.355 20975.01 20907.771 21042.248 20293.785 20293.287 20294.283 170865.64 170860.17 170871.11 165134.81 165131.06 165138.56 41491.012 41490.277 41491.746 41546.773 41546.109 41547.438 166891.17 166888.8 166893.55
4 2 1388.7727 1388.2306 1389.3148 5035.6616 5033.3149 5038.0083 9261.2324 9260.5088 9261.9561 18217.076 18216.133 18218.02 21049.338 21048.953 21049.723 20351.391 20350.875 20351.906 171111.95 171108.86 171115.05 165494.22 165491.36 165497.08 41585.855 41584.934 41586.777 41643.219 41642.5 41643.938 166984.53 166391.91 167577.16
5 3 1389.2549 1388.5906 1389.9192 5032.6069 5031.0239 5034.1899 9239.5078 9238.6982 9240.3174 18329.676 18328.783 18330.568 21036.42 21036.109 21036.73 20313.297 20312.844 20313.75 171033 171029.09 171036.91 165165.02 165160.69 165169.34 41426.988 41272.484 41581.492 41573.762 41572.508 41575.016 166972.52 166968.34 166976.69
@@ -0,0 +1,2 @@
workload_name,command,ip_blocks,timestamp,version,hostname,cpu_model,sbios,linux_distro,linux_kernel_version,amd_gpu_kernel_version,cpu_memory,gpu_memory,rocm_version,vbios,compute_partition,memory_partition,gpu_model,gpu_arch,gpu_l1,gpu_l2,cu_per_gpu,simd_per_cu,se_per_gpu,wave_size,workgroup_max_size,max_waves_per_cu,max_sclk,max_mclk,cur_sclk,cur_mclk,total_l2_chan,lds_banks_per_cu,sqc_per_gpu,pipes_per_gpu,hbm_bw,num_xcd
kernel,./tests/vcopy -n 1048576 -b 256 -i 3,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF|roofline,Thu 21 Mar 2024 05:04:59 PM (CDT),2,t007-002.hpcfund,AMD EPYC 7V13 64-Core Processor,American Megatrends Inc.0602,Rocky Linux 9.1 (Blue Onyx),5.14.0-162.18.1.el9_1.x86_64,,527650760,,6.0.2-115,113-D67301-059,NA,NA,MI200,gfx90a,16,8192,104,4,8,64,1024,32,1700,1600,1700,1600,32,32,56,4,1638.4,1
1 workload_name command ip_blocks timestamp version hostname cpu_model sbios linux_distro linux_kernel_version amd_gpu_kernel_version cpu_memory gpu_memory rocm_version vbios compute_partition memory_partition gpu_model gpu_arch gpu_l1 gpu_l2 cu_per_gpu simd_per_cu se_per_gpu wave_size workgroup_max_size max_waves_per_cu max_sclk max_mclk cur_sclk cur_mclk total_l2_chan lds_banks_per_cu sqc_per_gpu pipes_per_gpu hbm_bw num_xcd
2 kernel ./tests/vcopy -n 1048576 -b 256 -i 3 SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF|roofline Thu 21 Mar 2024 05:04:59 PM (CDT) 2 t007-002.hpcfund AMD EPYC 7V13 64-Core Processor American Megatrends Inc.0602 Rocky Linux 9.1 (Blue Onyx) 5.14.0-162.18.1.el9_1.x86_64 527650760 6.0.2-115 113-D67301-059 NA NA MI200 gfx90a 16 8192 104 4 8 64 1024 32 1700 1600 1700 1600 32 32 56 4 1638.4 1
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1463,1463,1048576,256,0,0,8,0,16,64,0x0,0x7f09a4408ec0,1414445656634195,1414445656659785,1414445656680105,1414445656694409
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1463,1463,1048576,256,0,0,8,0,16,64,0x0,0x7f09a4408ec0,1414445656691093,1414445656699625,1414445656715145,1414445656771244
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1463,1463,1048576,256,0,0,8,0,16,64,0x0,0x7f09a4408ec0,1414445656725698,1414445656774345,1414445656790665,1414445656791793
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1463 1463 1048576 256 0 0 8 0 16 64 0x0 0x7f09a4408ec0 1414445656634195 1414445656659785 1414445656680105 1414445656694409
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1463 1463 1048576 256 0 0 8 0 16 64 0x0 0x7f09a4408ec0 1414445656691093 1414445656699625 1414445656715145 1414445656771244
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1463 1463 1048576 256 0 0 8 0 16 64 0x0 0x7f09a4408ec0 1414445656725698 1414445656774345 1414445656790665 1414445656791793