Add 'projects/rocprofiler-compute/' from commit 'd2cec001161fc49761bd71a498474a447b1d6975'

git-subtree-dir: projects/rocprofiler-compute
git-subtree-mainline: 8a4d7262f8
git-subtree-split: d2cec00116
This commit is contained in:
systems-assistant[bot]
2025-07-17 18:13:42 +00:00
4484 changed files with 289584 additions and 0 deletions
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295056,1295056,1048576,256,0,0,8,8,16,64,0x0,0x7fcafc83cec0,47758,47758,16384,65536,14395,1831576,1414609382362391,1414621055425142,1414621055449622,1414609390882981
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295056,1295056,1048576,256,0,0,8,8,16,64,0x0,0x7fcafc83cec0,46493,46493,16384,65536,8138,1048588,1414609390906726,1414621055537142,1414621055556022,1414609391222231
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295056,1295056,1048576,256,0,0,8,8,16,64,0x0,0x7fcafc83cec0,44994,44994,16384,65536,8169,1048584,1414609391254421,1414621055576182,1414621055595542,1414609391431225
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295056 1295056 1048576 256 0 0 8 8 16 64 0x0 0x7fcafc83cec0 47758 47758 16384 65536 14395 1831576 1414609382362391 1414621055425142 1414621055449622 1414609390882981
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295056 1295056 1048576 256 0 0 8 8 16 64 0x0 0x7fcafc83cec0 46493 46493 16384 65536 8138 1048588 1414609390906726 1414621055537142 1414621055556022 1414609391222231
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295056 1295056 1048576 256 0 0 8 8 16 64 0x0 0x7fcafc83cec0 44994 44994 16384 65536 8169 1048584 1414609391254421 1414621055576182 1414621055595542 1414609391431225
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295257,1295257,1048576,256,0,0,8,8,16,64,0x0,0x7ff0afd60ec0,0,0,0,1414609872531785,1414621055425142,1414621055449622,1414609880166996
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295257,1295257,1048576,256,0,0,8,8,16,64,0x0,0x7ff0afd60ec0,0,0,0,1414609880186282,1414621055537142,1414621055556022,1414609880519821
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295257,1295257,1048576,256,0,0,8,8,16,64,0x0,0x7ff0afd60ec0,0,0,0,1414609880548024,1414621055576182,1414621055595542,1414609880734476
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295257 1295257 1048576 256 0 0 8 8 16 64 0x0 0x7ff0afd60ec0 0 0 0 1414609872531785 1414621055425142 1414621055449622 1414609880166996
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295257 1295257 1048576 256 0 0 8 8 16 64 0x0 0x7ff0afd60ec0 0 0 0 1414609880186282 1414621055537142 1414621055556022 1414609880519821
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295257 1295257 1048576 256 0 0 8 8 16 64 0x0 0x7ff0afd60ec0 0 0 0 1414609880548024 1414621055576182 1414621055595542 1414609880734476
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295459,1295459,1048576,256,0,0,8,8,16,64,0x0,0x7fbaa9898ec0,65536,192926,24614264,1414610346774400,1414621055425142,1414621055449622,1414610354390535
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295459,1295459,1048576,256,0,0,8,8,16,64,0x0,0x7fbaa9898ec0,65536,170544,21889328,1414610354409831,1414621055537142,1414621055556022,1414610354715538
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295459,1295459,1048576,256,0,0,8,8,16,64,0x0,0x7fbaa9898ec0,65536,173974,22234656,1414610354743811,1414621055576182,1414621055595542,1414610354923470
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295459 1295459 1048576 256 0 0 8 8 16 64 0x0 0x7fbaa9898ec0 65536 192926 24614264 1414610346774400 1414621055425142 1414621055449622 1414610354390535
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295459 1295459 1048576 256 0 0 8 8 16 64 0x0 0x7fbaa9898ec0 65536 170544 21889328 1414610354409831 1414621055537142 1414621055556022 1414610354715538
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295459 1295459 1048576 256 0 0 8 8 16 64 0x0 0x7fbaa9898ec0 65536 173974 22234656 1414610354743811 1414621055576182 1414621055595542 1414610354923470
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295661,1295661,1048576,256,0,0,8,8,16,64,0x0,0x7fcdb12dcec0,32768,664178,85014096,1414610810961751,1414621055425142,1414621055449622,1414610818650923
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295661,1295661,1048576,256,0,0,8,8,16,64,0x0,0x7fcdb12dcec0,32768,662767,84821212,1414610818672203,1414621055537142,1414621055556022,1414610818960106
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295661,1295661,1048576,256,0,0,8,8,16,64,0x0,0x7fcdb12dcec0,32768,655809,83963416,1414610818988480,1414621055576182,1414621055595542,1414610819205870
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295661 1295661 1048576 256 0 0 8 8 16 64 0x0 0x7fcdb12dcec0 32768 664178 85014096 1414610810961751 1414621055425142 1414621055449622 1414610818650923
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295661 1295661 1048576 256 0 0 8 8 16 64 0x0 0x7fcdb12dcec0 32768 662767 84821212 1414610818672203 1414621055537142 1414621055556022 1414610818960106
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295661 1295661 1048576 256 0 0 8 8 16 64 0x0 0x7fcdb12dcec0 32768 655809 83963416 1414610818988480 1414621055576182 1414621055595542 1414610819205870
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1295847,1295847,1048576,256,0,0,8,8,16,64,0x0,0x7f5c3f6e4ec0,48355,48355,17730,386848,16384,24958385,233718,0,100367700,1414611287019490,1414621055425142,1414621055449622,1414611294707580
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1295847,1295847,1048576,256,0,0,8,8,16,64,0x0,0x7f5c3f6e4ec0,43528,43528,13690,348232,16384,25762917,238429,0,103589552,1414611294736925,1414621055537142,1414621055556022,1414611295058863
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1295847,1295847,1048576,256,0,0,8,8,16,64,0x0,0x7f5c3f6e4ec0,43639,43639,13971,349120,16384,25009170,233467,0,100574040,1414611295096984,1414621055576182,1414621055595542,1414611295279148
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1295847 1295847 1048576 256 0 0 8 8 16 64 0x0 0x7f5c3f6e4ec0 48355 48355 17730 386848 16384 24958385 233718 0 100367700 1414611287019490 1414621055425142 1414621055449622 1414611294707580
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1295847 1295847 1048576 256 0 0 8 8 16 64 0x0 0x7f5c3f6e4ec0 43528 43528 13690 348232 16384 25762917 238429 0 103589552 1414611294736925 1414621055537142 1414621055556022 1414611295058863
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1295847 1295847 1048576 256 0 0 8 8 16 64 0x0 0x7f5c3f6e4ec0 43639 43639 13971 349120 16384 25009170 233467 0 100574040 1414611295096984 1414621055576182 1414621055595542 1414611295279148
@@ -0,0 +1,702 @@
Omniperf version: 2.0.0-RC1
Profiler choice: rocprofv1
Path: /home1/josantos/omniperf/tests/workloads/kernel/MI100
Target: MI100
Command: ./tests/vcopy -n 1048576 -b 256 -i 3
Kernel Selection: ['vecCopy']
Dispatch Selection: None
IP Blocks: All
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Collecting Performance Counters
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_IFETCH_LEVEL.txt
|-> [rocprof] RPL: on '240321_170755' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_IFETCH_LEVEL.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170755_1294896'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170755_1294896/input0_results_240321_170755'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170755_1294896/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 6 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, SQ_WAVES, SQ_IFETCH, SQ_IFETCH_LEVEL, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170755_1294896/input0_results_240321_170755
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_IFETCH_LEVEL.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_LDS.txt
|-> [rocprof] RPL: on '240321_170755' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_LDS.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170755_1295097'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170755_1295097/input0_results_240321_170755'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170755_1295097/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_LDS, SQ_INST_LEVEL_LDS, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170755_1295097/input0_results_240321_170755
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_INST_LEVEL_LDS.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_SMEM.txt
|-> [rocprof] RPL: on '240321_170756' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_SMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170756_1295299'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170756_1295299/input0_results_240321_170756'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170756_1295299/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_SMEM, SQ_INST_LEVEL_SMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170756_1295299/input0_results_240321_170756
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_INST_LEVEL_SMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_VMEM.txt
|-> [rocprof] RPL: on '240321_170756' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_INST_LEVEL_VMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170756_1295501'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170756_1295501/input0_results_240321_170756'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170756_1295501/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_VMEM, SQ_INST_LEVEL_VMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170756_1295501/input0_results_240321_170756
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_INST_LEVEL_VMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/SQ_LEVEL_WAVES.txt
|-> [rocprof] RPL: on '240321_170757' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/SQ_LEVEL_WAVES.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170757_1295687'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170757_1295687/input0_results_240321_170757'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170757_1295687/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 9 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, CPC_ME1_BUSY_FOR_PACKET_DECODE, SQ_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQ_BUSY_CYCLES, SQ_LEVEL_WAVES, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170757_1295687/input0_results_240321_170757
|-> [rocprof] File 'tests/workloads/kernel/MI100/SQ_LEVEL_WAVES.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_0.txt
|-> [rocprof] RPL: on '240321_170757' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_0.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170757_1295888'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170757_1295888/input0_results_240321_170757'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170757_1295888/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 28 metrics
|-> [rocprof] SQ_CYCLES, SQ_BUSY_CYCLES, SQ_BUSY_CU_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQC_TC_INST_REQ, SQC_TC_DATA_READ_REQ, SQC_TC_DATA_WRITE_REQ, GRBM_COUNT, GRBM_GUI_ACTIVE, TCP_GATE_EN1_sum, TCP_GATE_EN2_sum, TCP_TD_TCP_STALL_CYCLES_sum, TCP_TCR_TCP_STALL_CYCLES_sum, TA_TA_BUSY_sum, TA_BUFFER_WAVEFRONTS_sum, TD_TD_BUSY_sum, TD_TC_STALL_sum, SPI_CSN_WINDOW_VALID, SPI_CSN_BUSY, CPC_CPC_STAT_BUSY, CPC_CPC_STAT_IDLE, CPF_CPF_STAT_BUSY, CPF_CPF_STAT_STALL, TCC_CYCLE_sum, TCC_BUSY_sum, TCC_PROBE_sum, TCC_PROBE_ALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170757_1295888/input0_results_240321_170757
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_0.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_1.txt
|-> [rocprof] RPL: on '240321_170758' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_1.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170758_1296091'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170758_1296091/input0_results_240321_170758'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170758_1296091/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 27 metrics
|-> [rocprof] SQC_TC_DATA_ATOMIC_REQ, SQC_TC_STALL, SQC_TC_REQ, SQC_DCACHE_REQ_READ_16, SQC_ICACHE_REQ, SQC_ICACHE_HITS, SQC_ICACHE_MISSES, SQC_ICACHE_MISSES_DUPLICATE, GRBM_SPI_BUSY, TCP_READ_TAGCONFLICT_STALL_CYCLES_sum, TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum, TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum, TCP_TA_TCP_STATE_READ_sum, TA_BUFFER_READ_WAVEFRONTS_sum, TA_BUFFER_WRITE_WAVEFRONTS_sum, TD_COALESCABLE_WAVEFRONT_sum, TD_LOAD_WAVEFRONT_sum, SPI_CSN_NUM_THREADGROUPS, SPI_CSN_WAVE, CPC_CPC_TCIU_BUSY, CPC_CPC_TCIU_IDLE, CPF_CPF_TCIU_BUSY, CPF_CPF_TCIU_STALL, TCC_NC_REQ_sum, TCC_UC_REQ_sum, TCC_CC_REQ_sum, TCC_RW_REQ_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170758_1296091/input0_results_240321_170758
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_1.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_10.txt
|-> [rocprof] RPL: on '240321_170758' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_10.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170758_1296294'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170758_1296294/input0_results_240321_170758'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170758_1296294/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 1 metrics
|-> [rocprof] TCC_EA_ATOMIC_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170758_1296294/input0_results_240321_170758
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_10.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_11.txt
|-> [rocprof] RPL: on '240321_170759' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_11.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170759_1296495'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170759_1296495/input0_results_240321_170759'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170759_1296495/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_ATOMIC[0], TCC_CYCLE[0], TCC_EA_ATOMIC[0], TCC_EA_ATOMIC_LEVEL[0], TCC_ATOMIC[1], TCC_CYCLE[1], TCC_EA_ATOMIC[1], TCC_EA_ATOMIC_LEVEL[1], TCC_ATOMIC[2], TCC_CYCLE[2], TCC_EA_ATOMIC[2], TCC_EA_ATOMIC_LEVEL[2], TCC_ATOMIC[3], TCC_CYCLE[3], TCC_EA_ATOMIC[3], TCC_EA_ATOMIC_LEVEL[3], TCC_ATOMIC[4], TCC_CYCLE[4], TCC_EA_ATOMIC[4], TCC_EA_ATOMIC_LEVEL[4], TCC_ATOMIC[5], TCC_CYCLE[5], TCC_EA_ATOMIC[5], TCC_EA_ATOMIC_LEVEL[5], TCC_ATOMIC[6], TCC_CYCLE[6], TCC_EA_ATOMIC[6], TCC_EA_ATOMIC_LEVEL[6], TCC_ATOMIC[7], TCC_CYCLE[7], TCC_EA_ATOMIC[7], TCC_EA_ATOMIC_LEVEL[7], TCC_ATOMIC[8], TCC_CYCLE[8], TCC_EA_ATOMIC[8], TCC_EA_ATOMIC_LEVEL[8], TCC_ATOMIC[9], TCC_CYCLE[9], TCC_EA_ATOMIC[9], TCC_EA_ATOMIC_LEVEL[9], TCC_ATOMIC[10], TCC_CYCLE[10], TCC_EA_ATOMIC[10], TCC_EA_ATOMIC_LEVEL[10], TCC_ATOMIC[11], TCC_CYCLE[11], TCC_EA_ATOMIC[11], TCC_EA_ATOMIC_LEVEL[11], TCC_ATOMIC[12], TCC_CYCLE[12], TCC_EA_ATOMIC[12], TCC_EA_ATOMIC_LEVEL[12], TCC_ATOMIC[13], TCC_CYCLE[13], TCC_EA_ATOMIC[13], TCC_EA_ATOMIC_LEVEL[13], TCC_ATOMIC[14], TCC_CYCLE[14], TCC_EA_ATOMIC[14], TCC_EA_ATOMIC_LEVEL[14], TCC_ATOMIC[15], TCC_CYCLE[15], TCC_EA_ATOMIC[15], TCC_EA_ATOMIC_LEVEL[15], TCC_ATOMIC[16], TCC_CYCLE[16], TCC_EA_ATOMIC[16], TCC_EA_ATOMIC_LEVEL[16], TCC_ATOMIC[17], TCC_CYCLE[17], TCC_EA_ATOMIC[17], TCC_EA_ATOMIC_LEVEL[17], TCC_ATOMIC[18], TCC_CYCLE[18], TCC_EA_ATOMIC[18], TCC_EA_ATOMIC_LEVEL[18], TCC_ATOMIC[19], TCC_CYCLE[19], TCC_EA_ATOMIC[19], TCC_EA_ATOMIC_LEVEL[19], TCC_ATOMIC[20], TCC_CYCLE[20], TCC_EA_ATOMIC[20], TCC_EA_ATOMIC_LEVEL[20], TCC_ATOMIC[21], TCC_CYCLE[21], TCC_EA_ATOMIC[21], TCC_EA_ATOMIC_LEVEL[21], TCC_ATOMIC[22], TCC_CYCLE[22], TCC_EA_ATOMIC[22], TCC_EA_ATOMIC_LEVEL[22], TCC_ATOMIC[23], TCC_CYCLE[23], TCC_EA_ATOMIC[23], TCC_EA_ATOMIC_LEVEL[23], TCC_ATOMIC[24], TCC_CYCLE[24], TCC_EA_ATOMIC[24], TCC_EA_ATOMIC_LEVEL[24], TCC_ATOMIC[25], TCC_CYCLE[25], TCC_EA_ATOMIC[25], TCC_EA_ATOMIC_LEVEL[25], TCC_ATOMIC[26], TCC_CYCLE[26], TCC_EA_ATOMIC[26], TCC_EA_ATOMIC_LEVEL[26], TCC_ATOMIC[27], TCC_CYCLE[27], TCC_EA_ATOMIC[27], TCC_EA_ATOMIC_LEVEL[27], TCC_ATOMIC[28], TCC_CYCLE[28], TCC_EA_ATOMIC[28], TCC_EA_ATOMIC_LEVEL[28], TCC_ATOMIC[29], TCC_CYCLE[29], TCC_EA_ATOMIC[29], TCC_EA_ATOMIC_LEVEL[29], TCC_ATOMIC[30], TCC_CYCLE[30], TCC_EA_ATOMIC[30], TCC_EA_ATOMIC_LEVEL[30], TCC_ATOMIC[31], TCC_CYCLE[31], TCC_EA_ATOMIC[31], TCC_EA_ATOMIC_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170759_1296495/input0_results_240321_170759
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_11.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_12.txt
|-> [rocprof] RPL: on '240321_170759' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_12.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170759_1296697'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170759_1296697/input0_results_240321_170759'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170759_1296697/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ[0], TCC_EA_RDREQ_32B[0], TCC_EA_RDREQ_DRAM_CREDIT_STALL[0], TCC_EA_RDREQ_GMI_CREDIT_STALL[0], TCC_EA_RDREQ[1], TCC_EA_RDREQ_32B[1], TCC_EA_RDREQ_DRAM_CREDIT_STALL[1], TCC_EA_RDREQ_GMI_CREDIT_STALL[1], TCC_EA_RDREQ[2], TCC_EA_RDREQ_32B[2], TCC_EA_RDREQ_DRAM_CREDIT_STALL[2], TCC_EA_RDREQ_GMI_CREDIT_STALL[2], TCC_EA_RDREQ[3], TCC_EA_RDREQ_32B[3], TCC_EA_RDREQ_DRAM_CREDIT_STALL[3], TCC_EA_RDREQ_GMI_CREDIT_STALL[3], TCC_EA_RDREQ[4], TCC_EA_RDREQ_32B[4], TCC_EA_RDREQ_DRAM_CREDIT_STALL[4], TCC_EA_RDREQ_GMI_CREDIT_STALL[4], TCC_EA_RDREQ[5], TCC_EA_RDREQ_32B[5], TCC_EA_RDREQ_DRAM_CREDIT_STALL[5], TCC_EA_RDREQ_GMI_CREDIT_STALL[5], TCC_EA_RDREQ[6], TCC_EA_RDREQ_32B[6], TCC_EA_RDREQ_DRAM_CREDIT_STALL[6], TCC_EA_RDREQ_GMI_CREDIT_STALL[6], TCC_EA_RDREQ[7], TCC_EA_RDREQ_32B[7], TCC_EA_RDREQ_DRAM_CREDIT_STALL[7], TCC_EA_RDREQ_GMI_CREDIT_STALL[7], TCC_EA_RDREQ[8], TCC_EA_RDREQ_32B[8], TCC_EA_RDREQ_DRAM_CREDIT_STALL[8], TCC_EA_RDREQ_GMI_CREDIT_STALL[8], TCC_EA_RDREQ[9], TCC_EA_RDREQ_32B[9], TCC_EA_RDREQ_DRAM_CREDIT_STALL[9], TCC_EA_RDREQ_GMI_CREDIT_STALL[9], TCC_EA_RDREQ[10], TCC_EA_RDREQ_32B[10], TCC_EA_RDREQ_DRAM_CREDIT_STALL[10], TCC_EA_RDREQ_GMI_CREDIT_STALL[10], TCC_EA_RDREQ[11], TCC_EA_RDREQ_32B[11], TCC_EA_RDREQ_DRAM_CREDIT_STALL[11], TCC_EA_RDREQ_GMI_CREDIT_STALL[11], TCC_EA_RDREQ[12], TCC_EA_RDREQ_32B[12], TCC_EA_RDREQ_DRAM_CREDIT_STALL[12], TCC_EA_RDREQ_GMI_CREDIT_STALL[12], TCC_EA_RDREQ[13], TCC_EA_RDREQ_32B[13], TCC_EA_RDREQ_DRAM_CREDIT_STALL[13], TCC_EA_RDREQ_GMI_CREDIT_STALL[13], TCC_EA_RDREQ[14], TCC_EA_RDREQ_32B[14], TCC_EA_RDREQ_DRAM_CREDIT_STALL[14], TCC_EA_RDREQ_GMI_CREDIT_STALL[14], TCC_EA_RDREQ[15], TCC_EA_RDREQ_32B[15], TCC_EA_RDREQ_DRAM_CREDIT_STALL[15], TCC_EA_RDREQ_GMI_CREDIT_STALL[15], TCC_EA_RDREQ[16], TCC_EA_RDREQ_32B[16], TCC_EA_RDREQ_DRAM_CREDIT_STALL[16], TCC_EA_RDREQ_GMI_CREDIT_STALL[16], TCC_EA_RDREQ[17], TCC_EA_RDREQ_32B[17], TCC_EA_RDREQ_DRAM_CREDIT_STALL[17], TCC_EA_RDREQ_GMI_CREDIT_STALL[17], TCC_EA_RDREQ[18], TCC_EA_RDREQ_32B[18], TCC_EA_RDREQ_DRAM_CREDIT_STALL[18], TCC_EA_RDREQ_GMI_CREDIT_STALL[18], TCC_EA_RDREQ[19], TCC_EA_RDREQ_32B[19], TCC_EA_RDREQ_DRAM_CREDIT_STALL[19], TCC_EA_RDREQ_GMI_CREDIT_STALL[19], TCC_EA_RDREQ[20], TCC_EA_RDREQ_32B[20], TCC_EA_RDREQ_DRAM_CREDIT_STALL[20], TCC_EA_RDREQ_GMI_CREDIT_STALL[20], TCC_EA_RDREQ[21], TCC_EA_RDREQ_32B[21], TCC_EA_RDREQ_DRAM_CREDIT_STALL[21], TCC_EA_RDREQ_GMI_CREDIT_STALL[21], TCC_EA_RDREQ[22], TCC_EA_RDREQ_32B[22], TCC_EA_RDREQ_DRAM_CREDIT_STALL[22], TCC_EA_RDREQ_GMI_CREDIT_STALL[22], TCC_EA_RDREQ[23], TCC_EA_RDREQ_32B[23], TCC_EA_RDREQ_DRAM_CREDIT_STALL[23], TCC_EA_RDREQ_GMI_CREDIT_STALL[23], TCC_EA_RDREQ[24], TCC_EA_RDREQ_32B[24], TCC_EA_RDREQ_DRAM_CREDIT_STALL[24], TCC_EA_RDREQ_GMI_CREDIT_STALL[24], TCC_EA_RDREQ[25], TCC_EA_RDREQ_32B[25], TCC_EA_RDREQ_DRAM_CREDIT_STALL[25], TCC_EA_RDREQ_GMI_CREDIT_STALL[25], TCC_EA_RDREQ[26], TCC_EA_RDREQ_32B[26], TCC_EA_RDREQ_DRAM_CREDIT_STALL[26], TCC_EA_RDREQ_GMI_CREDIT_STALL[26], TCC_EA_RDREQ[27], TCC_EA_RDREQ_32B[27], TCC_EA_RDREQ_DRAM_CREDIT_STALL[27], TCC_EA_RDREQ_GMI_CREDIT_STALL[27], TCC_EA_RDREQ[28], TCC_EA_RDREQ_32B[28], TCC_EA_RDREQ_DRAM_CREDIT_STALL[28], TCC_EA_RDREQ_GMI_CREDIT_STALL[28], TCC_EA_RDREQ[29], TCC_EA_RDREQ_32B[29], TCC_EA_RDREQ_DRAM_CREDIT_STALL[29], TCC_EA_RDREQ_GMI_CREDIT_STALL[29], TCC_EA_RDREQ[30], TCC_EA_RDREQ_32B[30], TCC_EA_RDREQ_DRAM_CREDIT_STALL[30], TCC_EA_RDREQ_GMI_CREDIT_STALL[30], TCC_EA_RDREQ[31], TCC_EA_RDREQ_32B[31], TCC_EA_RDREQ_DRAM_CREDIT_STALL[31], TCC_EA_RDREQ_GMI_CREDIT_STALL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170759_1296697/input0_results_240321_170759
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_12.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_13.txt
|-> [rocprof] RPL: on '240321_170800' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_13.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170800_1296901'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170800_1296901/input0_results_240321_170800'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170800_1296901/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ_IO_CREDIT_STALL[0], TCC_EA_RDREQ_LEVEL[0], TCC_EA_WRREQ[0], TCC_EA_WRREQ_64B[0], TCC_EA_RDREQ_IO_CREDIT_STALL[1], TCC_EA_RDREQ_LEVEL[1], TCC_EA_WRREQ[1], TCC_EA_WRREQ_64B[1], TCC_EA_RDREQ_IO_CREDIT_STALL[2], TCC_EA_RDREQ_LEVEL[2], TCC_EA_WRREQ[2], TCC_EA_WRREQ_64B[2], TCC_EA_RDREQ_IO_CREDIT_STALL[3], TCC_EA_RDREQ_LEVEL[3], TCC_EA_WRREQ[3], TCC_EA_WRREQ_64B[3], TCC_EA_RDREQ_IO_CREDIT_STALL[4], TCC_EA_RDREQ_LEVEL[4], TCC_EA_WRREQ[4], TCC_EA_WRREQ_64B[4], TCC_EA_RDREQ_IO_CREDIT_STALL[5], TCC_EA_RDREQ_LEVEL[5], TCC_EA_WRREQ[5], TCC_EA_WRREQ_64B[5], TCC_EA_RDREQ_IO_CREDIT_STALL[6], TCC_EA_RDREQ_LEVEL[6], TCC_EA_WRREQ[6], TCC_EA_WRREQ_64B[6], TCC_EA_RDREQ_IO_CREDIT_STALL[7], TCC_EA_RDREQ_LEVEL[7], TCC_EA_WRREQ[7], TCC_EA_WRREQ_64B[7], TCC_EA_RDREQ_IO_CREDIT_STALL[8], TCC_EA_RDREQ_LEVEL[8], TCC_EA_WRREQ[8], TCC_EA_WRREQ_64B[8], TCC_EA_RDREQ_IO_CREDIT_STALL[9], TCC_EA_RDREQ_LEVEL[9], TCC_EA_WRREQ[9], TCC_EA_WRREQ_64B[9], TCC_EA_RDREQ_IO_CREDIT_STALL[10], TCC_EA_RDREQ_LEVEL[10], TCC_EA_WRREQ[10], TCC_EA_WRREQ_64B[10], TCC_EA_RDREQ_IO_CREDIT_STALL[11], TCC_EA_RDREQ_LEVEL[11], TCC_EA_WRREQ[11], TCC_EA_WRREQ_64B[11], TCC_EA_RDREQ_IO_CREDIT_STALL[12], TCC_EA_RDREQ_LEVEL[12], TCC_EA_WRREQ[12], TCC_EA_WRREQ_64B[12], TCC_EA_RDREQ_IO_CREDIT_STALL[13], TCC_EA_RDREQ_LEVEL[13], TCC_EA_WRREQ[13], TCC_EA_WRREQ_64B[13], TCC_EA_RDREQ_IO_CREDIT_STALL[14], TCC_EA_RDREQ_LEVEL[14], TCC_EA_WRREQ[14], TCC_EA_WRREQ_64B[14], TCC_EA_RDREQ_IO_CREDIT_STALL[15], TCC_EA_RDREQ_LEVEL[15], TCC_EA_WRREQ[15], TCC_EA_WRREQ_64B[15], TCC_EA_RDREQ_IO_CREDIT_STALL[16], TCC_EA_RDREQ_LEVEL[16], TCC_EA_WRREQ[16], TCC_EA_WRREQ_64B[16], TCC_EA_RDREQ_IO_CREDIT_STALL[17], TCC_EA_RDREQ_LEVEL[17], TCC_EA_WRREQ[17], TCC_EA_WRREQ_64B[17], TCC_EA_RDREQ_IO_CREDIT_STALL[18], TCC_EA_RDREQ_LEVEL[18], TCC_EA_WRREQ[18], TCC_EA_WRREQ_64B[18], TCC_EA_RDREQ_IO_CREDIT_STALL[19], TCC_EA_RDREQ_LEVEL[19], TCC_EA_WRREQ[19], TCC_EA_WRREQ_64B[19], TCC_EA_RDREQ_IO_CREDIT_STALL[20], TCC_EA_RDREQ_LEVEL[20], TCC_EA_WRREQ[20], TCC_EA_WRREQ_64B[20], TCC_EA_RDREQ_IO_CREDIT_STALL[21], TCC_EA_RDREQ_LEVEL[21], TCC_EA_WRREQ[21], TCC_EA_WRREQ_64B[21], TCC_EA_RDREQ_IO_CREDIT_STALL[22], TCC_EA_RDREQ_LEVEL[22], TCC_EA_WRREQ[22], TCC_EA_WRREQ_64B[22], TCC_EA_RDREQ_IO_CREDIT_STALL[23], TCC_EA_RDREQ_LEVEL[23], TCC_EA_WRREQ[23], TCC_EA_WRREQ_64B[23], TCC_EA_RDREQ_IO_CREDIT_STALL[24], TCC_EA_RDREQ_LEVEL[24], TCC_EA_WRREQ[24], TCC_EA_WRREQ_64B[24], TCC_EA_RDREQ_IO_CREDIT_STALL[25], TCC_EA_RDREQ_LEVEL[25], TCC_EA_WRREQ[25], TCC_EA_WRREQ_64B[25], TCC_EA_RDREQ_IO_CREDIT_STALL[26], TCC_EA_RDREQ_LEVEL[26], TCC_EA_WRREQ[26], TCC_EA_WRREQ_64B[26], TCC_EA_RDREQ_IO_CREDIT_STALL[27], TCC_EA_RDREQ_LEVEL[27], TCC_EA_WRREQ[27], TCC_EA_WRREQ_64B[27], TCC_EA_RDREQ_IO_CREDIT_STALL[28], TCC_EA_RDREQ_LEVEL[28], TCC_EA_WRREQ[28], TCC_EA_WRREQ_64B[28], TCC_EA_RDREQ_IO_CREDIT_STALL[29], TCC_EA_RDREQ_LEVEL[29], TCC_EA_WRREQ[29], TCC_EA_WRREQ_64B[29], TCC_EA_RDREQ_IO_CREDIT_STALL[30], TCC_EA_RDREQ_LEVEL[30], TCC_EA_WRREQ[30], TCC_EA_WRREQ_64B[30], TCC_EA_RDREQ_IO_CREDIT_STALL[31], TCC_EA_RDREQ_LEVEL[31], TCC_EA_WRREQ[31], TCC_EA_WRREQ_64B[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170800_1296901/input0_results_240321_170800
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_13.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_14.txt
|-> [rocprof] RPL: on '240321_170801' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_14.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170801_1297099'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170801_1297099/input0_results_240321_170801'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170801_1297099/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_WRREQ_DRAM_CREDIT_STALL[0], TCC_EA_WRREQ_GMI_CREDIT_STALL[0], TCC_EA_WRREQ_IO_CREDIT_STALL[0], TCC_EA_WRREQ_LEVEL[0], TCC_EA_WRREQ_DRAM_CREDIT_STALL[1], TCC_EA_WRREQ_GMI_CREDIT_STALL[1], TCC_EA_WRREQ_IO_CREDIT_STALL[1], TCC_EA_WRREQ_LEVEL[1], TCC_EA_WRREQ_DRAM_CREDIT_STALL[2], TCC_EA_WRREQ_GMI_CREDIT_STALL[2], TCC_EA_WRREQ_IO_CREDIT_STALL[2], TCC_EA_WRREQ_LEVEL[2], TCC_EA_WRREQ_DRAM_CREDIT_STALL[3], TCC_EA_WRREQ_GMI_CREDIT_STALL[3], TCC_EA_WRREQ_IO_CREDIT_STALL[3], TCC_EA_WRREQ_LEVEL[3], TCC_EA_WRREQ_DRAM_CREDIT_STALL[4], TCC_EA_WRREQ_GMI_CREDIT_STALL[4], TCC_EA_WRREQ_IO_CREDIT_STALL[4], TCC_EA_WRREQ_LEVEL[4], TCC_EA_WRREQ_DRAM_CREDIT_STALL[5], TCC_EA_WRREQ_GMI_CREDIT_STALL[5], TCC_EA_WRREQ_IO_CREDIT_STALL[5], TCC_EA_WRREQ_LEVEL[5], TCC_EA_WRREQ_DRAM_CREDIT_STALL[6], TCC_EA_WRREQ_GMI_CREDIT_STALL[6], TCC_EA_WRREQ_IO_CREDIT_STALL[6], TCC_EA_WRREQ_LEVEL[6], TCC_EA_WRREQ_DRAM_CREDIT_STALL[7], TCC_EA_WRREQ_GMI_CREDIT_STALL[7], TCC_EA_WRREQ_IO_CREDIT_STALL[7], TCC_EA_WRREQ_LEVEL[7], TCC_EA_WRREQ_DRAM_CREDIT_STALL[8], TCC_EA_WRREQ_GMI_CREDIT_STALL[8], TCC_EA_WRREQ_IO_CREDIT_STALL[8], TCC_EA_WRREQ_LEVEL[8], TCC_EA_WRREQ_DRAM_CREDIT_STALL[9], TCC_EA_WRREQ_GMI_CREDIT_STALL[9], TCC_EA_WRREQ_IO_CREDIT_STALL[9], TCC_EA_WRREQ_LEVEL[9], TCC_EA_WRREQ_DRAM_CREDIT_STALL[10], TCC_EA_WRREQ_GMI_CREDIT_STALL[10], TCC_EA_WRREQ_IO_CREDIT_STALL[10], TCC_EA_WRREQ_LEVEL[10], TCC_EA_WRREQ_DRAM_CREDIT_STALL[11], TCC_EA_WRREQ_GMI_CREDIT_STALL[11], TCC_EA_WRREQ_IO_CREDIT_STALL[11], TCC_EA_WRREQ_LEVEL[11], TCC_EA_WRREQ_DRAM_CREDIT_STALL[12], TCC_EA_WRREQ_GMI_CREDIT_STALL[12], TCC_EA_WRREQ_IO_CREDIT_STALL[12], TCC_EA_WRREQ_LEVEL[12], TCC_EA_WRREQ_DRAM_CREDIT_STALL[13], TCC_EA_WRREQ_GMI_CREDIT_STALL[13], TCC_EA_WRREQ_IO_CREDIT_STALL[13], TCC_EA_WRREQ_LEVEL[13], TCC_EA_WRREQ_DRAM_CREDIT_STALL[14], TCC_EA_WRREQ_GMI_CREDIT_STALL[14], TCC_EA_WRREQ_IO_CREDIT_STALL[14], TCC_EA_WRREQ_LEVEL[14], TCC_EA_WRREQ_DRAM_CREDIT_STALL[15], TCC_EA_WRREQ_GMI_CREDIT_STALL[15], TCC_EA_WRREQ_IO_CREDIT_STALL[15], TCC_EA_WRREQ_LEVEL[15], TCC_EA_WRREQ_DRAM_CREDIT_STALL[16], TCC_EA_WRREQ_GMI_CREDIT_STALL[16], TCC_EA_WRREQ_IO_CREDIT_STALL[16], TCC_EA_WRREQ_LEVEL[16], TCC_EA_WRREQ_DRAM_CREDIT_STALL[17], TCC_EA_WRREQ_GMI_CREDIT_STALL[17], TCC_EA_WRREQ_IO_CREDIT_STALL[17], TCC_EA_WRREQ_LEVEL[17], TCC_EA_WRREQ_DRAM_CREDIT_STALL[18], TCC_EA_WRREQ_GMI_CREDIT_STALL[18], TCC_EA_WRREQ_IO_CREDIT_STALL[18], TCC_EA_WRREQ_LEVEL[18], TCC_EA_WRREQ_DRAM_CREDIT_STALL[19], TCC_EA_WRREQ_GMI_CREDIT_STALL[19], TCC_EA_WRREQ_IO_CREDIT_STALL[19], TCC_EA_WRREQ_LEVEL[19], TCC_EA_WRREQ_DRAM_CREDIT_STALL[20], TCC_EA_WRREQ_GMI_CREDIT_STALL[20], TCC_EA_WRREQ_IO_CREDIT_STALL[20], TCC_EA_WRREQ_LEVEL[20], TCC_EA_WRREQ_DRAM_CREDIT_STALL[21], TCC_EA_WRREQ_GMI_CREDIT_STALL[21], TCC_EA_WRREQ_IO_CREDIT_STALL[21], TCC_EA_WRREQ_LEVEL[21], TCC_EA_WRREQ_DRAM_CREDIT_STALL[22], TCC_EA_WRREQ_GMI_CREDIT_STALL[22], TCC_EA_WRREQ_IO_CREDIT_STALL[22], TCC_EA_WRREQ_LEVEL[22], TCC_EA_WRREQ_DRAM_CREDIT_STALL[23], TCC_EA_WRREQ_GMI_CREDIT_STALL[23], TCC_EA_WRREQ_IO_CREDIT_STALL[23], TCC_EA_WRREQ_LEVEL[23], TCC_EA_WRREQ_DRAM_CREDIT_STALL[24], TCC_EA_WRREQ_GMI_CREDIT_STALL[24], TCC_EA_WRREQ_IO_CREDIT_STALL[24], TCC_EA_WRREQ_LEVEL[24], TCC_EA_WRREQ_DRAM_CREDIT_STALL[25], TCC_EA_WRREQ_GMI_CREDIT_STALL[25], TCC_EA_WRREQ_IO_CREDIT_STALL[25], TCC_EA_WRREQ_LEVEL[25], TCC_EA_WRREQ_DRAM_CREDIT_STALL[26], TCC_EA_WRREQ_GMI_CREDIT_STALL[26], TCC_EA_WRREQ_IO_CREDIT_STALL[26], TCC_EA_WRREQ_LEVEL[26], TCC_EA_WRREQ_DRAM_CREDIT_STALL[27], TCC_EA_WRREQ_GMI_CREDIT_STALL[27], TCC_EA_WRREQ_IO_CREDIT_STALL[27], TCC_EA_WRREQ_LEVEL[27], TCC_EA_WRREQ_DRAM_CREDIT_STALL[28], TCC_EA_WRREQ_GMI_CREDIT_STALL[28], TCC_EA_WRREQ_IO_CREDIT_STALL[28], TCC_EA_WRREQ_LEVEL[28], TCC_EA_WRREQ_DRAM_CREDIT_STALL[29], TCC_EA_WRREQ_GMI_CREDIT_STALL[29], TCC_EA_WRREQ_IO_CREDIT_STALL[29], TCC_EA_WRREQ_LEVEL[29], TCC_EA_WRREQ_DRAM_CREDIT_STALL[30], TCC_EA_WRREQ_GMI_CREDIT_STALL[30], TCC_EA_WRREQ_IO_CREDIT_STALL[30], TCC_EA_WRREQ_LEVEL[30], TCC_EA_WRREQ_DRAM_CREDIT_STALL[31], TCC_EA_WRREQ_GMI_CREDIT_STALL[31], TCC_EA_WRREQ_IO_CREDIT_STALL[31], TCC_EA_WRREQ_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170801_1297099/input0_results_240321_170801
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_14.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_15.txt
|-> [rocprof] RPL: on '240321_170801' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_15.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170801_1297303'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170801_1297303/input0_results_240321_170801'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170801_1297303/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_HIT[0], TCC_MISS[0], TCC_READ[0], TCC_REQ[0], TCC_HIT[1], TCC_MISS[1], TCC_READ[1], TCC_REQ[1], TCC_HIT[2], TCC_MISS[2], TCC_READ[2], TCC_REQ[2], TCC_HIT[3], TCC_MISS[3], TCC_READ[3], TCC_REQ[3], TCC_HIT[4], TCC_MISS[4], TCC_READ[4], TCC_REQ[4], TCC_HIT[5], TCC_MISS[5], TCC_READ[5], TCC_REQ[5], TCC_HIT[6], TCC_MISS[6], TCC_READ[6], TCC_REQ[6], TCC_HIT[7], TCC_MISS[7], TCC_READ[7], TCC_REQ[7], TCC_HIT[8], TCC_MISS[8], TCC_READ[8], TCC_REQ[8], TCC_HIT[9], TCC_MISS[9], TCC_READ[9], TCC_REQ[9], TCC_HIT[10], TCC_MISS[10], TCC_READ[10], TCC_REQ[10], TCC_HIT[11], TCC_MISS[11], TCC_READ[11], TCC_REQ[11], TCC_HIT[12], TCC_MISS[12], TCC_READ[12], TCC_REQ[12], TCC_HIT[13], TCC_MISS[13], TCC_READ[13], TCC_REQ[13], TCC_HIT[14], TCC_MISS[14], TCC_READ[14], TCC_REQ[14], TCC_HIT[15], TCC_MISS[15], TCC_READ[15], TCC_REQ[15], TCC_HIT[16], TCC_MISS[16], TCC_READ[16], TCC_REQ[16], TCC_HIT[17], TCC_MISS[17], TCC_READ[17], TCC_REQ[17], TCC_HIT[18], TCC_MISS[18], TCC_READ[18], TCC_REQ[18], TCC_HIT[19], TCC_MISS[19], TCC_READ[19], TCC_REQ[19], TCC_HIT[20], TCC_MISS[20], TCC_READ[20], TCC_REQ[20], TCC_HIT[21], TCC_MISS[21], TCC_READ[21], TCC_REQ[21], TCC_HIT[22], TCC_MISS[22], TCC_READ[22], TCC_REQ[22], TCC_HIT[23], TCC_MISS[23], TCC_READ[23], TCC_REQ[23], TCC_HIT[24], TCC_MISS[24], TCC_READ[24], TCC_REQ[24], TCC_HIT[25], TCC_MISS[25], TCC_READ[25], TCC_REQ[25], TCC_HIT[26], TCC_MISS[26], TCC_READ[26], TCC_REQ[26], TCC_HIT[27], TCC_MISS[27], TCC_READ[27], TCC_REQ[27], TCC_HIT[28], TCC_MISS[28], TCC_READ[28], TCC_REQ[28], TCC_HIT[29], TCC_MISS[29], TCC_READ[29], TCC_REQ[29], TCC_HIT[30], TCC_MISS[30], TCC_READ[30], TCC_REQ[30], TCC_HIT[31], TCC_MISS[31], TCC_READ[31], TCC_REQ[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170801_1297303/input0_results_240321_170801
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_15.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_16.txt
|-> [rocprof] RPL: on '240321_170802' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_16.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170802_1297506'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170802_1297506/input0_results_240321_170802'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170802_1297506/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 96 metrics
|-> [rocprof] TCC_RW_REQ[0], TCC_TOO_MANY_EA_WRREQS_STALL[0], TCC_WRITE[0], TCC_RW_REQ[1], TCC_TOO_MANY_EA_WRREQS_STALL[1], TCC_WRITE[1], TCC_RW_REQ[2], TCC_TOO_MANY_EA_WRREQS_STALL[2], TCC_WRITE[2], TCC_RW_REQ[3], TCC_TOO_MANY_EA_WRREQS_STALL[3], TCC_WRITE[3], TCC_RW_REQ[4], TCC_TOO_MANY_EA_WRREQS_STALL[4], TCC_WRITE[4], TCC_RW_REQ[5], TCC_TOO_MANY_EA_WRREQS_STALL[5], TCC_WRITE[5], TCC_RW_REQ[6], TCC_TOO_MANY_EA_WRREQS_STALL[6], TCC_WRITE[6], TCC_RW_REQ[7], TCC_TOO_MANY_EA_WRREQS_STALL[7], TCC_WRITE[7], TCC_RW_REQ[8], TCC_TOO_MANY_EA_WRREQS_STALL[8], TCC_WRITE[8], TCC_RW_REQ[9], TCC_TOO_MANY_EA_WRREQS_STALL[9], TCC_WRITE[9], TCC_RW_REQ[10], TCC_TOO_MANY_EA_WRREQS_STALL[10], TCC_WRITE[10], TCC_RW_REQ[11], TCC_TOO_MANY_EA_WRREQS_STALL[11], TCC_WRITE[11], TCC_RW_REQ[12], TCC_TOO_MANY_EA_WRREQS_STALL[12], TCC_WRITE[12], TCC_RW_REQ[13], TCC_TOO_MANY_EA_WRREQS_STALL[13], TCC_WRITE[13], TCC_RW_REQ[14], TCC_TOO_MANY_EA_WRREQS_STALL[14], TCC_WRITE[14], TCC_RW_REQ[15], TCC_TOO_MANY_EA_WRREQS_STALL[15], TCC_WRITE[15], TCC_RW_REQ[16], TCC_TOO_MANY_EA_WRREQS_STALL[16], TCC_WRITE[16], TCC_RW_REQ[17], TCC_TOO_MANY_EA_WRREQS_STALL[17], TCC_WRITE[17], TCC_RW_REQ[18], TCC_TOO_MANY_EA_WRREQS_STALL[18], TCC_WRITE[18], TCC_RW_REQ[19], TCC_TOO_MANY_EA_WRREQS_STALL[19], TCC_WRITE[19], TCC_RW_REQ[20], TCC_TOO_MANY_EA_WRREQS_STALL[20], TCC_WRITE[20], TCC_RW_REQ[21], TCC_TOO_MANY_EA_WRREQS_STALL[21], TCC_WRITE[21], TCC_RW_REQ[22], TCC_TOO_MANY_EA_WRREQS_STALL[22], TCC_WRITE[22], TCC_RW_REQ[23], TCC_TOO_MANY_EA_WRREQS_STALL[23], TCC_WRITE[23], TCC_RW_REQ[24], TCC_TOO_MANY_EA_WRREQS_STALL[24], TCC_WRITE[24], TCC_RW_REQ[25], TCC_TOO_MANY_EA_WRREQS_STALL[25], TCC_WRITE[25], TCC_RW_REQ[26], TCC_TOO_MANY_EA_WRREQS_STALL[26], TCC_WRITE[26], TCC_RW_REQ[27], TCC_TOO_MANY_EA_WRREQS_STALL[27], TCC_WRITE[27], TCC_RW_REQ[28], TCC_TOO_MANY_EA_WRREQS_STALL[28], TCC_WRITE[28], TCC_RW_REQ[29], TCC_TOO_MANY_EA_WRREQS_STALL[29], TCC_WRITE[29], TCC_RW_REQ[30], TCC_TOO_MANY_EA_WRREQS_STALL[30], TCC_WRITE[30], TCC_RW_REQ[31], TCC_TOO_MANY_EA_WRREQS_STALL[31], TCC_WRITE[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170802_1297506/input0_results_240321_170802
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_16.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_2.txt
|-> [rocprof] RPL: on '240321_170803' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_2.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170803_1297703'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170803_1297703/input0_results_240321_170803'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170803_1297703/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 26 metrics
|-> [rocprof] SQC_DCACHE_INPUT_VALID_READYB, SQC_DCACHE_ATOMIC, SQC_DCACHE_REQ_READ_8, SQC_DCACHE_REQ, SQC_DCACHE_HITS, SQC_DCACHE_MISSES, SQC_DCACHE_MISSES_DUPLICATE, SQC_DCACHE_REQ_READ_1, TCP_VOLATILE_sum, TCP_TOTAL_ACCESSES_sum, TCP_TOTAL_READ_sum, TCP_TOTAL_WRITE_sum, TA_BUFFER_ATOMIC_WAVEFRONTS_sum, TA_BUFFER_TOTAL_CYCLES_sum, TD_ATOMIC_WAVEFRONT_sum, TD_STORE_WAVEFRONT_sum, SPI_RA_REQ_NO_ALLOC, SPI_RA_REQ_NO_ALLOC_CSN, CPC_CPC_STAT_STALL, CPC_UTCL1_STALL_ON_TRANSLATION, CPF_CPF_STAT_IDLE, CPF_CPF_TCIU_IDLE, TCC_REQ_sum, TCC_STREAMING_REQ_sum, TCC_HIT_sum, TCC_MISS_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170803_1297703/input0_results_240321_170803
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_2.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_3.txt
|-> [rocprof] RPL: on '240321_170803' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_3.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170803_1297890'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170803_1297890/input0_results_240321_170803'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170803_1297890/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 23 metrics
|-> [rocprof] SQC_DCACHE_REQ_READ_2, SQC_DCACHE_REQ_READ_4, SQ_INSTS_VMEM_WR, SQ_INSTS_VMEM_RD, SQ_INSTS_VMEM, SQ_INSTS_SALU, SQ_INSTS_VSKIPPED, SQ_INSTS_SMEM, TCP_TOTAL_ATOMIC_WITH_RET_sum, TCP_TOTAL_ATOMIC_WITHOUT_RET_sum, TCP_TOTAL_WRITEBACK_INVALIDATES_sum, TCP_TOTAL_CACHE_ACCESSES_sum, TA_BUFFER_COALESCED_READ_CYCLES_sum, TA_BUFFER_COALESCED_WRITE_CYCLES_sum, SPI_RA_RES_STALL_CSN, SPI_RA_TMP_STALL_CSN, CPC_CPC_UTCL2IU_BUSY, CPC_CPC_UTCL2IU_IDLE, CPF_CMP_UTCL1_STALL_ON_TRANSLATION, TCC_READ_sum, TCC_WRITE_sum, TCC_ATOMIC_sum, TCC_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170803_1297890/input0_results_240321_170803
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_3.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_4.txt
|-> [rocprof] RPL: on '240321_170804' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_4.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170804_1298091'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170804_1298091/input0_results_240321_170804'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170804_1298091/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 22 metrics
|-> [rocprof] SQ_INSTS_FLAT, SQ_INSTS_LDS, SQ_INSTS_GDS, SQ_INSTS_EXP_GDS, SQ_INSTS_BRANCH, SQ_INSTS_SENDMSG, SQ_INSTS, SQ_WAIT_ANY, TCP_UTCL1_TRANSLATION_MISS_sum, TCP_UTCL1_TRANSLATION_HIT_sum, TCP_UTCL1_PERMISSION_MISS_sum, TCP_UTCL1_REQUEST_sum, TA_ADDR_STALLED_BY_TC_CYCLES_sum, TA_TOTAL_WAVEFRONTS_sum, SPI_RA_WAVE_SIMD_FULL_CSN, SPI_RA_VGPR_SIMD_FULL_CSN, CPC_CPC_UTCL2IU_STALL, CPC_ME1_BUSY_FOR_PACKET_DECODE, TCC_EA_WRREQ_sum, TCC_EA_WRREQ_64B_sum, TCC_EA_WR_UNCACHED_32B_sum, TCC_EA_WRREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170804_1298091/input0_results_240321_170804
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_4.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_5.txt
|-> [rocprof] RPL: on '240321_170804' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_5.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170804_1298276'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170804_1298276/input0_results_240321_170804'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170804_1298276/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 21 metrics
|-> [rocprof] SQ_WAIT_INST_ANY, SQ_ACTIVE_INST_ANY, SQ_INSTS_VALU, SQ_ACTIVE_INST_VMEM, SQ_ACTIVE_INST_LDS, SQ_ACTIVE_INST_VALU, SQ_ACTIVE_INST_SCA, SQ_ACTIVE_INST_EXP_GDS, TCP_TCP_LATENCY_sum, TCP_TCC_READ_REQ_LATENCY_sum, TCP_TCC_WRITE_REQ_LATENCY_sum, TCP_TCC_READ_REQ_sum, TA_ADDR_STALLED_BY_TD_CYCLES_sum, TA_DATA_STALLED_BY_TC_CYCLES_sum, SPI_RA_SGPR_SIMD_FULL_CSN, SPI_RA_LDS_CU_FULL_CSN, CPC_ME1_DC0_SPI_BUSY, TCC_EA_WRREQ_STALL_sum, TCC_EA_WRREQ_IO_CREDIT_STALL_sum, TCC_EA_WRREQ_GMI_CREDIT_STALL_sum, TCC_EA_WRREQ_DRAM_CREDIT_STALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170804_1298276/input0_results_240321_170804
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_5.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_6.txt
|-> [rocprof] RPL: on '240321_170805' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_6.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170805_1298477'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170805_1298477/input0_results_240321_170805'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170805_1298477/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_ACTIVE_INST_MISC, SQ_ACTIVE_INST_FLAT, SQ_INST_CYCLES_VMEM_WR, SQ_INST_CYCLES_VMEM_RD, SQ_INST_CYCLES_SMEM, SQ_INST_CYCLES_SALU, SQ_THREAD_CYCLES_VALU, SQ_IFETCH, TCP_TCC_WRITE_REQ_sum, TCP_TCC_ATOMIC_WITH_RET_REQ_sum, TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum, TCP_TCC_NC_READ_REQ_sum, TA_FLAT_WAVEFRONTS_sum, TA_FLAT_READ_WAVEFRONTS_sum, SPI_RA_BAR_CU_FULL_CSN, SPI_RA_TGLIM_CU_FULL_CSN, TCC_EA_RDREQ_sum, TCC_EA_RDREQ_32B_sum, TCC_EA_RD_UNCACHED_32B_sum, TCC_EA_RDREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170805_1298477/input0_results_240321_170805
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_6.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_7.txt
|-> [rocprof] RPL: on '240321_170805' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_7.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170805_1298679'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170805_1298679/input0_results_240321_170805'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170805_1298679/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_LDS_BANK_CONFLICT, SQ_LDS_ADDR_CONFLICT, SQ_LDS_UNALIGNED_STALL, SQ_WAVES_EQ_64, SQ_WAVES_LT_64, SQ_WAVES_LT_48, SQ_WAVES_LT_32, SQ_WAVES_LT_16, TCP_TCC_NC_WRITE_REQ_sum, TCP_TCC_NC_ATOMIC_REQ_sum, TCP_TCC_UC_READ_REQ_sum, TCP_TCC_UC_WRITE_REQ_sum, TA_FLAT_WRITE_WAVEFRONTS_sum, TA_FLAT_ATOMIC_WAVEFRONTS_sum, SPI_RA_WVLIM_STALL_CSN, SPI_SWC_CSC_WR, TCC_EA_RDREQ_IO_CREDIT_STALL_sum, TCC_EA_RDREQ_GMI_CREDIT_STALL_sum, TCC_EA_RDREQ_DRAM_CREDIT_STALL_sum, TCC_TAG_STALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170805_1298679/input0_results_240321_170805
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_7.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_8.txt
|-> [rocprof] RPL: on '240321_170806' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_8.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170806_1298868'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170806_1298868/input0_results_240321_170806'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170806_1298868/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 17 metrics
|-> [rocprof] SQ_ITEMS, SQ_LDS_MEM_VIOLATIONS, SQ_LDS_ATOMIC_RETURN, SQ_LDS_IDX_ACTIVE, SQ_WAVES_RESTORED, SQ_WAVES_SAVED, SQ_INSTS_SMEM_NORM, TCP_TCC_UC_ATOMIC_REQ_sum, TCP_TCC_CC_READ_REQ_sum, TCP_TCC_CC_WRITE_REQ_sum, TCP_TCC_CC_ATOMIC_REQ_sum, SPI_VWC_CSC_WR, SPI_RA_BULKY_CU_FULL_CSN, TCC_NORMAL_WRITEBACK_sum, TCC_ALL_TC_OP_WB_WRITEBACK_sum, TCC_NORMAL_EVICT_sum, TCC_ALL_TC_OP_INV_EVICT_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170806_1298868/input0_results_240321_170806
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_8.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/pmc_perf_9.txt
|-> [rocprof] RPL: on '240321_170806' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/pmc_perf_9.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170806_1299070'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170806_1299070/input0_results_240321_170806'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170806_1299070/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] TCP_TCC_RW_READ_REQ_sum, TCP_TCC_RW_WRITE_REQ_sum, TCP_TCC_RW_ATOMIC_REQ_sum, TCP_PENDING_STALL_CYCLES_sum, TCC_TOO_MANY_EA_WRREQS_STALL_sum, TCC_EA_ATOMIC_sum, TCC_EA_RDREQ_LEVEL_sum, TCC_EA_WRREQ_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170806_1299070/input0_results_240321_170806
|-> [rocprof] File 'tests/workloads/kernel/MI100/pmc_perf_9.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI100/perfmon/timestamps.txt
|-> [rocprof] RPL: on '240321_170807' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI100/perfmon/timestamps.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170807_1299273'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170807_1299273/input0_results_240321_170807'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170807_1299273/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 0 metrics
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170807_1299273/input0_results_240321_170807
|-> [rocprof] File 'tests/workloads/kernel/MI100/timestamps.csv' is generating
|-> [rocprof]
@@ -0,0 +1,5 @@
pmc: GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_CYCLES SQ_BUSY_CYCLES SQ_BUSY_CU_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQC_TC_INST_REQ SQC_TC_DATA_READ_REQ SQC_TC_DATA_WRITE_REQ GRBM_COUNT GRBM_GUI_ACTIVE TCP_GATE_EN1_sum TCP_GATE_EN2_sum TCP_TD_TCP_STALL_CYCLES_sum TCP_TCR_TCP_STALL_CYCLES_sum TA_TA_BUSY_sum TA_BUFFER_WAVEFRONTS_sum TD_TD_BUSY_sum TD_TC_STALL_sum SPI_CSN_WINDOW_VALID SPI_CSN_BUSY CPC_CPC_STAT_BUSY CPC_CPC_STAT_IDLE CPF_CPF_STAT_BUSY CPF_CPF_STAT_STALL TCC_CYCLE_sum TCC_BUSY_sum TCC_PROBE_sum TCC_PROBE_ALL_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQC_TC_DATA_ATOMIC_REQ SQC_TC_STALL SQC_TC_REQ SQC_DCACHE_REQ_READ_16 SQC_ICACHE_REQ SQC_ICACHE_HITS SQC_ICACHE_MISSES SQC_ICACHE_MISSES_DUPLICATE GRBM_SPI_BUSY TCP_READ_TAGCONFLICT_STALL_CYCLES_sum TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum TCP_TA_TCP_STATE_READ_sum TA_BUFFER_READ_WAVEFRONTS_sum TA_BUFFER_WRITE_WAVEFRONTS_sum TD_COALESCABLE_WAVEFRONT_sum TD_LOAD_WAVEFRONT_sum SPI_CSN_NUM_THREADGROUPS SPI_CSN_WAVE CPC_CPC_TCIU_BUSY CPC_CPC_TCIU_IDLE CPF_CPF_TCIU_BUSY CPF_CPF_TCIU_STALL TCC_NC_REQ_sum TCC_UC_REQ_sum TCC_CC_REQ_sum TCC_RW_REQ_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_EA_ATOMIC_LEVEL_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_ATOMIC[0] TCC_CYCLE[0] TCC_EA_ATOMIC[0] TCC_EA_ATOMIC_LEVEL[0] TCC_ATOMIC[1] TCC_CYCLE[1] TCC_EA_ATOMIC[1] TCC_EA_ATOMIC_LEVEL[1] TCC_ATOMIC[2] TCC_CYCLE[2] TCC_EA_ATOMIC[2] TCC_EA_ATOMIC_LEVEL[2] TCC_ATOMIC[3] TCC_CYCLE[3] TCC_EA_ATOMIC[3] TCC_EA_ATOMIC_LEVEL[3] TCC_ATOMIC[4] TCC_CYCLE[4] TCC_EA_ATOMIC[4] TCC_EA_ATOMIC_LEVEL[4] TCC_ATOMIC[5] TCC_CYCLE[5] TCC_EA_ATOMIC[5] TCC_EA_ATOMIC_LEVEL[5] TCC_ATOMIC[6] TCC_CYCLE[6] TCC_EA_ATOMIC[6] TCC_EA_ATOMIC_LEVEL[6] TCC_ATOMIC[7] TCC_CYCLE[7] TCC_EA_ATOMIC[7] TCC_EA_ATOMIC_LEVEL[7] TCC_ATOMIC[8] TCC_CYCLE[8] TCC_EA_ATOMIC[8] TCC_EA_ATOMIC_LEVEL[8] TCC_ATOMIC[9] TCC_CYCLE[9] TCC_EA_ATOMIC[9] TCC_EA_ATOMIC_LEVEL[9] TCC_ATOMIC[10] TCC_CYCLE[10] TCC_EA_ATOMIC[10] TCC_EA_ATOMIC_LEVEL[10] TCC_ATOMIC[11] TCC_CYCLE[11] TCC_EA_ATOMIC[11] TCC_EA_ATOMIC_LEVEL[11] TCC_ATOMIC[12] TCC_CYCLE[12] TCC_EA_ATOMIC[12] TCC_EA_ATOMIC_LEVEL[12] TCC_ATOMIC[13] TCC_CYCLE[13] TCC_EA_ATOMIC[13] TCC_EA_ATOMIC_LEVEL[13] TCC_ATOMIC[14] TCC_CYCLE[14] TCC_EA_ATOMIC[14] TCC_EA_ATOMIC_LEVEL[14] TCC_ATOMIC[15] TCC_CYCLE[15] TCC_EA_ATOMIC[15] TCC_EA_ATOMIC_LEVEL[15] TCC_ATOMIC[16] TCC_CYCLE[16] TCC_EA_ATOMIC[16] TCC_EA_ATOMIC_LEVEL[16] TCC_ATOMIC[17] TCC_CYCLE[17] TCC_EA_ATOMIC[17] TCC_EA_ATOMIC_LEVEL[17] TCC_ATOMIC[18] TCC_CYCLE[18] TCC_EA_ATOMIC[18] TCC_EA_ATOMIC_LEVEL[18] TCC_ATOMIC[19] TCC_CYCLE[19] TCC_EA_ATOMIC[19] TCC_EA_ATOMIC_LEVEL[19] TCC_ATOMIC[20] TCC_CYCLE[20] TCC_EA_ATOMIC[20] TCC_EA_ATOMIC_LEVEL[20] TCC_ATOMIC[21] TCC_CYCLE[21] TCC_EA_ATOMIC[21] TCC_EA_ATOMIC_LEVEL[21] TCC_ATOMIC[22] TCC_CYCLE[22] TCC_EA_ATOMIC[22] TCC_EA_ATOMIC_LEVEL[22] TCC_ATOMIC[23] TCC_CYCLE[23] TCC_EA_ATOMIC[23] TCC_EA_ATOMIC_LEVEL[23] TCC_ATOMIC[24] TCC_CYCLE[24] TCC_EA_ATOMIC[24] TCC_EA_ATOMIC_LEVEL[24] TCC_ATOMIC[25] TCC_CYCLE[25] TCC_EA_ATOMIC[25] TCC_EA_ATOMIC_LEVEL[25] TCC_ATOMIC[26] TCC_CYCLE[26] TCC_EA_ATOMIC[26] TCC_EA_ATOMIC_LEVEL[26] TCC_ATOMIC[27] TCC_CYCLE[27] TCC_EA_ATOMIC[27] TCC_EA_ATOMIC_LEVEL[27] TCC_ATOMIC[28] TCC_CYCLE[28] TCC_EA_ATOMIC[28] TCC_EA_ATOMIC_LEVEL[28] TCC_ATOMIC[29] TCC_CYCLE[29] TCC_EA_ATOMIC[29] TCC_EA_ATOMIC_LEVEL[29] TCC_ATOMIC[30] TCC_CYCLE[30] TCC_EA_ATOMIC[30] TCC_EA_ATOMIC_LEVEL[30] TCC_ATOMIC[31] TCC_CYCLE[31] TCC_EA_ATOMIC[31] TCC_EA_ATOMIC_LEVEL[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_EA_RDREQ[0] TCC_EA_RDREQ_32B[0] TCC_EA_RDREQ_DRAM_CREDIT_STALL[0] TCC_EA_RDREQ_GMI_CREDIT_STALL[0] TCC_EA_RDREQ[1] TCC_EA_RDREQ_32B[1] TCC_EA_RDREQ_DRAM_CREDIT_STALL[1] TCC_EA_RDREQ_GMI_CREDIT_STALL[1] TCC_EA_RDREQ[2] TCC_EA_RDREQ_32B[2] TCC_EA_RDREQ_DRAM_CREDIT_STALL[2] TCC_EA_RDREQ_GMI_CREDIT_STALL[2] TCC_EA_RDREQ[3] TCC_EA_RDREQ_32B[3] TCC_EA_RDREQ_DRAM_CREDIT_STALL[3] TCC_EA_RDREQ_GMI_CREDIT_STALL[3] TCC_EA_RDREQ[4] TCC_EA_RDREQ_32B[4] TCC_EA_RDREQ_DRAM_CREDIT_STALL[4] TCC_EA_RDREQ_GMI_CREDIT_STALL[4] TCC_EA_RDREQ[5] TCC_EA_RDREQ_32B[5] TCC_EA_RDREQ_DRAM_CREDIT_STALL[5] TCC_EA_RDREQ_GMI_CREDIT_STALL[5] TCC_EA_RDREQ[6] TCC_EA_RDREQ_32B[6] TCC_EA_RDREQ_DRAM_CREDIT_STALL[6] TCC_EA_RDREQ_GMI_CREDIT_STALL[6] TCC_EA_RDREQ[7] TCC_EA_RDREQ_32B[7] TCC_EA_RDREQ_DRAM_CREDIT_STALL[7] TCC_EA_RDREQ_GMI_CREDIT_STALL[7] TCC_EA_RDREQ[8] TCC_EA_RDREQ_32B[8] TCC_EA_RDREQ_DRAM_CREDIT_STALL[8] TCC_EA_RDREQ_GMI_CREDIT_STALL[8] TCC_EA_RDREQ[9] TCC_EA_RDREQ_32B[9] TCC_EA_RDREQ_DRAM_CREDIT_STALL[9] TCC_EA_RDREQ_GMI_CREDIT_STALL[9] TCC_EA_RDREQ[10] TCC_EA_RDREQ_32B[10] TCC_EA_RDREQ_DRAM_CREDIT_STALL[10] TCC_EA_RDREQ_GMI_CREDIT_STALL[10] TCC_EA_RDREQ[11] TCC_EA_RDREQ_32B[11] TCC_EA_RDREQ_DRAM_CREDIT_STALL[11] TCC_EA_RDREQ_GMI_CREDIT_STALL[11] TCC_EA_RDREQ[12] TCC_EA_RDREQ_32B[12] TCC_EA_RDREQ_DRAM_CREDIT_STALL[12] TCC_EA_RDREQ_GMI_CREDIT_STALL[12] TCC_EA_RDREQ[13] TCC_EA_RDREQ_32B[13] TCC_EA_RDREQ_DRAM_CREDIT_STALL[13] TCC_EA_RDREQ_GMI_CREDIT_STALL[13] TCC_EA_RDREQ[14] TCC_EA_RDREQ_32B[14] TCC_EA_RDREQ_DRAM_CREDIT_STALL[14] TCC_EA_RDREQ_GMI_CREDIT_STALL[14] TCC_EA_RDREQ[15] TCC_EA_RDREQ_32B[15] TCC_EA_RDREQ_DRAM_CREDIT_STALL[15] TCC_EA_RDREQ_GMI_CREDIT_STALL[15] TCC_EA_RDREQ[16] TCC_EA_RDREQ_32B[16] TCC_EA_RDREQ_DRAM_CREDIT_STALL[16] TCC_EA_RDREQ_GMI_CREDIT_STALL[16] TCC_EA_RDREQ[17] TCC_EA_RDREQ_32B[17] TCC_EA_RDREQ_DRAM_CREDIT_STALL[17] TCC_EA_RDREQ_GMI_CREDIT_STALL[17] TCC_EA_RDREQ[18] TCC_EA_RDREQ_32B[18] TCC_EA_RDREQ_DRAM_CREDIT_STALL[18] TCC_EA_RDREQ_GMI_CREDIT_STALL[18] TCC_EA_RDREQ[19] TCC_EA_RDREQ_32B[19] TCC_EA_RDREQ_DRAM_CREDIT_STALL[19] TCC_EA_RDREQ_GMI_CREDIT_STALL[19] TCC_EA_RDREQ[20] TCC_EA_RDREQ_32B[20] TCC_EA_RDREQ_DRAM_CREDIT_STALL[20] TCC_EA_RDREQ_GMI_CREDIT_STALL[20] TCC_EA_RDREQ[21] TCC_EA_RDREQ_32B[21] TCC_EA_RDREQ_DRAM_CREDIT_STALL[21] TCC_EA_RDREQ_GMI_CREDIT_STALL[21] TCC_EA_RDREQ[22] TCC_EA_RDREQ_32B[22] TCC_EA_RDREQ_DRAM_CREDIT_STALL[22] TCC_EA_RDREQ_GMI_CREDIT_STALL[22] TCC_EA_RDREQ[23] TCC_EA_RDREQ_32B[23] TCC_EA_RDREQ_DRAM_CREDIT_STALL[23] TCC_EA_RDREQ_GMI_CREDIT_STALL[23] TCC_EA_RDREQ[24] TCC_EA_RDREQ_32B[24] TCC_EA_RDREQ_DRAM_CREDIT_STALL[24] TCC_EA_RDREQ_GMI_CREDIT_STALL[24] TCC_EA_RDREQ[25] TCC_EA_RDREQ_32B[25] TCC_EA_RDREQ_DRAM_CREDIT_STALL[25] TCC_EA_RDREQ_GMI_CREDIT_STALL[25] TCC_EA_RDREQ[26] TCC_EA_RDREQ_32B[26] TCC_EA_RDREQ_DRAM_CREDIT_STALL[26] TCC_EA_RDREQ_GMI_CREDIT_STALL[26] TCC_EA_RDREQ[27] TCC_EA_RDREQ_32B[27] TCC_EA_RDREQ_DRAM_CREDIT_STALL[27] TCC_EA_RDREQ_GMI_CREDIT_STALL[27] TCC_EA_RDREQ[28] TCC_EA_RDREQ_32B[28] TCC_EA_RDREQ_DRAM_CREDIT_STALL[28] TCC_EA_RDREQ_GMI_CREDIT_STALL[28] TCC_EA_RDREQ[29] TCC_EA_RDREQ_32B[29] TCC_EA_RDREQ_DRAM_CREDIT_STALL[29] TCC_EA_RDREQ_GMI_CREDIT_STALL[29] TCC_EA_RDREQ[30] TCC_EA_RDREQ_32B[30] TCC_EA_RDREQ_DRAM_CREDIT_STALL[30] TCC_EA_RDREQ_GMI_CREDIT_STALL[30] TCC_EA_RDREQ[31] TCC_EA_RDREQ_32B[31] TCC_EA_RDREQ_DRAM_CREDIT_STALL[31] TCC_EA_RDREQ_GMI_CREDIT_STALL[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_EA_RDREQ_IO_CREDIT_STALL[0] TCC_EA_RDREQ_LEVEL[0] TCC_EA_WRREQ[0] TCC_EA_WRREQ_64B[0] TCC_EA_RDREQ_IO_CREDIT_STALL[1] TCC_EA_RDREQ_LEVEL[1] TCC_EA_WRREQ[1] TCC_EA_WRREQ_64B[1] TCC_EA_RDREQ_IO_CREDIT_STALL[2] TCC_EA_RDREQ_LEVEL[2] TCC_EA_WRREQ[2] TCC_EA_WRREQ_64B[2] TCC_EA_RDREQ_IO_CREDIT_STALL[3] TCC_EA_RDREQ_LEVEL[3] TCC_EA_WRREQ[3] TCC_EA_WRREQ_64B[3] TCC_EA_RDREQ_IO_CREDIT_STALL[4] TCC_EA_RDREQ_LEVEL[4] TCC_EA_WRREQ[4] TCC_EA_WRREQ_64B[4] TCC_EA_RDREQ_IO_CREDIT_STALL[5] TCC_EA_RDREQ_LEVEL[5] TCC_EA_WRREQ[5] TCC_EA_WRREQ_64B[5] TCC_EA_RDREQ_IO_CREDIT_STALL[6] TCC_EA_RDREQ_LEVEL[6] TCC_EA_WRREQ[6] TCC_EA_WRREQ_64B[6] TCC_EA_RDREQ_IO_CREDIT_STALL[7] TCC_EA_RDREQ_LEVEL[7] TCC_EA_WRREQ[7] TCC_EA_WRREQ_64B[7] TCC_EA_RDREQ_IO_CREDIT_STALL[8] TCC_EA_RDREQ_LEVEL[8] TCC_EA_WRREQ[8] TCC_EA_WRREQ_64B[8] TCC_EA_RDREQ_IO_CREDIT_STALL[9] TCC_EA_RDREQ_LEVEL[9] TCC_EA_WRREQ[9] TCC_EA_WRREQ_64B[9] TCC_EA_RDREQ_IO_CREDIT_STALL[10] TCC_EA_RDREQ_LEVEL[10] TCC_EA_WRREQ[10] TCC_EA_WRREQ_64B[10] TCC_EA_RDREQ_IO_CREDIT_STALL[11] TCC_EA_RDREQ_LEVEL[11] TCC_EA_WRREQ[11] TCC_EA_WRREQ_64B[11] TCC_EA_RDREQ_IO_CREDIT_STALL[12] TCC_EA_RDREQ_LEVEL[12] TCC_EA_WRREQ[12] TCC_EA_WRREQ_64B[12] TCC_EA_RDREQ_IO_CREDIT_STALL[13] TCC_EA_RDREQ_LEVEL[13] TCC_EA_WRREQ[13] TCC_EA_WRREQ_64B[13] TCC_EA_RDREQ_IO_CREDIT_STALL[14] TCC_EA_RDREQ_LEVEL[14] TCC_EA_WRREQ[14] TCC_EA_WRREQ_64B[14] TCC_EA_RDREQ_IO_CREDIT_STALL[15] TCC_EA_RDREQ_LEVEL[15] TCC_EA_WRREQ[15] TCC_EA_WRREQ_64B[15] TCC_EA_RDREQ_IO_CREDIT_STALL[16] TCC_EA_RDREQ_LEVEL[16] TCC_EA_WRREQ[16] TCC_EA_WRREQ_64B[16] TCC_EA_RDREQ_IO_CREDIT_STALL[17] TCC_EA_RDREQ_LEVEL[17] TCC_EA_WRREQ[17] TCC_EA_WRREQ_64B[17] TCC_EA_RDREQ_IO_CREDIT_STALL[18] TCC_EA_RDREQ_LEVEL[18] TCC_EA_WRREQ[18] TCC_EA_WRREQ_64B[18] TCC_EA_RDREQ_IO_CREDIT_STALL[19] TCC_EA_RDREQ_LEVEL[19] TCC_EA_WRREQ[19] TCC_EA_WRREQ_64B[19] TCC_EA_RDREQ_IO_CREDIT_STALL[20] TCC_EA_RDREQ_LEVEL[20] TCC_EA_WRREQ[20] TCC_EA_WRREQ_64B[20] TCC_EA_RDREQ_IO_CREDIT_STALL[21] TCC_EA_RDREQ_LEVEL[21] TCC_EA_WRREQ[21] TCC_EA_WRREQ_64B[21] TCC_EA_RDREQ_IO_CREDIT_STALL[22] TCC_EA_RDREQ_LEVEL[22] TCC_EA_WRREQ[22] TCC_EA_WRREQ_64B[22] TCC_EA_RDREQ_IO_CREDIT_STALL[23] TCC_EA_RDREQ_LEVEL[23] TCC_EA_WRREQ[23] TCC_EA_WRREQ_64B[23] TCC_EA_RDREQ_IO_CREDIT_STALL[24] TCC_EA_RDREQ_LEVEL[24] TCC_EA_WRREQ[24] TCC_EA_WRREQ_64B[24] TCC_EA_RDREQ_IO_CREDIT_STALL[25] TCC_EA_RDREQ_LEVEL[25] TCC_EA_WRREQ[25] TCC_EA_WRREQ_64B[25] TCC_EA_RDREQ_IO_CREDIT_STALL[26] TCC_EA_RDREQ_LEVEL[26] TCC_EA_WRREQ[26] TCC_EA_WRREQ_64B[26] TCC_EA_RDREQ_IO_CREDIT_STALL[27] TCC_EA_RDREQ_LEVEL[27] TCC_EA_WRREQ[27] TCC_EA_WRREQ_64B[27] TCC_EA_RDREQ_IO_CREDIT_STALL[28] TCC_EA_RDREQ_LEVEL[28] TCC_EA_WRREQ[28] TCC_EA_WRREQ_64B[28] TCC_EA_RDREQ_IO_CREDIT_STALL[29] TCC_EA_RDREQ_LEVEL[29] TCC_EA_WRREQ[29] TCC_EA_WRREQ_64B[29] TCC_EA_RDREQ_IO_CREDIT_STALL[30] TCC_EA_RDREQ_LEVEL[30] TCC_EA_WRREQ[30] TCC_EA_WRREQ_64B[30] TCC_EA_RDREQ_IO_CREDIT_STALL[31] TCC_EA_RDREQ_LEVEL[31] TCC_EA_WRREQ[31] TCC_EA_WRREQ_64B[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_EA_WRREQ_DRAM_CREDIT_STALL[0] TCC_EA_WRREQ_GMI_CREDIT_STALL[0] TCC_EA_WRREQ_IO_CREDIT_STALL[0] TCC_EA_WRREQ_LEVEL[0] TCC_EA_WRREQ_DRAM_CREDIT_STALL[1] TCC_EA_WRREQ_GMI_CREDIT_STALL[1] TCC_EA_WRREQ_IO_CREDIT_STALL[1] TCC_EA_WRREQ_LEVEL[1] TCC_EA_WRREQ_DRAM_CREDIT_STALL[2] TCC_EA_WRREQ_GMI_CREDIT_STALL[2] TCC_EA_WRREQ_IO_CREDIT_STALL[2] TCC_EA_WRREQ_LEVEL[2] TCC_EA_WRREQ_DRAM_CREDIT_STALL[3] TCC_EA_WRREQ_GMI_CREDIT_STALL[3] TCC_EA_WRREQ_IO_CREDIT_STALL[3] TCC_EA_WRREQ_LEVEL[3] TCC_EA_WRREQ_DRAM_CREDIT_STALL[4] TCC_EA_WRREQ_GMI_CREDIT_STALL[4] TCC_EA_WRREQ_IO_CREDIT_STALL[4] TCC_EA_WRREQ_LEVEL[4] TCC_EA_WRREQ_DRAM_CREDIT_STALL[5] TCC_EA_WRREQ_GMI_CREDIT_STALL[5] TCC_EA_WRREQ_IO_CREDIT_STALL[5] TCC_EA_WRREQ_LEVEL[5] TCC_EA_WRREQ_DRAM_CREDIT_STALL[6] TCC_EA_WRREQ_GMI_CREDIT_STALL[6] TCC_EA_WRREQ_IO_CREDIT_STALL[6] TCC_EA_WRREQ_LEVEL[6] TCC_EA_WRREQ_DRAM_CREDIT_STALL[7] TCC_EA_WRREQ_GMI_CREDIT_STALL[7] TCC_EA_WRREQ_IO_CREDIT_STALL[7] TCC_EA_WRREQ_LEVEL[7] TCC_EA_WRREQ_DRAM_CREDIT_STALL[8] TCC_EA_WRREQ_GMI_CREDIT_STALL[8] TCC_EA_WRREQ_IO_CREDIT_STALL[8] TCC_EA_WRREQ_LEVEL[8] TCC_EA_WRREQ_DRAM_CREDIT_STALL[9] TCC_EA_WRREQ_GMI_CREDIT_STALL[9] TCC_EA_WRREQ_IO_CREDIT_STALL[9] TCC_EA_WRREQ_LEVEL[9] TCC_EA_WRREQ_DRAM_CREDIT_STALL[10] TCC_EA_WRREQ_GMI_CREDIT_STALL[10] TCC_EA_WRREQ_IO_CREDIT_STALL[10] TCC_EA_WRREQ_LEVEL[10] TCC_EA_WRREQ_DRAM_CREDIT_STALL[11] TCC_EA_WRREQ_GMI_CREDIT_STALL[11] TCC_EA_WRREQ_IO_CREDIT_STALL[11] TCC_EA_WRREQ_LEVEL[11] TCC_EA_WRREQ_DRAM_CREDIT_STALL[12] TCC_EA_WRREQ_GMI_CREDIT_STALL[12] TCC_EA_WRREQ_IO_CREDIT_STALL[12] TCC_EA_WRREQ_LEVEL[12] TCC_EA_WRREQ_DRAM_CREDIT_STALL[13] TCC_EA_WRREQ_GMI_CREDIT_STALL[13] TCC_EA_WRREQ_IO_CREDIT_STALL[13] TCC_EA_WRREQ_LEVEL[13] TCC_EA_WRREQ_DRAM_CREDIT_STALL[14] TCC_EA_WRREQ_GMI_CREDIT_STALL[14] TCC_EA_WRREQ_IO_CREDIT_STALL[14] TCC_EA_WRREQ_LEVEL[14] TCC_EA_WRREQ_DRAM_CREDIT_STALL[15] TCC_EA_WRREQ_GMI_CREDIT_STALL[15] TCC_EA_WRREQ_IO_CREDIT_STALL[15] TCC_EA_WRREQ_LEVEL[15] TCC_EA_WRREQ_DRAM_CREDIT_STALL[16] TCC_EA_WRREQ_GMI_CREDIT_STALL[16] TCC_EA_WRREQ_IO_CREDIT_STALL[16] TCC_EA_WRREQ_LEVEL[16] TCC_EA_WRREQ_DRAM_CREDIT_STALL[17] TCC_EA_WRREQ_GMI_CREDIT_STALL[17] TCC_EA_WRREQ_IO_CREDIT_STALL[17] TCC_EA_WRREQ_LEVEL[17] TCC_EA_WRREQ_DRAM_CREDIT_STALL[18] TCC_EA_WRREQ_GMI_CREDIT_STALL[18] TCC_EA_WRREQ_IO_CREDIT_STALL[18] TCC_EA_WRREQ_LEVEL[18] TCC_EA_WRREQ_DRAM_CREDIT_STALL[19] TCC_EA_WRREQ_GMI_CREDIT_STALL[19] TCC_EA_WRREQ_IO_CREDIT_STALL[19] TCC_EA_WRREQ_LEVEL[19] TCC_EA_WRREQ_DRAM_CREDIT_STALL[20] TCC_EA_WRREQ_GMI_CREDIT_STALL[20] TCC_EA_WRREQ_IO_CREDIT_STALL[20] TCC_EA_WRREQ_LEVEL[20] TCC_EA_WRREQ_DRAM_CREDIT_STALL[21] TCC_EA_WRREQ_GMI_CREDIT_STALL[21] TCC_EA_WRREQ_IO_CREDIT_STALL[21] TCC_EA_WRREQ_LEVEL[21] TCC_EA_WRREQ_DRAM_CREDIT_STALL[22] TCC_EA_WRREQ_GMI_CREDIT_STALL[22] TCC_EA_WRREQ_IO_CREDIT_STALL[22] TCC_EA_WRREQ_LEVEL[22] TCC_EA_WRREQ_DRAM_CREDIT_STALL[23] TCC_EA_WRREQ_GMI_CREDIT_STALL[23] TCC_EA_WRREQ_IO_CREDIT_STALL[23] TCC_EA_WRREQ_LEVEL[23] TCC_EA_WRREQ_DRAM_CREDIT_STALL[24] TCC_EA_WRREQ_GMI_CREDIT_STALL[24] TCC_EA_WRREQ_IO_CREDIT_STALL[24] TCC_EA_WRREQ_LEVEL[24] TCC_EA_WRREQ_DRAM_CREDIT_STALL[25] TCC_EA_WRREQ_GMI_CREDIT_STALL[25] TCC_EA_WRREQ_IO_CREDIT_STALL[25] TCC_EA_WRREQ_LEVEL[25] TCC_EA_WRREQ_DRAM_CREDIT_STALL[26] TCC_EA_WRREQ_GMI_CREDIT_STALL[26] TCC_EA_WRREQ_IO_CREDIT_STALL[26] TCC_EA_WRREQ_LEVEL[26] TCC_EA_WRREQ_DRAM_CREDIT_STALL[27] TCC_EA_WRREQ_GMI_CREDIT_STALL[27] TCC_EA_WRREQ_IO_CREDIT_STALL[27] TCC_EA_WRREQ_LEVEL[27] TCC_EA_WRREQ_DRAM_CREDIT_STALL[28] TCC_EA_WRREQ_GMI_CREDIT_STALL[28] TCC_EA_WRREQ_IO_CREDIT_STALL[28] TCC_EA_WRREQ_LEVEL[28] TCC_EA_WRREQ_DRAM_CREDIT_STALL[29] TCC_EA_WRREQ_GMI_CREDIT_STALL[29] TCC_EA_WRREQ_IO_CREDIT_STALL[29] TCC_EA_WRREQ_LEVEL[29] TCC_EA_WRREQ_DRAM_CREDIT_STALL[30] TCC_EA_WRREQ_GMI_CREDIT_STALL[30] TCC_EA_WRREQ_IO_CREDIT_STALL[30] TCC_EA_WRREQ_LEVEL[30] TCC_EA_WRREQ_DRAM_CREDIT_STALL[31] TCC_EA_WRREQ_GMI_CREDIT_STALL[31] TCC_EA_WRREQ_IO_CREDIT_STALL[31] TCC_EA_WRREQ_LEVEL[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_HIT[0] TCC_MISS[0] TCC_READ[0] TCC_REQ[0] TCC_HIT[1] TCC_MISS[1] TCC_READ[1] TCC_REQ[1] TCC_HIT[2] TCC_MISS[2] TCC_READ[2] TCC_REQ[2] TCC_HIT[3] TCC_MISS[3] TCC_READ[3] TCC_REQ[3] TCC_HIT[4] TCC_MISS[4] TCC_READ[4] TCC_REQ[4] TCC_HIT[5] TCC_MISS[5] TCC_READ[5] TCC_REQ[5] TCC_HIT[6] TCC_MISS[6] TCC_READ[6] TCC_REQ[6] TCC_HIT[7] TCC_MISS[7] TCC_READ[7] TCC_REQ[7] TCC_HIT[8] TCC_MISS[8] TCC_READ[8] TCC_REQ[8] TCC_HIT[9] TCC_MISS[9] TCC_READ[9] TCC_REQ[9] TCC_HIT[10] TCC_MISS[10] TCC_READ[10] TCC_REQ[10] TCC_HIT[11] TCC_MISS[11] TCC_READ[11] TCC_REQ[11] TCC_HIT[12] TCC_MISS[12] TCC_READ[12] TCC_REQ[12] TCC_HIT[13] TCC_MISS[13] TCC_READ[13] TCC_REQ[13] TCC_HIT[14] TCC_MISS[14] TCC_READ[14] TCC_REQ[14] TCC_HIT[15] TCC_MISS[15] TCC_READ[15] TCC_REQ[15] TCC_HIT[16] TCC_MISS[16] TCC_READ[16] TCC_REQ[16] TCC_HIT[17] TCC_MISS[17] TCC_READ[17] TCC_REQ[17] TCC_HIT[18] TCC_MISS[18] TCC_READ[18] TCC_REQ[18] TCC_HIT[19] TCC_MISS[19] TCC_READ[19] TCC_REQ[19] TCC_HIT[20] TCC_MISS[20] TCC_READ[20] TCC_REQ[20] TCC_HIT[21] TCC_MISS[21] TCC_READ[21] TCC_REQ[21] TCC_HIT[22] TCC_MISS[22] TCC_READ[22] TCC_REQ[22] TCC_HIT[23] TCC_MISS[23] TCC_READ[23] TCC_REQ[23] TCC_HIT[24] TCC_MISS[24] TCC_READ[24] TCC_REQ[24] TCC_HIT[25] TCC_MISS[25] TCC_READ[25] TCC_REQ[25] TCC_HIT[26] TCC_MISS[26] TCC_READ[26] TCC_REQ[26] TCC_HIT[27] TCC_MISS[27] TCC_READ[27] TCC_REQ[27] TCC_HIT[28] TCC_MISS[28] TCC_READ[28] TCC_REQ[28] TCC_HIT[29] TCC_MISS[29] TCC_READ[29] TCC_REQ[29] TCC_HIT[30] TCC_MISS[30] TCC_READ[30] TCC_REQ[30] TCC_HIT[31] TCC_MISS[31] TCC_READ[31] TCC_REQ[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_RW_REQ[0] TCC_TOO_MANY_EA_WRREQS_STALL[0] TCC_WRITE[0] TCC_RW_REQ[1] TCC_TOO_MANY_EA_WRREQS_STALL[1] TCC_WRITE[1] TCC_RW_REQ[2] TCC_TOO_MANY_EA_WRREQS_STALL[2] TCC_WRITE[2] TCC_RW_REQ[3] TCC_TOO_MANY_EA_WRREQS_STALL[3] TCC_WRITE[3] TCC_RW_REQ[4] TCC_TOO_MANY_EA_WRREQS_STALL[4] TCC_WRITE[4] TCC_RW_REQ[5] TCC_TOO_MANY_EA_WRREQS_STALL[5] TCC_WRITE[5] TCC_RW_REQ[6] TCC_TOO_MANY_EA_WRREQS_STALL[6] TCC_WRITE[6] TCC_RW_REQ[7] TCC_TOO_MANY_EA_WRREQS_STALL[7] TCC_WRITE[7] TCC_RW_REQ[8] TCC_TOO_MANY_EA_WRREQS_STALL[8] TCC_WRITE[8] TCC_RW_REQ[9] TCC_TOO_MANY_EA_WRREQS_STALL[9] TCC_WRITE[9] TCC_RW_REQ[10] TCC_TOO_MANY_EA_WRREQS_STALL[10] TCC_WRITE[10] TCC_RW_REQ[11] TCC_TOO_MANY_EA_WRREQS_STALL[11] TCC_WRITE[11] TCC_RW_REQ[12] TCC_TOO_MANY_EA_WRREQS_STALL[12] TCC_WRITE[12] TCC_RW_REQ[13] TCC_TOO_MANY_EA_WRREQS_STALL[13] TCC_WRITE[13] TCC_RW_REQ[14] TCC_TOO_MANY_EA_WRREQS_STALL[14] TCC_WRITE[14] TCC_RW_REQ[15] TCC_TOO_MANY_EA_WRREQS_STALL[15] TCC_WRITE[15] TCC_RW_REQ[16] TCC_TOO_MANY_EA_WRREQS_STALL[16] TCC_WRITE[16] TCC_RW_REQ[17] TCC_TOO_MANY_EA_WRREQS_STALL[17] TCC_WRITE[17] TCC_RW_REQ[18] TCC_TOO_MANY_EA_WRREQS_STALL[18] TCC_WRITE[18] TCC_RW_REQ[19] TCC_TOO_MANY_EA_WRREQS_STALL[19] TCC_WRITE[19] TCC_RW_REQ[20] TCC_TOO_MANY_EA_WRREQS_STALL[20] TCC_WRITE[20] TCC_RW_REQ[21] TCC_TOO_MANY_EA_WRREQS_STALL[21] TCC_WRITE[21] TCC_RW_REQ[22] TCC_TOO_MANY_EA_WRREQS_STALL[22] TCC_WRITE[22] TCC_RW_REQ[23] TCC_TOO_MANY_EA_WRREQS_STALL[23] TCC_WRITE[23] TCC_RW_REQ[24] TCC_TOO_MANY_EA_WRREQS_STALL[24] TCC_WRITE[24] TCC_RW_REQ[25] TCC_TOO_MANY_EA_WRREQS_STALL[25] TCC_WRITE[25] TCC_RW_REQ[26] TCC_TOO_MANY_EA_WRREQS_STALL[26] TCC_WRITE[26] TCC_RW_REQ[27] TCC_TOO_MANY_EA_WRREQS_STALL[27] TCC_WRITE[27] TCC_RW_REQ[28] TCC_TOO_MANY_EA_WRREQS_STALL[28] TCC_WRITE[28] TCC_RW_REQ[29] TCC_TOO_MANY_EA_WRREQS_STALL[29] TCC_WRITE[29] TCC_RW_REQ[30] TCC_TOO_MANY_EA_WRREQS_STALL[30] TCC_WRITE[30] TCC_RW_REQ[31] TCC_TOO_MANY_EA_WRREQS_STALL[31] TCC_WRITE[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQC_DCACHE_INPUT_VALID_READYB SQC_DCACHE_ATOMIC SQC_DCACHE_REQ_READ_8 SQC_DCACHE_REQ SQC_DCACHE_HITS SQC_DCACHE_MISSES SQC_DCACHE_MISSES_DUPLICATE SQC_DCACHE_REQ_READ_1 TCP_VOLATILE_sum TCP_TOTAL_ACCESSES_sum TCP_TOTAL_READ_sum TCP_TOTAL_WRITE_sum TA_BUFFER_ATOMIC_WAVEFRONTS_sum TA_BUFFER_TOTAL_CYCLES_sum TD_ATOMIC_WAVEFRONT_sum TD_STORE_WAVEFRONT_sum SPI_RA_REQ_NO_ALLOC SPI_RA_REQ_NO_ALLOC_CSN CPC_CPC_STAT_STALL CPC_UTCL1_STALL_ON_TRANSLATION CPF_CPF_STAT_IDLE CPF_CPF_TCIU_IDLE TCC_REQ_sum TCC_STREAMING_REQ_sum TCC_HIT_sum TCC_MISS_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQC_DCACHE_REQ_READ_2 SQC_DCACHE_REQ_READ_4 SQ_INSTS_VMEM_WR SQ_INSTS_VMEM_RD SQ_INSTS_VMEM SQ_INSTS_SALU SQ_INSTS_VSKIPPED SQ_INSTS_SMEM TCP_TOTAL_ATOMIC_WITH_RET_sum TCP_TOTAL_ATOMIC_WITHOUT_RET_sum TCP_TOTAL_WRITEBACK_INVALIDATES_sum TCP_TOTAL_CACHE_ACCESSES_sum TA_BUFFER_COALESCED_READ_CYCLES_sum TA_BUFFER_COALESCED_WRITE_CYCLES_sum SPI_RA_RES_STALL_CSN SPI_RA_TMP_STALL_CSN CPC_CPC_UTCL2IU_BUSY CPC_CPC_UTCL2IU_IDLE CPF_CMP_UTCL1_STALL_ON_TRANSLATION TCC_READ_sum TCC_WRITE_sum TCC_ATOMIC_sum TCC_WRITEBACK_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_FLAT SQ_INSTS_LDS SQ_INSTS_GDS SQ_INSTS_EXP_GDS SQ_INSTS_BRANCH SQ_INSTS_SENDMSG SQ_INSTS SQ_WAIT_ANY TCP_UTCL1_TRANSLATION_MISS_sum TCP_UTCL1_TRANSLATION_HIT_sum TCP_UTCL1_PERMISSION_MISS_sum TCP_UTCL1_REQUEST_sum TA_ADDR_STALLED_BY_TC_CYCLES_sum TA_TOTAL_WAVEFRONTS_sum SPI_RA_WAVE_SIMD_FULL_CSN SPI_RA_VGPR_SIMD_FULL_CSN CPC_CPC_UTCL2IU_STALL CPC_ME1_BUSY_FOR_PACKET_DECODE TCC_EA_WRREQ_sum TCC_EA_WRREQ_64B_sum TCC_EA_WR_UNCACHED_32B_sum TCC_EA_WRREQ_DRAM_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_WAIT_INST_ANY SQ_ACTIVE_INST_ANY SQ_INSTS_VALU SQ_ACTIVE_INST_VMEM SQ_ACTIVE_INST_LDS SQ_ACTIVE_INST_VALU SQ_ACTIVE_INST_SCA SQ_ACTIVE_INST_EXP_GDS TCP_TCP_LATENCY_sum TCP_TCC_READ_REQ_LATENCY_sum TCP_TCC_WRITE_REQ_LATENCY_sum TCP_TCC_READ_REQ_sum TA_ADDR_STALLED_BY_TD_CYCLES_sum TA_DATA_STALLED_BY_TC_CYCLES_sum SPI_RA_SGPR_SIMD_FULL_CSN SPI_RA_LDS_CU_FULL_CSN CPC_ME1_DC0_SPI_BUSY TCC_EA_WRREQ_STALL_sum TCC_EA_WRREQ_IO_CREDIT_STALL_sum TCC_EA_WRREQ_GMI_CREDIT_STALL_sum TCC_EA_WRREQ_DRAM_CREDIT_STALL_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_ACTIVE_INST_MISC SQ_ACTIVE_INST_FLAT SQ_INST_CYCLES_VMEM_WR SQ_INST_CYCLES_VMEM_RD SQ_INST_CYCLES_SMEM SQ_INST_CYCLES_SALU SQ_THREAD_CYCLES_VALU SQ_IFETCH TCP_TCC_WRITE_REQ_sum TCP_TCC_ATOMIC_WITH_RET_REQ_sum TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum TCP_TCC_NC_READ_REQ_sum TA_FLAT_WAVEFRONTS_sum TA_FLAT_READ_WAVEFRONTS_sum SPI_RA_BAR_CU_FULL_CSN SPI_RA_TGLIM_CU_FULL_CSN TCC_EA_RDREQ_sum TCC_EA_RDREQ_32B_sum TCC_EA_RD_UNCACHED_32B_sum TCC_EA_RDREQ_DRAM_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_LDS_BANK_CONFLICT SQ_LDS_ADDR_CONFLICT SQ_LDS_UNALIGNED_STALL SQ_WAVES_EQ_64 SQ_WAVES_LT_64 SQ_WAVES_LT_48 SQ_WAVES_LT_32 SQ_WAVES_LT_16 TCP_TCC_NC_WRITE_REQ_sum TCP_TCC_NC_ATOMIC_REQ_sum TCP_TCC_UC_READ_REQ_sum TCP_TCC_UC_WRITE_REQ_sum TA_FLAT_WRITE_WAVEFRONTS_sum TA_FLAT_ATOMIC_WAVEFRONTS_sum SPI_RA_WVLIM_STALL_CSN SPI_SWC_CSC_WR TCC_EA_RDREQ_IO_CREDIT_STALL_sum TCC_EA_RDREQ_GMI_CREDIT_STALL_sum TCC_EA_RDREQ_DRAM_CREDIT_STALL_sum TCC_TAG_STALL_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_ITEMS SQ_LDS_MEM_VIOLATIONS SQ_LDS_ATOMIC_RETURN SQ_LDS_IDX_ACTIVE SQ_WAVES_RESTORED SQ_WAVES_SAVED SQ_INSTS_SMEM_NORM TCP_TCC_UC_ATOMIC_REQ_sum TCP_TCC_CC_READ_REQ_sum TCP_TCC_CC_WRITE_REQ_sum TCP_TCC_CC_ATOMIC_REQ_sum SPI_VWC_CSC_WR SPI_RA_BULKY_CU_FULL_CSN TCC_NORMAL_WRITEBACK_sum TCC_ALL_TC_OP_WB_WRITEBACK_sum TCC_NORMAL_EVICT_sum TCC_ALL_TC_OP_INV_EVICT_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCP_TCC_RW_READ_REQ_sum TCP_TCC_RW_WRITE_REQ_sum TCP_TCC_RW_ATOMIC_REQ_sum TCP_PENDING_STALL_CYCLES_sum TCC_TOO_MANY_EA_WRREQS_STALL_sum TCC_EA_ATOMIC_sum TCC_EA_RDREQ_LEVEL_sum TCC_EA_WRREQ_LEVEL_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc:
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1 Dispatch_ID Kernel_Name GPU_ID
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
File diff suppressed because one or more lines are too long
@@ -0,0 +1,2 @@
workload_name,command,ip_blocks,timestamp,version,hostname,cpu_model,sbios,linux_distro,linux_kernel_version,amd_gpu_kernel_version,cpu_memory,gpu_memory,rocm_version,vbios,compute_partition,memory_partition,gpu_model,gpu_arch,gpu_l1,gpu_l2,cu_per_gpu,simd_per_cu,se_per_gpu,wave_size,workgroup_max_size,max_waves_per_cu,max_sclk,max_mclk,cur_sclk,cur_mclk,total_l2_chan,lds_banks_per_cu,sqc_per_gpu,pipes_per_gpu,hbm_bw,num_xcd,num_hbm_channels
path,./tests/vcopy -n 1048576 -b 256 -i 3,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF,Thu 21 Mar 2024 03:52:12 PM (CDT),2,t007-001.hpcfund,AMD EPYC 7V13 64-Core Processor,American Megatrends Inc.0602,Rocky Linux 9.1 (Blue Onyx),5.14.0-162.18.1.el9_1.x86_64,,527651008,,6.0.2-115,113-D3431401-100,NA,NA,MI100,gfx908,16,8192,120,4,8,64,1024,40,1502,1200,1502,1200,32,32,64,4,1228.8,1,32
1 workload_name command ip_blocks timestamp version hostname cpu_model sbios linux_distro linux_kernel_version amd_gpu_kernel_version cpu_memory gpu_memory rocm_version vbios compute_partition memory_partition gpu_model gpu_arch gpu_l1 gpu_l2 cu_per_gpu simd_per_cu se_per_gpu wave_size workgroup_max_size max_waves_per_cu max_sclk max_mclk cur_sclk cur_mclk total_l2_chan lds_banks_per_cu sqc_per_gpu pipes_per_gpu hbm_bw num_xcd num_hbm_channels
2 path ./tests/vcopy -n 1048576 -b 256 -i 3 SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF Thu 21 Mar 2024 03:52:12 PM (CDT) 2 t007-001.hpcfund AMD EPYC 7V13 64-Core Processor American Megatrends Inc.0602 Rocky Linux 9.1 (Blue Onyx) 5.14.0-162.18.1.el9_1.x86_64 527651008 6.0.2-115 113-D3431401-100 NA NA MI100 gfx908 16 8192 120 4 8 64 1024 40 1502 1200 1502 1200 32 32 64 4 1228.8 1 32
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1299433,1299433,1048576,256,0,0,8,8,16,64,0x0,0x7fc619c60ec0,1414621055400032,1414621055425142,1414621055449622,1414621055460846
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1299433,1299433,1048576,256,0,0,8,8,16,64,0x0,0x7fc619c60ec0,1414621055462319,1414621055537142,1414621055556022,1414621055557449
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1299433,1299433,1048576,256,0,0,8,8,16,64,0x0,0x7fc619c60ec0,1414621055567848,1414621055576182,1414621055595542,1414621055596913
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1299433 1299433 1048576 256 0 0 8 8 16 64 0x0 0x7fc619c60ec0 1414621055400032 1414621055425142 1414621055449622 1414621055460846
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1299433 1299433 1048576 256 0 0 8 8 16 64 0x0 0x7fc619c60ec0 1414621055462319 1414621055537142 1414621055556022 1414621055557449
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1299433 1299433 1048576 256 0 0 8 8 16 64 0x0 0x7fc619c60ec0 1414621055567848 1414621055576182 1414621055595542 1414621055596913
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4190428,4190428,1048576,256,0,0,8,0,16,64,0x0,0x7f94512c0ec0,27378,27378,16384,65536,13161,1467220,1414432756404789,1414445656659785,1414445656680105,1414432772257328
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4190428,4190428,1048576,256,0,0,8,0,16,64,0x0,0x7f94512c0ec0,40627,40627,16384,65536,9359,1048624,1414432772278037,1414445656699625,1414445656715145,1414432772686309
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4190428,4190428,1048576,256,0,0,8,0,16,64,0x0,0x7f94512c0ec0,40747,40747,16384,65536,9106,1048600,1414432772715755,1414445656774345,1414445656790665,1414432772875517
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4190428 4190428 1048576 256 0 0 8 0 16 64 0x0 0x7f94512c0ec0 27378 27378 16384 65536 13161 1467220 1414432756404789 1414445656659785 1414445656680105 1414432772257328
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4190428 4190428 1048576 256 0 0 8 0 16 64 0x0 0x7f94512c0ec0 40627 40627 16384 65536 9359 1048624 1414432772278037 1414445656699625 1414445656715145 1414432772686309
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4190428 4190428 1048576 256 0 0 8 0 16 64 0x0 0x7f94512c0ec0 40747 40747 16384 65536 9106 1048600 1414432772715755 1414445656774345 1414445656790665 1414432772875517
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4190642,4190642,1048576,256,0,0,8,0,16,64,0x0,0x7f6c840ecec0,0,0,0,1414433244984736,1414445656659785,1414445656680105,1414433261546075
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4190642,4190642,1048576,256,0,0,8,0,16,64,0x0,0x7f6c840ecec0,0,0,0,1414433261564100,1414445656699625,1414445656715145,1414433261863666
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4190642,4190642,1048576,256,0,0,8,0,16,64,0x0,0x7f6c840ecec0,0,0,0,1414433261890016,1414445656774345,1414445656790665,1414433262043226
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4190642 4190642 1048576 256 0 0 8 0 16 64 0x0 0x7f6c840ecec0 0 0 0 1414433244984736 1414445656659785 1414445656680105 1414433261546075
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4190642 4190642 1048576 256 0 0 8 0 16 64 0x0 0x7f6c840ecec0 0 0 0 1414433261564100 1414445656699625 1414445656715145 1414433261863666
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4190642 4190642 1048576 256 0 0 8 0 16 64 0x0 0x7f6c840ecec0 0 0 0 1414433261890016 1414445656774345 1414445656790665 1414433262043226
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4190847,4190847,1048576,256,0,0,8,0,16,64,0x0,0x7f19e81d0ec0,65536,187730,20996312,1414433732354835,1414445656659785,1414445656680105,1414433748542708
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4190847,4190847,1048576,256,0,0,8,0,16,64,0x0,0x7f19e81d0ec0,65536,281018,31378144,1414433748560412,1414445656699625,1414445656715145,1414433748861932
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4190847,4190847,1048576,256,0,0,8,0,16,64,0x0,0x7f19e81d0ec0,65536,258024,28932200,1414433748888122,1414445656774345,1414445656790665,1414433749041842
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4190847 4190847 1048576 256 0 0 8 0 16 64 0x0 0x7f19e81d0ec0 65536 187730 20996312 1414433732354835 1414445656659785 1414445656680105 1414433748542708
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4190847 4190847 1048576 256 0 0 8 0 16 64 0x0 0x7f19e81d0ec0 65536 281018 31378144 1414433748560412 1414445656699625 1414445656715145 1414433748861932
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4190847 4190847 1048576 256 0 0 8 0 16 64 0x0 0x7f19e81d0ec0 65536 258024 28932200 1414433748888122 1414445656774345 1414445656790665 1414433749041842
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4191050,4191050,1048576,256,0,0,8,0,16,64,0x0,0x7fcf14168ec0,32768,306417,34321340,1414434204510803,1414445656659785,1414445656680105,1414434220573068
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4191050,4191050,1048576,256,0,0,8,0,16,64,0x0,0x7fcf14168ec0,32768,580620,65031856,1414434220596282,1414445656699625,1414445656715145,1414434220884157
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4191050,4191050,1048576,256,0,0,8,0,16,64,0x0,0x7fcf14168ec0,32768,586983,65717864,1414434220910687,1414445656774345,1414445656790665,1414434221039700
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4191050 4191050 1048576 256 0 0 8 0 16 64 0x0 0x7fcf14168ec0 32768 306417 34321340 1414434204510803 1414445656659785 1414445656680105 1414434220573068
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4191050 4191050 1048576 256 0 0 8 0 16 64 0x0 0x7fcf14168ec0 32768 580620 65031856 1414434220596282 1414445656699625 1414445656715145 1414434220884157
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4191050 4191050 1048576 256 0 0 8 0 16 64 0x0 0x7fcf14168ec0 32768 586983 65717864 1414434220910687 1414445656774345 1414445656790665 1414434221039700
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,4191237,4191237,1048576,256,0,0,8,0,16,64,0x0,0x7ffbe6b9cec0,27648,27648,11103,221192,16384,10666434,128024,0,43180252,1414434690769320,1414445656659785,1414445656680105,1414434706762735
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,4191237,4191237,1048576,256,0,0,8,0,16,64,0x0,0x7ffbe6b9cec0,41673,41673,14419,333392,16384,19327058,223251,0,77801560,1414434706790227,1414445656699625,1414445656715145,1414434707222514
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,4191237,4191237,1048576,256,0,0,8,0,16,64,0x0,0x7ffbe6b9cec0,41027,41027,13871,328224,16384,18889841,217781,0,76050388,1414434707257661,1414445656774345,1414445656790665,1414434707434395
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 4191237 4191237 1048576 256 0 0 8 0 16 64 0x0 0x7ffbe6b9cec0 27648 27648 11103 221192 16384 10666434 128024 0 43180252 1414434690769320 1414445656659785 1414445656680105 1414434706762735
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 4191237 4191237 1048576 256 0 0 8 0 16 64 0x0 0x7ffbe6b9cec0 41673 41673 14419 333392 16384 19327058 223251 0 77801560 1414434706790227 1414445656699625 1414445656715145 1414434707222514
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 4191237 4191237 1048576 256 0 0 8 0 16 64 0x0 0x7ffbe6b9cec0 41027 41027 13871 328224 16384 18889841 217781 0 76050388 1414434707257661 1414445656774345 1414445656790665 1414434707434395
@@ -0,0 +1,764 @@
Omniperf version: 2.0.0-RC1
Profiler choice: rocprofv1
Path: /home1/josantos/omniperf/tests/workloads/kernel/MI200
Target: MI200
Command: ./tests/vcopy -n 1048576 -b 256 -i 3
Kernel Selection: ['vecCopy']
Dispatch Selection: None
IP Blocks: All
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Collecting Performance Counters
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_IFETCH_LEVEL.txt
|-> [rocprof] RPL: on '240321_170500' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_IFETCH_LEVEL.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170500_4190268'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170500_4190268/input0_results_240321_170500'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170500_4190268/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 6 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, SQ_WAVES, SQ_IFETCH, SQ_IFETCH_LEVEL, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170500_4190268/input0_results_240321_170500
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_IFETCH_LEVEL.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_LDS.txt
|-> [rocprof] RPL: on '240321_170501' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_LDS.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170501_4190482'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170501_4190482/input0_results_240321_170501'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170501_4190482/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_LDS, SQ_INST_LEVEL_LDS, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170501_4190482/input0_results_240321_170501
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_INST_LEVEL_LDS.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_SMEM.txt
|-> [rocprof] RPL: on '240321_170501' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_SMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170501_4190684'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170501_4190684/input0_results_240321_170501'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170501_4190684/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_SMEM, SQ_INST_LEVEL_SMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170501_4190684/input0_results_240321_170501
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_INST_LEVEL_SMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_VMEM.txt
|-> [rocprof] RPL: on '240321_170502' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_INST_LEVEL_VMEM.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170502_4190890'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170502_4190890/input0_results_240321_170502'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170502_4190890/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQ_INSTS_VMEM, SQ_INST_LEVEL_VMEM, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170502_4190890/input0_results_240321_170502
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_INST_LEVEL_VMEM.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/SQ_LEVEL_WAVES.txt
|-> [rocprof] RPL: on '240321_170502' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/SQ_LEVEL_WAVES.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170502_4191077'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170502_4191077/input0_results_240321_170502'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170502_4191077/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 9 metrics
|-> [rocprof] GRBM_COUNT, GRBM_GUI_ACTIVE, CPC_ME1_BUSY_FOR_PACKET_DECODE, SQ_CYCLES, SQ_WAVES, SQ_WAVE_CYCLES, SQ_BUSY_CYCLES, SQ_LEVEL_WAVES, SQ_ACCUM_PREV_HIRES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170502_4191077/input0_results_240321_170502
|-> [rocprof] File 'tests/workloads/kernel/MI200/SQ_LEVEL_WAVES.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_0.txt
|-> [rocprof] RPL: on '240321_170502' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_0.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170502_4191279'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170502_4191279/input0_results_240321_170502'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170502_4191279/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 28 metrics
|-> [rocprof] SQ_CYCLES, SQ_BUSY_CYCLES, SQ_WAVES, SQ_INSTS_VALU_CVT, SQ_INSTS_VMEM_WR, SQ_INSTS_VMEM_RD, SQ_INSTS_VMEM, SQ_INSTS_SALU, GRBM_COUNT, GRBM_GUI_ACTIVE, TCP_GATE_EN1_sum, TCP_GATE_EN2_sum, TCP_TD_TCP_STALL_CYCLES_sum, TCP_TCR_TCP_STALL_CYCLES_sum, TA_TA_BUSY_sum, TA_BUFFER_WAVEFRONTS_sum, TD_TD_BUSY_sum, TD_TC_STALL_sum, SPI_CSN_WINDOW_VALID, SPI_CSN_BUSY, CPC_CPC_STAT_BUSY, CPC_CPC_STAT_IDLE, CPF_CPF_STAT_BUSY, CPF_CPF_STAT_STALL, TCC_CYCLE_sum, TCC_BUSY_sum, TCC_PROBE_sum, TCC_PROBE_ALL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170502_4191279/input0_results_240321_170502
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_0.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_1.txt
|-> [rocprof] RPL: on '240321_170503' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_1.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170503_4191481'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170503_4191481/input0_results_240321_170503'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170503_4191481/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 27 metrics
|-> [rocprof] SQ_INSTS_VSKIPPED, SQ_INSTS, SQ_INSTS_VALU, SQ_INSTS_VALU_ADD_F16, SQ_INSTS_VALU_MUL_F16, SQ_INSTS_VALU_FMA_F16, SQ_INSTS_VALU_TRANS_F16, SQ_INSTS_VALU_ADD_F32, GRBM_SPI_BUSY, TCP_READ_TAGCONFLICT_STALL_CYCLES_sum, TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum, TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum, TCP_TA_TCP_STATE_READ_sum, TA_BUFFER_READ_WAVEFRONTS_sum, TA_BUFFER_WRITE_WAVEFRONTS_sum, TD_SPI_STALL_sum, TD_LOAD_WAVEFRONT_sum, SPI_CSN_NUM_THREADGROUPS, SPI_CSN_WAVE, CPC_CPC_TCIU_BUSY, CPC_CPC_TCIU_IDLE, CPF_CPF_TCIU_BUSY, CPF_CPF_TCIU_STALL, TCC_NC_REQ_sum, TCC_UC_REQ_sum, TCC_CC_REQ_sum, TCC_RW_REQ_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170503_4191481/input0_results_240321_170503
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_1.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_10.txt
|-> [rocprof] RPL: on '240321_170503' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_10.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170503_4191685'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170503_4191685/input0_results_240321_170503'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170503_4191685/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] SQC_TC_DATA_WRITE_REQ, SQC_TC_DATA_ATOMIC_REQ, SQC_TC_STALL, SQC_TC_REQ, SQC_DCACHE_REQ_READ_16, SQC_ICACHE_REQ, SQC_ICACHE_HITS, SQC_ICACHE_MISSES
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170503_4191685/input0_results_240321_170503
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_10.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_11.txt
|-> [rocprof] RPL: on '240321_170504' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_11.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170504_4191871'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170504_4191871/input0_results_240321_170504'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170504_4191871/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 8 metrics
|-> [rocprof] SQC_ICACHE_MISSES_DUPLICATE, SQC_DCACHE_INPUT_VALID_READYB, SQC_DCACHE_ATOMIC, SQC_DCACHE_REQ_READ_8, SQC_DCACHE_REQ, SQC_DCACHE_HITS, SQC_DCACHE_MISSES, SQC_DCACHE_MISSES_DUPLICATE
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170504_4191871/input0_results_240321_170504
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_11.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_12.txt
|-> [rocprof] RPL: on '240321_170504' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_12.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170504_4192073'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170504_4192073/input0_results_240321_170504'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170504_4192073/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 3 metrics
|-> [rocprof] SQC_DCACHE_REQ_READ_1, SQC_DCACHE_REQ_READ_2, SQC_DCACHE_REQ_READ_4
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170504_4192073/input0_results_240321_170504
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_12.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_13.txt
|-> [rocprof] RPL: on '240321_170505' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_13.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170505_4192275'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170505_4192275/input0_results_240321_170505'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170505_4192275/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_ATOMIC[0], TCC_CYCLE[0], TCC_EA_ATOMIC[0], TCC_EA_ATOMIC_LEVEL[0], TCC_ATOMIC[1], TCC_CYCLE[1], TCC_EA_ATOMIC[1], TCC_EA_ATOMIC_LEVEL[1], TCC_ATOMIC[2], TCC_CYCLE[2], TCC_EA_ATOMIC[2], TCC_EA_ATOMIC_LEVEL[2], TCC_ATOMIC[3], TCC_CYCLE[3], TCC_EA_ATOMIC[3], TCC_EA_ATOMIC_LEVEL[3], TCC_ATOMIC[4], TCC_CYCLE[4], TCC_EA_ATOMIC[4], TCC_EA_ATOMIC_LEVEL[4], TCC_ATOMIC[5], TCC_CYCLE[5], TCC_EA_ATOMIC[5], TCC_EA_ATOMIC_LEVEL[5], TCC_ATOMIC[6], TCC_CYCLE[6], TCC_EA_ATOMIC[6], TCC_EA_ATOMIC_LEVEL[6], TCC_ATOMIC[7], TCC_CYCLE[7], TCC_EA_ATOMIC[7], TCC_EA_ATOMIC_LEVEL[7], TCC_ATOMIC[8], TCC_CYCLE[8], TCC_EA_ATOMIC[8], TCC_EA_ATOMIC_LEVEL[8], TCC_ATOMIC[9], TCC_CYCLE[9], TCC_EA_ATOMIC[9], TCC_EA_ATOMIC_LEVEL[9], TCC_ATOMIC[10], TCC_CYCLE[10], TCC_EA_ATOMIC[10], TCC_EA_ATOMIC_LEVEL[10], TCC_ATOMIC[11], TCC_CYCLE[11], TCC_EA_ATOMIC[11], TCC_EA_ATOMIC_LEVEL[11], TCC_ATOMIC[12], TCC_CYCLE[12], TCC_EA_ATOMIC[12], TCC_EA_ATOMIC_LEVEL[12], TCC_ATOMIC[13], TCC_CYCLE[13], TCC_EA_ATOMIC[13], TCC_EA_ATOMIC_LEVEL[13], TCC_ATOMIC[14], TCC_CYCLE[14], TCC_EA_ATOMIC[14], TCC_EA_ATOMIC_LEVEL[14], TCC_ATOMIC[15], TCC_CYCLE[15], TCC_EA_ATOMIC[15], TCC_EA_ATOMIC_LEVEL[15], TCC_ATOMIC[16], TCC_CYCLE[16], TCC_EA_ATOMIC[16], TCC_EA_ATOMIC_LEVEL[16], TCC_ATOMIC[17], TCC_CYCLE[17], TCC_EA_ATOMIC[17], TCC_EA_ATOMIC_LEVEL[17], TCC_ATOMIC[18], TCC_CYCLE[18], TCC_EA_ATOMIC[18], TCC_EA_ATOMIC_LEVEL[18], TCC_ATOMIC[19], TCC_CYCLE[19], TCC_EA_ATOMIC[19], TCC_EA_ATOMIC_LEVEL[19], TCC_ATOMIC[20], TCC_CYCLE[20], TCC_EA_ATOMIC[20], TCC_EA_ATOMIC_LEVEL[20], TCC_ATOMIC[21], TCC_CYCLE[21], TCC_EA_ATOMIC[21], TCC_EA_ATOMIC_LEVEL[21], TCC_ATOMIC[22], TCC_CYCLE[22], TCC_EA_ATOMIC[22], TCC_EA_ATOMIC_LEVEL[22], TCC_ATOMIC[23], TCC_CYCLE[23], TCC_EA_ATOMIC[23], TCC_EA_ATOMIC_LEVEL[23], TCC_ATOMIC[24], TCC_CYCLE[24], TCC_EA_ATOMIC[24], TCC_EA_ATOMIC_LEVEL[24], TCC_ATOMIC[25], TCC_CYCLE[25], TCC_EA_ATOMIC[25], TCC_EA_ATOMIC_LEVEL[25], TCC_ATOMIC[26], TCC_CYCLE[26], TCC_EA_ATOMIC[26], TCC_EA_ATOMIC_LEVEL[26], TCC_ATOMIC[27], TCC_CYCLE[27], TCC_EA_ATOMIC[27], TCC_EA_ATOMIC_LEVEL[27], TCC_ATOMIC[28], TCC_CYCLE[28], TCC_EA_ATOMIC[28], TCC_EA_ATOMIC_LEVEL[28], TCC_ATOMIC[29], TCC_CYCLE[29], TCC_EA_ATOMIC[29], TCC_EA_ATOMIC_LEVEL[29], TCC_ATOMIC[30], TCC_CYCLE[30], TCC_EA_ATOMIC[30], TCC_EA_ATOMIC_LEVEL[30], TCC_ATOMIC[31], TCC_CYCLE[31], TCC_EA_ATOMIC[31], TCC_EA_ATOMIC_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170505_4192275/input0_results_240321_170505
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_13.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_14.txt
|-> [rocprof] RPL: on '240321_170506' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_14.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170506_4192465'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170506_4192465/input0_results_240321_170506'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170506_4192465/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ[0], TCC_EA_RDREQ_32B[0], TCC_EA_RDREQ_DRAM_CREDIT_STALL[0], TCC_EA_RDREQ_GMI_CREDIT_STALL[0], TCC_EA_RDREQ[1], TCC_EA_RDREQ_32B[1], TCC_EA_RDREQ_DRAM_CREDIT_STALL[1], TCC_EA_RDREQ_GMI_CREDIT_STALL[1], TCC_EA_RDREQ[2], TCC_EA_RDREQ_32B[2], TCC_EA_RDREQ_DRAM_CREDIT_STALL[2], TCC_EA_RDREQ_GMI_CREDIT_STALL[2], TCC_EA_RDREQ[3], TCC_EA_RDREQ_32B[3], TCC_EA_RDREQ_DRAM_CREDIT_STALL[3], TCC_EA_RDREQ_GMI_CREDIT_STALL[3], TCC_EA_RDREQ[4], TCC_EA_RDREQ_32B[4], TCC_EA_RDREQ_DRAM_CREDIT_STALL[4], TCC_EA_RDREQ_GMI_CREDIT_STALL[4], TCC_EA_RDREQ[5], TCC_EA_RDREQ_32B[5], TCC_EA_RDREQ_DRAM_CREDIT_STALL[5], TCC_EA_RDREQ_GMI_CREDIT_STALL[5], TCC_EA_RDREQ[6], TCC_EA_RDREQ_32B[6], TCC_EA_RDREQ_DRAM_CREDIT_STALL[6], TCC_EA_RDREQ_GMI_CREDIT_STALL[6], TCC_EA_RDREQ[7], TCC_EA_RDREQ_32B[7], TCC_EA_RDREQ_DRAM_CREDIT_STALL[7], TCC_EA_RDREQ_GMI_CREDIT_STALL[7], TCC_EA_RDREQ[8], TCC_EA_RDREQ_32B[8], TCC_EA_RDREQ_DRAM_CREDIT_STALL[8], TCC_EA_RDREQ_GMI_CREDIT_STALL[8], TCC_EA_RDREQ[9], TCC_EA_RDREQ_32B[9], TCC_EA_RDREQ_DRAM_CREDIT_STALL[9], TCC_EA_RDREQ_GMI_CREDIT_STALL[9], TCC_EA_RDREQ[10], TCC_EA_RDREQ_32B[10], TCC_EA_RDREQ_DRAM_CREDIT_STALL[10], TCC_EA_RDREQ_GMI_CREDIT_STALL[10], TCC_EA_RDREQ[11], TCC_EA_RDREQ_32B[11], TCC_EA_RDREQ_DRAM_CREDIT_STALL[11], TCC_EA_RDREQ_GMI_CREDIT_STALL[11], TCC_EA_RDREQ[12], TCC_EA_RDREQ_32B[12], TCC_EA_RDREQ_DRAM_CREDIT_STALL[12], TCC_EA_RDREQ_GMI_CREDIT_STALL[12], TCC_EA_RDREQ[13], TCC_EA_RDREQ_32B[13], TCC_EA_RDREQ_DRAM_CREDIT_STALL[13], TCC_EA_RDREQ_GMI_CREDIT_STALL[13], TCC_EA_RDREQ[14], TCC_EA_RDREQ_32B[14], TCC_EA_RDREQ_DRAM_CREDIT_STALL[14], TCC_EA_RDREQ_GMI_CREDIT_STALL[14], TCC_EA_RDREQ[15], TCC_EA_RDREQ_32B[15], TCC_EA_RDREQ_DRAM_CREDIT_STALL[15], TCC_EA_RDREQ_GMI_CREDIT_STALL[15], TCC_EA_RDREQ[16], TCC_EA_RDREQ_32B[16], TCC_EA_RDREQ_DRAM_CREDIT_STALL[16], TCC_EA_RDREQ_GMI_CREDIT_STALL[16], TCC_EA_RDREQ[17], TCC_EA_RDREQ_32B[17], TCC_EA_RDREQ_DRAM_CREDIT_STALL[17], TCC_EA_RDREQ_GMI_CREDIT_STALL[17], TCC_EA_RDREQ[18], TCC_EA_RDREQ_32B[18], TCC_EA_RDREQ_DRAM_CREDIT_STALL[18], TCC_EA_RDREQ_GMI_CREDIT_STALL[18], TCC_EA_RDREQ[19], TCC_EA_RDREQ_32B[19], TCC_EA_RDREQ_DRAM_CREDIT_STALL[19], TCC_EA_RDREQ_GMI_CREDIT_STALL[19], TCC_EA_RDREQ[20], TCC_EA_RDREQ_32B[20], TCC_EA_RDREQ_DRAM_CREDIT_STALL[20], TCC_EA_RDREQ_GMI_CREDIT_STALL[20], TCC_EA_RDREQ[21], TCC_EA_RDREQ_32B[21], TCC_EA_RDREQ_DRAM_CREDIT_STALL[21], TCC_EA_RDREQ_GMI_CREDIT_STALL[21], TCC_EA_RDREQ[22], TCC_EA_RDREQ_32B[22], TCC_EA_RDREQ_DRAM_CREDIT_STALL[22], TCC_EA_RDREQ_GMI_CREDIT_STALL[22], TCC_EA_RDREQ[23], TCC_EA_RDREQ_32B[23], TCC_EA_RDREQ_DRAM_CREDIT_STALL[23], TCC_EA_RDREQ_GMI_CREDIT_STALL[23], TCC_EA_RDREQ[24], TCC_EA_RDREQ_32B[24], TCC_EA_RDREQ_DRAM_CREDIT_STALL[24], TCC_EA_RDREQ_GMI_CREDIT_STALL[24], TCC_EA_RDREQ[25], TCC_EA_RDREQ_32B[25], TCC_EA_RDREQ_DRAM_CREDIT_STALL[25], TCC_EA_RDREQ_GMI_CREDIT_STALL[25], TCC_EA_RDREQ[26], TCC_EA_RDREQ_32B[26], TCC_EA_RDREQ_DRAM_CREDIT_STALL[26], TCC_EA_RDREQ_GMI_CREDIT_STALL[26], TCC_EA_RDREQ[27], TCC_EA_RDREQ_32B[27], TCC_EA_RDREQ_DRAM_CREDIT_STALL[27], TCC_EA_RDREQ_GMI_CREDIT_STALL[27], TCC_EA_RDREQ[28], TCC_EA_RDREQ_32B[28], TCC_EA_RDREQ_DRAM_CREDIT_STALL[28], TCC_EA_RDREQ_GMI_CREDIT_STALL[28], TCC_EA_RDREQ[29], TCC_EA_RDREQ_32B[29], TCC_EA_RDREQ_DRAM_CREDIT_STALL[29], TCC_EA_RDREQ_GMI_CREDIT_STALL[29], TCC_EA_RDREQ[30], TCC_EA_RDREQ_32B[30], TCC_EA_RDREQ_DRAM_CREDIT_STALL[30], TCC_EA_RDREQ_GMI_CREDIT_STALL[30], TCC_EA_RDREQ[31], TCC_EA_RDREQ_32B[31], TCC_EA_RDREQ_DRAM_CREDIT_STALL[31], TCC_EA_RDREQ_GMI_CREDIT_STALL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170506_4192465/input0_results_240321_170506
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_14.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_15.txt
|-> [rocprof] RPL: on '240321_170506' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_15.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170506_4192664'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170506_4192664/input0_results_240321_170506'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170506_4192664/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_RDREQ_IO_CREDIT_STALL[0], TCC_EA_RDREQ_LEVEL[0], TCC_EA_WRREQ[0], TCC_EA_WRREQ_64B[0], TCC_EA_RDREQ_IO_CREDIT_STALL[1], TCC_EA_RDREQ_LEVEL[1], TCC_EA_WRREQ[1], TCC_EA_WRREQ_64B[1], TCC_EA_RDREQ_IO_CREDIT_STALL[2], TCC_EA_RDREQ_LEVEL[2], TCC_EA_WRREQ[2], TCC_EA_WRREQ_64B[2], TCC_EA_RDREQ_IO_CREDIT_STALL[3], TCC_EA_RDREQ_LEVEL[3], TCC_EA_WRREQ[3], TCC_EA_WRREQ_64B[3], TCC_EA_RDREQ_IO_CREDIT_STALL[4], TCC_EA_RDREQ_LEVEL[4], TCC_EA_WRREQ[4], TCC_EA_WRREQ_64B[4], TCC_EA_RDREQ_IO_CREDIT_STALL[5], TCC_EA_RDREQ_LEVEL[5], TCC_EA_WRREQ[5], TCC_EA_WRREQ_64B[5], TCC_EA_RDREQ_IO_CREDIT_STALL[6], TCC_EA_RDREQ_LEVEL[6], TCC_EA_WRREQ[6], TCC_EA_WRREQ_64B[6], TCC_EA_RDREQ_IO_CREDIT_STALL[7], TCC_EA_RDREQ_LEVEL[7], TCC_EA_WRREQ[7], TCC_EA_WRREQ_64B[7], TCC_EA_RDREQ_IO_CREDIT_STALL[8], TCC_EA_RDREQ_LEVEL[8], TCC_EA_WRREQ[8], TCC_EA_WRREQ_64B[8], TCC_EA_RDREQ_IO_CREDIT_STALL[9], TCC_EA_RDREQ_LEVEL[9], TCC_EA_WRREQ[9], TCC_EA_WRREQ_64B[9], TCC_EA_RDREQ_IO_CREDIT_STALL[10], TCC_EA_RDREQ_LEVEL[10], TCC_EA_WRREQ[10], TCC_EA_WRREQ_64B[10], TCC_EA_RDREQ_IO_CREDIT_STALL[11], TCC_EA_RDREQ_LEVEL[11], TCC_EA_WRREQ[11], TCC_EA_WRREQ_64B[11], TCC_EA_RDREQ_IO_CREDIT_STALL[12], TCC_EA_RDREQ_LEVEL[12], TCC_EA_WRREQ[12], TCC_EA_WRREQ_64B[12], TCC_EA_RDREQ_IO_CREDIT_STALL[13], TCC_EA_RDREQ_LEVEL[13], TCC_EA_WRREQ[13], TCC_EA_WRREQ_64B[13], TCC_EA_RDREQ_IO_CREDIT_STALL[14], TCC_EA_RDREQ_LEVEL[14], TCC_EA_WRREQ[14], TCC_EA_WRREQ_64B[14], TCC_EA_RDREQ_IO_CREDIT_STALL[15], TCC_EA_RDREQ_LEVEL[15], TCC_EA_WRREQ[15], TCC_EA_WRREQ_64B[15], TCC_EA_RDREQ_IO_CREDIT_STALL[16], TCC_EA_RDREQ_LEVEL[16], TCC_EA_WRREQ[16], TCC_EA_WRREQ_64B[16], TCC_EA_RDREQ_IO_CREDIT_STALL[17], TCC_EA_RDREQ_LEVEL[17], TCC_EA_WRREQ[17], TCC_EA_WRREQ_64B[17], TCC_EA_RDREQ_IO_CREDIT_STALL[18], TCC_EA_RDREQ_LEVEL[18], TCC_EA_WRREQ[18], TCC_EA_WRREQ_64B[18], TCC_EA_RDREQ_IO_CREDIT_STALL[19], TCC_EA_RDREQ_LEVEL[19], TCC_EA_WRREQ[19], TCC_EA_WRREQ_64B[19], TCC_EA_RDREQ_IO_CREDIT_STALL[20], TCC_EA_RDREQ_LEVEL[20], TCC_EA_WRREQ[20], TCC_EA_WRREQ_64B[20], TCC_EA_RDREQ_IO_CREDIT_STALL[21], TCC_EA_RDREQ_LEVEL[21], TCC_EA_WRREQ[21], TCC_EA_WRREQ_64B[21], TCC_EA_RDREQ_IO_CREDIT_STALL[22], TCC_EA_RDREQ_LEVEL[22], TCC_EA_WRREQ[22], TCC_EA_WRREQ_64B[22], TCC_EA_RDREQ_IO_CREDIT_STALL[23], TCC_EA_RDREQ_LEVEL[23], TCC_EA_WRREQ[23], TCC_EA_WRREQ_64B[23], TCC_EA_RDREQ_IO_CREDIT_STALL[24], TCC_EA_RDREQ_LEVEL[24], TCC_EA_WRREQ[24], TCC_EA_WRREQ_64B[24], TCC_EA_RDREQ_IO_CREDIT_STALL[25], TCC_EA_RDREQ_LEVEL[25], TCC_EA_WRREQ[25], TCC_EA_WRREQ_64B[25], TCC_EA_RDREQ_IO_CREDIT_STALL[26], TCC_EA_RDREQ_LEVEL[26], TCC_EA_WRREQ[26], TCC_EA_WRREQ_64B[26], TCC_EA_RDREQ_IO_CREDIT_STALL[27], TCC_EA_RDREQ_LEVEL[27], TCC_EA_WRREQ[27], TCC_EA_WRREQ_64B[27], TCC_EA_RDREQ_IO_CREDIT_STALL[28], TCC_EA_RDREQ_LEVEL[28], TCC_EA_WRREQ[28], TCC_EA_WRREQ_64B[28], TCC_EA_RDREQ_IO_CREDIT_STALL[29], TCC_EA_RDREQ_LEVEL[29], TCC_EA_WRREQ[29], TCC_EA_WRREQ_64B[29], TCC_EA_RDREQ_IO_CREDIT_STALL[30], TCC_EA_RDREQ_LEVEL[30], TCC_EA_WRREQ[30], TCC_EA_WRREQ_64B[30], TCC_EA_RDREQ_IO_CREDIT_STALL[31], TCC_EA_RDREQ_LEVEL[31], TCC_EA_WRREQ[31], TCC_EA_WRREQ_64B[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170506_4192664/input0_results_240321_170506
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_15.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_16.txt
|-> [rocprof] RPL: on '240321_170507' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_16.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170507_4192866'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170507_4192866/input0_results_240321_170507'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170507_4192866/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_EA_WRREQ_DRAM_CREDIT_STALL[0], TCC_EA_WRREQ_GMI_CREDIT_STALL[0], TCC_EA_WRREQ_IO_CREDIT_STALL[0], TCC_EA_WRREQ_LEVEL[0], TCC_EA_WRREQ_DRAM_CREDIT_STALL[1], TCC_EA_WRREQ_GMI_CREDIT_STALL[1], TCC_EA_WRREQ_IO_CREDIT_STALL[1], TCC_EA_WRREQ_LEVEL[1], TCC_EA_WRREQ_DRAM_CREDIT_STALL[2], TCC_EA_WRREQ_GMI_CREDIT_STALL[2], TCC_EA_WRREQ_IO_CREDIT_STALL[2], TCC_EA_WRREQ_LEVEL[2], TCC_EA_WRREQ_DRAM_CREDIT_STALL[3], TCC_EA_WRREQ_GMI_CREDIT_STALL[3], TCC_EA_WRREQ_IO_CREDIT_STALL[3], TCC_EA_WRREQ_LEVEL[3], TCC_EA_WRREQ_DRAM_CREDIT_STALL[4], TCC_EA_WRREQ_GMI_CREDIT_STALL[4], TCC_EA_WRREQ_IO_CREDIT_STALL[4], TCC_EA_WRREQ_LEVEL[4], TCC_EA_WRREQ_DRAM_CREDIT_STALL[5], TCC_EA_WRREQ_GMI_CREDIT_STALL[5], TCC_EA_WRREQ_IO_CREDIT_STALL[5], TCC_EA_WRREQ_LEVEL[5], TCC_EA_WRREQ_DRAM_CREDIT_STALL[6], TCC_EA_WRREQ_GMI_CREDIT_STALL[6], TCC_EA_WRREQ_IO_CREDIT_STALL[6], TCC_EA_WRREQ_LEVEL[6], TCC_EA_WRREQ_DRAM_CREDIT_STALL[7], TCC_EA_WRREQ_GMI_CREDIT_STALL[7], TCC_EA_WRREQ_IO_CREDIT_STALL[7], TCC_EA_WRREQ_LEVEL[7], TCC_EA_WRREQ_DRAM_CREDIT_STALL[8], TCC_EA_WRREQ_GMI_CREDIT_STALL[8], TCC_EA_WRREQ_IO_CREDIT_STALL[8], TCC_EA_WRREQ_LEVEL[8], TCC_EA_WRREQ_DRAM_CREDIT_STALL[9], TCC_EA_WRREQ_GMI_CREDIT_STALL[9], TCC_EA_WRREQ_IO_CREDIT_STALL[9], TCC_EA_WRREQ_LEVEL[9], TCC_EA_WRREQ_DRAM_CREDIT_STALL[10], TCC_EA_WRREQ_GMI_CREDIT_STALL[10], TCC_EA_WRREQ_IO_CREDIT_STALL[10], TCC_EA_WRREQ_LEVEL[10], TCC_EA_WRREQ_DRAM_CREDIT_STALL[11], TCC_EA_WRREQ_GMI_CREDIT_STALL[11], TCC_EA_WRREQ_IO_CREDIT_STALL[11], TCC_EA_WRREQ_LEVEL[11], TCC_EA_WRREQ_DRAM_CREDIT_STALL[12], TCC_EA_WRREQ_GMI_CREDIT_STALL[12], TCC_EA_WRREQ_IO_CREDIT_STALL[12], TCC_EA_WRREQ_LEVEL[12], TCC_EA_WRREQ_DRAM_CREDIT_STALL[13], TCC_EA_WRREQ_GMI_CREDIT_STALL[13], TCC_EA_WRREQ_IO_CREDIT_STALL[13], TCC_EA_WRREQ_LEVEL[13], TCC_EA_WRREQ_DRAM_CREDIT_STALL[14], TCC_EA_WRREQ_GMI_CREDIT_STALL[14], TCC_EA_WRREQ_IO_CREDIT_STALL[14], TCC_EA_WRREQ_LEVEL[14], TCC_EA_WRREQ_DRAM_CREDIT_STALL[15], TCC_EA_WRREQ_GMI_CREDIT_STALL[15], TCC_EA_WRREQ_IO_CREDIT_STALL[15], TCC_EA_WRREQ_LEVEL[15], TCC_EA_WRREQ_DRAM_CREDIT_STALL[16], TCC_EA_WRREQ_GMI_CREDIT_STALL[16], TCC_EA_WRREQ_IO_CREDIT_STALL[16], TCC_EA_WRREQ_LEVEL[16], TCC_EA_WRREQ_DRAM_CREDIT_STALL[17], TCC_EA_WRREQ_GMI_CREDIT_STALL[17], TCC_EA_WRREQ_IO_CREDIT_STALL[17], TCC_EA_WRREQ_LEVEL[17], TCC_EA_WRREQ_DRAM_CREDIT_STALL[18], TCC_EA_WRREQ_GMI_CREDIT_STALL[18], TCC_EA_WRREQ_IO_CREDIT_STALL[18], TCC_EA_WRREQ_LEVEL[18], TCC_EA_WRREQ_DRAM_CREDIT_STALL[19], TCC_EA_WRREQ_GMI_CREDIT_STALL[19], TCC_EA_WRREQ_IO_CREDIT_STALL[19], TCC_EA_WRREQ_LEVEL[19], TCC_EA_WRREQ_DRAM_CREDIT_STALL[20], TCC_EA_WRREQ_GMI_CREDIT_STALL[20], TCC_EA_WRREQ_IO_CREDIT_STALL[20], TCC_EA_WRREQ_LEVEL[20], TCC_EA_WRREQ_DRAM_CREDIT_STALL[21], TCC_EA_WRREQ_GMI_CREDIT_STALL[21], TCC_EA_WRREQ_IO_CREDIT_STALL[21], TCC_EA_WRREQ_LEVEL[21], TCC_EA_WRREQ_DRAM_CREDIT_STALL[22], TCC_EA_WRREQ_GMI_CREDIT_STALL[22], TCC_EA_WRREQ_IO_CREDIT_STALL[22], TCC_EA_WRREQ_LEVEL[22], TCC_EA_WRREQ_DRAM_CREDIT_STALL[23], TCC_EA_WRREQ_GMI_CREDIT_STALL[23], TCC_EA_WRREQ_IO_CREDIT_STALL[23], TCC_EA_WRREQ_LEVEL[23], TCC_EA_WRREQ_DRAM_CREDIT_STALL[24], TCC_EA_WRREQ_GMI_CREDIT_STALL[24], TCC_EA_WRREQ_IO_CREDIT_STALL[24], TCC_EA_WRREQ_LEVEL[24], TCC_EA_WRREQ_DRAM_CREDIT_STALL[25], TCC_EA_WRREQ_GMI_CREDIT_STALL[25], TCC_EA_WRREQ_IO_CREDIT_STALL[25], TCC_EA_WRREQ_LEVEL[25], TCC_EA_WRREQ_DRAM_CREDIT_STALL[26], TCC_EA_WRREQ_GMI_CREDIT_STALL[26], TCC_EA_WRREQ_IO_CREDIT_STALL[26], TCC_EA_WRREQ_LEVEL[26], TCC_EA_WRREQ_DRAM_CREDIT_STALL[27], TCC_EA_WRREQ_GMI_CREDIT_STALL[27], TCC_EA_WRREQ_IO_CREDIT_STALL[27], TCC_EA_WRREQ_LEVEL[27], TCC_EA_WRREQ_DRAM_CREDIT_STALL[28], TCC_EA_WRREQ_GMI_CREDIT_STALL[28], TCC_EA_WRREQ_IO_CREDIT_STALL[28], TCC_EA_WRREQ_LEVEL[28], TCC_EA_WRREQ_DRAM_CREDIT_STALL[29], TCC_EA_WRREQ_GMI_CREDIT_STALL[29], TCC_EA_WRREQ_IO_CREDIT_STALL[29], TCC_EA_WRREQ_LEVEL[29], TCC_EA_WRREQ_DRAM_CREDIT_STALL[30], TCC_EA_WRREQ_GMI_CREDIT_STALL[30], TCC_EA_WRREQ_IO_CREDIT_STALL[30], TCC_EA_WRREQ_LEVEL[30], TCC_EA_WRREQ_DRAM_CREDIT_STALL[31], TCC_EA_WRREQ_GMI_CREDIT_STALL[31], TCC_EA_WRREQ_IO_CREDIT_STALL[31], TCC_EA_WRREQ_LEVEL[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170507_4192866/input0_results_240321_170507
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_16.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_17.txt
|-> [rocprof] RPL: on '240321_170508' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_17.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170508_4193068'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170508_4193068/input0_results_240321_170508'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170508_4193068/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 128 metrics
|-> [rocprof] TCC_HIT[0], TCC_MISS[0], TCC_READ[0], TCC_REQ[0], TCC_HIT[1], TCC_MISS[1], TCC_READ[1], TCC_REQ[1], TCC_HIT[2], TCC_MISS[2], TCC_READ[2], TCC_REQ[2], TCC_HIT[3], TCC_MISS[3], TCC_READ[3], TCC_REQ[3], TCC_HIT[4], TCC_MISS[4], TCC_READ[4], TCC_REQ[4], TCC_HIT[5], TCC_MISS[5], TCC_READ[5], TCC_REQ[5], TCC_HIT[6], TCC_MISS[6], TCC_READ[6], TCC_REQ[6], TCC_HIT[7], TCC_MISS[7], TCC_READ[7], TCC_REQ[7], TCC_HIT[8], TCC_MISS[8], TCC_READ[8], TCC_REQ[8], TCC_HIT[9], TCC_MISS[9], TCC_READ[9], TCC_REQ[9], TCC_HIT[10], TCC_MISS[10], TCC_READ[10], TCC_REQ[10], TCC_HIT[11], TCC_MISS[11], TCC_READ[11], TCC_REQ[11], TCC_HIT[12], TCC_MISS[12], TCC_READ[12], TCC_REQ[12], TCC_HIT[13], TCC_MISS[13], TCC_READ[13], TCC_REQ[13], TCC_HIT[14], TCC_MISS[14], TCC_READ[14], TCC_REQ[14], TCC_HIT[15], TCC_MISS[15], TCC_READ[15], TCC_REQ[15], TCC_HIT[16], TCC_MISS[16], TCC_READ[16], TCC_REQ[16], TCC_HIT[17], TCC_MISS[17], TCC_READ[17], TCC_REQ[17], TCC_HIT[18], TCC_MISS[18], TCC_READ[18], TCC_REQ[18], TCC_HIT[19], TCC_MISS[19], TCC_READ[19], TCC_REQ[19], TCC_HIT[20], TCC_MISS[20], TCC_READ[20], TCC_REQ[20], TCC_HIT[21], TCC_MISS[21], TCC_READ[21], TCC_REQ[21], TCC_HIT[22], TCC_MISS[22], TCC_READ[22], TCC_REQ[22], TCC_HIT[23], TCC_MISS[23], TCC_READ[23], TCC_REQ[23], TCC_HIT[24], TCC_MISS[24], TCC_READ[24], TCC_REQ[24], TCC_HIT[25], TCC_MISS[25], TCC_READ[25], TCC_REQ[25], TCC_HIT[26], TCC_MISS[26], TCC_READ[26], TCC_REQ[26], TCC_HIT[27], TCC_MISS[27], TCC_READ[27], TCC_REQ[27], TCC_HIT[28], TCC_MISS[28], TCC_READ[28], TCC_REQ[28], TCC_HIT[29], TCC_MISS[29], TCC_READ[29], TCC_REQ[29], TCC_HIT[30], TCC_MISS[30], TCC_READ[30], TCC_REQ[30], TCC_HIT[31], TCC_MISS[31], TCC_READ[31], TCC_REQ[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170508_4193068/input0_results_240321_170508
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_17.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_18.txt
|-> [rocprof] RPL: on '240321_170508' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_18.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170508_4193269'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170508_4193269/input0_results_240321_170508'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170508_4193269/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 96 metrics
|-> [rocprof] TCC_RW_REQ[0], TCC_TOO_MANY_EA_WRREQS_STALL[0], TCC_WRITE[0], TCC_RW_REQ[1], TCC_TOO_MANY_EA_WRREQS_STALL[1], TCC_WRITE[1], TCC_RW_REQ[2], TCC_TOO_MANY_EA_WRREQS_STALL[2], TCC_WRITE[2], TCC_RW_REQ[3], TCC_TOO_MANY_EA_WRREQS_STALL[3], TCC_WRITE[3], TCC_RW_REQ[4], TCC_TOO_MANY_EA_WRREQS_STALL[4], TCC_WRITE[4], TCC_RW_REQ[5], TCC_TOO_MANY_EA_WRREQS_STALL[5], TCC_WRITE[5], TCC_RW_REQ[6], TCC_TOO_MANY_EA_WRREQS_STALL[6], TCC_WRITE[6], TCC_RW_REQ[7], TCC_TOO_MANY_EA_WRREQS_STALL[7], TCC_WRITE[7], TCC_RW_REQ[8], TCC_TOO_MANY_EA_WRREQS_STALL[8], TCC_WRITE[8], TCC_RW_REQ[9], TCC_TOO_MANY_EA_WRREQS_STALL[9], TCC_WRITE[9], TCC_RW_REQ[10], TCC_TOO_MANY_EA_WRREQS_STALL[10], TCC_WRITE[10], TCC_RW_REQ[11], TCC_TOO_MANY_EA_WRREQS_STALL[11], TCC_WRITE[11], TCC_RW_REQ[12], TCC_TOO_MANY_EA_WRREQS_STALL[12], TCC_WRITE[12], TCC_RW_REQ[13], TCC_TOO_MANY_EA_WRREQS_STALL[13], TCC_WRITE[13], TCC_RW_REQ[14], TCC_TOO_MANY_EA_WRREQS_STALL[14], TCC_WRITE[14], TCC_RW_REQ[15], TCC_TOO_MANY_EA_WRREQS_STALL[15], TCC_WRITE[15], TCC_RW_REQ[16], TCC_TOO_MANY_EA_WRREQS_STALL[16], TCC_WRITE[16], TCC_RW_REQ[17], TCC_TOO_MANY_EA_WRREQS_STALL[17], TCC_WRITE[17], TCC_RW_REQ[18], TCC_TOO_MANY_EA_WRREQS_STALL[18], TCC_WRITE[18], TCC_RW_REQ[19], TCC_TOO_MANY_EA_WRREQS_STALL[19], TCC_WRITE[19], TCC_RW_REQ[20], TCC_TOO_MANY_EA_WRREQS_STALL[20], TCC_WRITE[20], TCC_RW_REQ[21], TCC_TOO_MANY_EA_WRREQS_STALL[21], TCC_WRITE[21], TCC_RW_REQ[22], TCC_TOO_MANY_EA_WRREQS_STALL[22], TCC_WRITE[22], TCC_RW_REQ[23], TCC_TOO_MANY_EA_WRREQS_STALL[23], TCC_WRITE[23], TCC_RW_REQ[24], TCC_TOO_MANY_EA_WRREQS_STALL[24], TCC_WRITE[24], TCC_RW_REQ[25], TCC_TOO_MANY_EA_WRREQS_STALL[25], TCC_WRITE[25], TCC_RW_REQ[26], TCC_TOO_MANY_EA_WRREQS_STALL[26], TCC_WRITE[26], TCC_RW_REQ[27], TCC_TOO_MANY_EA_WRREQS_STALL[27], TCC_WRITE[27], TCC_RW_REQ[28], TCC_TOO_MANY_EA_WRREQS_STALL[28], TCC_WRITE[28], TCC_RW_REQ[29], TCC_TOO_MANY_EA_WRREQS_STALL[29], TCC_WRITE[29], TCC_RW_REQ[30], TCC_TOO_MANY_EA_WRREQS_STALL[30], TCC_WRITE[30], TCC_RW_REQ[31], TCC_TOO_MANY_EA_WRREQS_STALL[31], TCC_WRITE[31]
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170508_4193269/input0_results_240321_170508
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_18.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_2.txt
|-> [rocprof] RPL: on '240321_170509' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_2.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170509_4193471'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170509_4193471/input0_results_240321_170509'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170509_4193471/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 26 metrics
|-> [rocprof] SQ_INSTS_VALU_MUL_F32, SQ_INSTS_VALU_FMA_F32, SQ_INSTS_VALU_TRANS_F32, SQ_INSTS_VALU_ADD_F64, SQ_INSTS_VALU_MUL_F64, SQ_INSTS_VALU_FMA_F64, SQ_INSTS_VALU_TRANS_F64, SQ_INSTS_VALU_INT32, TCP_VOLATILE_sum, TCP_TOTAL_ACCESSES_sum, TCP_TOTAL_READ_sum, TCP_TOTAL_WRITE_sum, TA_BUFFER_ATOMIC_WAVEFRONTS_sum, TA_BUFFER_TOTAL_CYCLES_sum, TD_ATOMIC_WAVEFRONT_sum, TD_STORE_WAVEFRONT_sum, SPI_RA_REQ_NO_ALLOC, SPI_RA_REQ_NO_ALLOC_CSN, CPC_CPC_STAT_STALL, CPC_UTCL1_STALL_ON_TRANSLATION, CPF_CPF_STAT_IDLE, CPF_CPF_TCIU_IDLE, TCC_REQ_sum, TCC_STREAMING_REQ_sum, TCC_HIT_sum, TCC_MISS_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170509_4193471/input0_results_240321_170509
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_2.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_3.txt
|-> [rocprof] RPL: on '240321_170510' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_3.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170510_4193673'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170510_4193673/input0_results_240321_170510'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170510_4193673/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 24 metrics
|-> [rocprof] SQ_INSTS_VALU_INT64, SQ_INSTS_SMEM, SQ_INSTS_FLAT, SQ_INSTS_LDS, SQ_INSTS_GDS, SQ_INSTS_EXP_GDS, SQ_INSTS_BRANCH, SQ_INSTS_SENDMSG, TCP_TOTAL_ATOMIC_WITH_RET_sum, TCP_TOTAL_ATOMIC_WITHOUT_RET_sum, TCP_TOTAL_WRITEBACK_INVALIDATES_sum, TCP_TOTAL_CACHE_ACCESSES_sum, TA_BUFFER_COALESCED_READ_CYCLES_sum, TA_BUFFER_COALESCED_WRITE_CYCLES_sum, TD_COALESCABLE_WAVEFRONT_sum, SPI_RA_RES_STALL_CSN, SPI_RA_TMP_STALL_CSN, CPC_CPC_UTCL2IU_BUSY, CPC_CPC_UTCL2IU_IDLE, CPF_CMP_UTCL1_STALL_ON_TRANSLATION, TCC_READ_sum, TCC_WRITE_sum, TCC_ATOMIC_sum, TCC_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170510_4193673/input0_results_240321_170510
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_3.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_4.txt
|-> [rocprof] RPL: on '240321_170510' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_4.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170510_4193875'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170510_4193875/input0_results_240321_170510'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170510_4193875/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 22 metrics
|-> [rocprof] SQ_WAVE_CYCLES, SQ_WAIT_ANY, SQ_WAIT_INST_ANY, SQ_ACTIVE_INST_ANY, SQ_BUSY_CU_CYCLES, SQ_ACTIVE_INST_VMEM, SQ_ACTIVE_INST_LDS, SQ_ACTIVE_INST_VALU, TCP_UTCL1_TRANSLATION_MISS_sum, TCP_UTCL1_TRANSLATION_HIT_sum, TCP_UTCL1_PERMISSION_MISS_sum, TCP_UTCL1_REQUEST_sum, TA_ADDR_STALLED_BY_TC_CYCLES_sum, TA_TOTAL_WAVEFRONTS_sum, SPI_RA_WAVE_SIMD_FULL_CSN, SPI_RA_VGPR_SIMD_FULL_CSN, CPC_CPC_UTCL2IU_STALL, CPC_ME1_BUSY_FOR_PACKET_DECODE, TCC_EA_WRREQ_sum, TCC_EA_WRREQ_64B_sum, TCC_EA_WR_UNCACHED_32B_sum, TCC_EA_WRREQ_DRAM_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170510_4193875/input0_results_240321_170510
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_4.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_5.txt
|-> [rocprof] RPL: on '240321_170510' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_5.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170510_4194077'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170510_4194077/input0_results_240321_170510'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170510_4194077/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 21 metrics
|-> [rocprof] SQ_ACTIVE_INST_SCA, SQ_ACTIVE_INST_EXP_GDS, SQ_ACTIVE_INST_MISC, SQ_ACTIVE_INST_FLAT, SQ_INST_CYCLES_VMEM_WR, SQ_INST_CYCLES_VMEM_RD, SQ_INST_CYCLES_SMEM, SQ_INST_CYCLES_SALU, TCP_TCP_LATENCY_sum, TCP_TCC_READ_REQ_LATENCY_sum, TCP_TCC_WRITE_REQ_LATENCY_sum, TCP_TCC_READ_REQ_sum, TA_ADDR_STALLED_BY_TD_CYCLES_sum, TA_DATA_STALLED_BY_TC_CYCLES_sum, SPI_RA_SGPR_SIMD_FULL_CSN, SPI_RA_LDS_CU_FULL_CSN, CPC_ME1_DC0_SPI_BUSY, TCC_EA_WRREQ_STALL_sum, TCC_EA_RDREQ_sum, TCC_EA_RDREQ_32B_sum, TCC_EA_RD_UNCACHED_32B_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170510_4194077/input0_results_240321_170510
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_5.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_6.txt
|-> [rocprof] RPL: on '240321_170511' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_6.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170511_4194261'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170511_4194261/input0_results_240321_170511'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170511_4194261/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_THREAD_CYCLES_VALU, SQ_IFETCH, SQ_LDS_BANK_CONFLICT, SQ_LDS_ADDR_CONFLICT, SQ_LDS_UNALIGNED_STALL, SQ_WAVES_EQ_64, SQ_WAVES_LT_64, SQ_WAVES_LT_48, TCP_TCC_WRITE_REQ_sum, TCP_TCC_ATOMIC_WITH_RET_REQ_sum, TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum, TCP_TCC_NC_READ_REQ_sum, TA_FLAT_WAVEFRONTS_sum, TA_FLAT_READ_WAVEFRONTS_sum, SPI_RA_BAR_CU_FULL_CSN, SPI_RA_TGLIM_CU_FULL_CSN, TCC_EA_RDREQ_DRAM_sum, TCC_TAG_STALL_sum, TCC_NORMAL_WRITEBACK_sum, TCC_ALL_TC_OP_WB_WRITEBACK_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170511_4194261/input0_results_240321_170511
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_6.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_7.txt
|-> [rocprof] RPL: on '240321_170511' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_7.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170511_672'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170511_672/input0_results_240321_170511'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170511_672/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 20 metrics
|-> [rocprof] SQ_WAVES_LT_32, SQ_WAVES_LT_16, SQ_ITEMS, SQ_LDS_MEM_VIOLATIONS, SQ_LDS_ATOMIC_RETURN, SQ_LDS_IDX_ACTIVE, SQ_WAVES_RESTORED, SQ_WAVES_SAVED, TCP_TCC_NC_WRITE_REQ_sum, TCP_TCC_NC_ATOMIC_REQ_sum, TCP_TCC_UC_READ_REQ_sum, TCP_TCC_UC_WRITE_REQ_sum, TA_FLAT_WRITE_WAVEFRONTS_sum, TA_FLAT_ATOMIC_WAVEFRONTS_sum, SPI_RA_WVLIM_STALL_CSN, SPI_SWC_CSC_WR, TCC_NORMAL_EVICT_sum, TCC_ALL_TC_OP_INV_EVICT_sum, TCC_TOO_MANY_EA_WRREQS_STALL_sum, TCC_EA_ATOMIC_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170511_672/input0_results_240321_170511
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_7.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_8.txt
|-> [rocprof] RPL: on '240321_170512' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_8.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170512_894'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170512_894/input0_results_240321_170512'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170512_894/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 17 metrics
|-> [rocprof] SQ_INSTS_SMEM_NORM, SQ_INSTS_MFMA, SQ_INSTS_VALU_MFMA_I8, SQ_INSTS_VALU_MFMA_F16, SQ_INSTS_VALU_MFMA_BF16, SQ_INSTS_VALU_MFMA_F32, SQ_INSTS_VALU_MFMA_F64, SQ_VALU_MFMA_BUSY_CYCLES, TCP_TCC_UC_ATOMIC_REQ_sum, TCP_TCC_CC_READ_REQ_sum, TCP_TCC_CC_WRITE_REQ_sum, TCP_TCC_CC_ATOMIC_REQ_sum, SPI_VWC_CSC_WR, SPI_RA_BULKY_CU_FULL_CSN, TCC_EA_RDREQ_LEVEL_sum, TCC_EA_WRREQ_LEVEL_sum, TCC_EA_ATOMIC_LEVEL_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170512_894/input0_results_240321_170512
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_8.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/pmc_perf_9.txt
|-> [rocprof] RPL: on '240321_170512' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/pmc_perf_9.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170512_1097'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170512_1097/input0_results_240321_170512'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170512_1097/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 12 metrics
|-> [rocprof] SQ_INSTS_FLAT_LDS_ONLY, SQ_INSTS_VALU_MFMA_MOPS_I8, SQ_INSTS_VALU_MFMA_MOPS_F16, SQ_INSTS_VALU_MFMA_MOPS_BF16, SQ_INSTS_VALU_MFMA_MOPS_F32, SQ_INSTS_VALU_MFMA_MOPS_F64, SQC_TC_INST_REQ, SQC_TC_DATA_READ_REQ, TCP_TCC_RW_READ_REQ_sum, TCP_TCC_RW_WRITE_REQ_sum, TCP_TCC_RW_ATOMIC_REQ_sum, TCP_PENDING_STALL_CYCLES_sum
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170512_1097/input0_results_240321_170512
|-> [rocprof] File 'tests/workloads/kernel/MI200/pmc_perf_9.csv' is generating
|-> [rocprof]
[profiling] Current input file: tests/workloads/kernel/MI200/perfmon/timestamps.txt
|-> [rocprof] RPL: on '240321_170513' from '/opt/rocm-6.0.2' in '/home1/josantos/omniperf'
|-> [rocprof] RPL: profiling '""./tests/vcopy -n 1048576 -b 256 -i 3""'
|-> [rocprof] RPL: input file 'tests/workloads/kernel/MI200/perfmon/timestamps.txt'
|-> [rocprof] RPL: output dir '/tmp/rpl_data_240321_170513_1301'
|-> [rocprof] RPL: result dir '/tmp/rpl_data_240321_170513_1301/input0_results_240321_170513'
|-> [rocprof] ROCProfiler: input from "/tmp/rpl_data_240321_170513_1301/input0.xml"
|-> [rocprof] gpu_index =
|-> [rocprof] kernel = vecCopy
|-> [rocprof] range =
|-> [rocprof] 0 metrics
|-> [rocprof] vcopy testing on GCD 0
|-> [rocprof] Finished allocating vectors on the CPU
|-> [rocprof] Finished allocating vectors on the GPU
|-> [rocprof] Finished copying vectors to the GPU
|-> [rocprof] sw thinks it moved 1.000000 KB per wave
|-> [rocprof] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [rocprof] Launching the kernel on the GPU
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished executing kernel
|-> [rocprof] Finished copying the output vector from the GPU to the CPU
|-> [rocprof] Releasing GPU memory
|-> [rocprof] Releasing CPU memory
|-> [rocprof]
|-> [rocprof] ROCPRofiler: 3 contexts collected, output directory /tmp/rpl_data_240321_170513_1301/input0_results_240321_170513
|-> [rocprof] File 'tests/workloads/kernel/MI200/timestamps.csv' is generating
|-> [rocprof]
[roofline] Checking for roofline.csv in tests/workloads/kernel/MI200
[roofline] No roofline data found. Generating...
@@ -0,0 +1,5 @@
pmc: GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_CYCLES SQ_BUSY_CYCLES SQ_WAVES SQ_INSTS_VALU_CVT SQ_INSTS_VMEM_WR SQ_INSTS_VMEM_RD SQ_INSTS_VMEM SQ_INSTS_SALU GRBM_COUNT GRBM_GUI_ACTIVE TCP_GATE_EN1_sum TCP_GATE_EN2_sum TCP_TD_TCP_STALL_CYCLES_sum TCP_TCR_TCP_STALL_CYCLES_sum TA_TA_BUSY_sum TA_BUFFER_WAVEFRONTS_sum TD_TD_BUSY_sum TD_TC_STALL_sum SPI_CSN_WINDOW_VALID SPI_CSN_BUSY CPC_CPC_STAT_BUSY CPC_CPC_STAT_IDLE CPF_CPF_STAT_BUSY CPF_CPF_STAT_STALL TCC_CYCLE_sum TCC_BUSY_sum TCC_PROBE_sum TCC_PROBE_ALL_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VSKIPPED SQ_INSTS SQ_INSTS_VALU SQ_INSTS_VALU_ADD_F16 SQ_INSTS_VALU_MUL_F16 SQ_INSTS_VALU_FMA_F16 SQ_INSTS_VALU_TRANS_F16 SQ_INSTS_VALU_ADD_F32 GRBM_SPI_BUSY TCP_READ_TAGCONFLICT_STALL_CYCLES_sum TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum TCP_TA_TCP_STATE_READ_sum TA_BUFFER_READ_WAVEFRONTS_sum TA_BUFFER_WRITE_WAVEFRONTS_sum TD_SPI_STALL_sum TD_LOAD_WAVEFRONT_sum SPI_CSN_NUM_THREADGROUPS SPI_CSN_WAVE CPC_CPC_TCIU_BUSY CPC_CPC_TCIU_IDLE CPF_CPF_TCIU_BUSY CPF_CPF_TCIU_STALL TCC_NC_REQ_sum TCC_UC_REQ_sum TCC_CC_REQ_sum TCC_RW_REQ_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQC_TC_DATA_WRITE_REQ SQC_TC_DATA_ATOMIC_REQ SQC_TC_STALL SQC_TC_REQ SQC_DCACHE_REQ_READ_16 SQC_ICACHE_REQ SQC_ICACHE_HITS SQC_ICACHE_MISSES
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQC_ICACHE_MISSES_DUPLICATE SQC_DCACHE_INPUT_VALID_READYB SQC_DCACHE_ATOMIC SQC_DCACHE_REQ_READ_8 SQC_DCACHE_REQ SQC_DCACHE_HITS SQC_DCACHE_MISSES SQC_DCACHE_MISSES_DUPLICATE
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQC_DCACHE_REQ_READ_1 SQC_DCACHE_REQ_READ_2 SQC_DCACHE_REQ_READ_4
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_ATOMIC[0] TCC_CYCLE[0] TCC_EA_ATOMIC[0] TCC_EA_ATOMIC_LEVEL[0] TCC_ATOMIC[1] TCC_CYCLE[1] TCC_EA_ATOMIC[1] TCC_EA_ATOMIC_LEVEL[1] TCC_ATOMIC[2] TCC_CYCLE[2] TCC_EA_ATOMIC[2] TCC_EA_ATOMIC_LEVEL[2] TCC_ATOMIC[3] TCC_CYCLE[3] TCC_EA_ATOMIC[3] TCC_EA_ATOMIC_LEVEL[3] TCC_ATOMIC[4] TCC_CYCLE[4] TCC_EA_ATOMIC[4] TCC_EA_ATOMIC_LEVEL[4] TCC_ATOMIC[5] TCC_CYCLE[5] TCC_EA_ATOMIC[5] TCC_EA_ATOMIC_LEVEL[5] TCC_ATOMIC[6] TCC_CYCLE[6] TCC_EA_ATOMIC[6] TCC_EA_ATOMIC_LEVEL[6] TCC_ATOMIC[7] TCC_CYCLE[7] TCC_EA_ATOMIC[7] TCC_EA_ATOMIC_LEVEL[7] TCC_ATOMIC[8] TCC_CYCLE[8] TCC_EA_ATOMIC[8] TCC_EA_ATOMIC_LEVEL[8] TCC_ATOMIC[9] TCC_CYCLE[9] TCC_EA_ATOMIC[9] TCC_EA_ATOMIC_LEVEL[9] TCC_ATOMIC[10] TCC_CYCLE[10] TCC_EA_ATOMIC[10] TCC_EA_ATOMIC_LEVEL[10] TCC_ATOMIC[11] TCC_CYCLE[11] TCC_EA_ATOMIC[11] TCC_EA_ATOMIC_LEVEL[11] TCC_ATOMIC[12] TCC_CYCLE[12] TCC_EA_ATOMIC[12] TCC_EA_ATOMIC_LEVEL[12] TCC_ATOMIC[13] TCC_CYCLE[13] TCC_EA_ATOMIC[13] TCC_EA_ATOMIC_LEVEL[13] TCC_ATOMIC[14] TCC_CYCLE[14] TCC_EA_ATOMIC[14] TCC_EA_ATOMIC_LEVEL[14] TCC_ATOMIC[15] TCC_CYCLE[15] TCC_EA_ATOMIC[15] TCC_EA_ATOMIC_LEVEL[15] TCC_ATOMIC[16] TCC_CYCLE[16] TCC_EA_ATOMIC[16] TCC_EA_ATOMIC_LEVEL[16] TCC_ATOMIC[17] TCC_CYCLE[17] TCC_EA_ATOMIC[17] TCC_EA_ATOMIC_LEVEL[17] TCC_ATOMIC[18] TCC_CYCLE[18] TCC_EA_ATOMIC[18] TCC_EA_ATOMIC_LEVEL[18] TCC_ATOMIC[19] TCC_CYCLE[19] TCC_EA_ATOMIC[19] TCC_EA_ATOMIC_LEVEL[19] TCC_ATOMIC[20] TCC_CYCLE[20] TCC_EA_ATOMIC[20] TCC_EA_ATOMIC_LEVEL[20] TCC_ATOMIC[21] TCC_CYCLE[21] TCC_EA_ATOMIC[21] TCC_EA_ATOMIC_LEVEL[21] TCC_ATOMIC[22] TCC_CYCLE[22] TCC_EA_ATOMIC[22] TCC_EA_ATOMIC_LEVEL[22] TCC_ATOMIC[23] TCC_CYCLE[23] TCC_EA_ATOMIC[23] TCC_EA_ATOMIC_LEVEL[23] TCC_ATOMIC[24] TCC_CYCLE[24] TCC_EA_ATOMIC[24] TCC_EA_ATOMIC_LEVEL[24] TCC_ATOMIC[25] TCC_CYCLE[25] TCC_EA_ATOMIC[25] TCC_EA_ATOMIC_LEVEL[25] TCC_ATOMIC[26] TCC_CYCLE[26] TCC_EA_ATOMIC[26] TCC_EA_ATOMIC_LEVEL[26] TCC_ATOMIC[27] TCC_CYCLE[27] TCC_EA_ATOMIC[27] TCC_EA_ATOMIC_LEVEL[27] TCC_ATOMIC[28] TCC_CYCLE[28] TCC_EA_ATOMIC[28] TCC_EA_ATOMIC_LEVEL[28] TCC_ATOMIC[29] TCC_CYCLE[29] TCC_EA_ATOMIC[29] TCC_EA_ATOMIC_LEVEL[29] TCC_ATOMIC[30] TCC_CYCLE[30] TCC_EA_ATOMIC[30] TCC_EA_ATOMIC_LEVEL[30] TCC_ATOMIC[31] TCC_CYCLE[31] TCC_EA_ATOMIC[31] TCC_EA_ATOMIC_LEVEL[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_EA_RDREQ[0] TCC_EA_RDREQ_32B[0] TCC_EA_RDREQ_DRAM_CREDIT_STALL[0] TCC_EA_RDREQ_GMI_CREDIT_STALL[0] TCC_EA_RDREQ[1] TCC_EA_RDREQ_32B[1] TCC_EA_RDREQ_DRAM_CREDIT_STALL[1] TCC_EA_RDREQ_GMI_CREDIT_STALL[1] TCC_EA_RDREQ[2] TCC_EA_RDREQ_32B[2] TCC_EA_RDREQ_DRAM_CREDIT_STALL[2] TCC_EA_RDREQ_GMI_CREDIT_STALL[2] TCC_EA_RDREQ[3] TCC_EA_RDREQ_32B[3] TCC_EA_RDREQ_DRAM_CREDIT_STALL[3] TCC_EA_RDREQ_GMI_CREDIT_STALL[3] TCC_EA_RDREQ[4] TCC_EA_RDREQ_32B[4] TCC_EA_RDREQ_DRAM_CREDIT_STALL[4] TCC_EA_RDREQ_GMI_CREDIT_STALL[4] TCC_EA_RDREQ[5] TCC_EA_RDREQ_32B[5] TCC_EA_RDREQ_DRAM_CREDIT_STALL[5] TCC_EA_RDREQ_GMI_CREDIT_STALL[5] TCC_EA_RDREQ[6] TCC_EA_RDREQ_32B[6] TCC_EA_RDREQ_DRAM_CREDIT_STALL[6] TCC_EA_RDREQ_GMI_CREDIT_STALL[6] TCC_EA_RDREQ[7] TCC_EA_RDREQ_32B[7] TCC_EA_RDREQ_DRAM_CREDIT_STALL[7] TCC_EA_RDREQ_GMI_CREDIT_STALL[7] TCC_EA_RDREQ[8] TCC_EA_RDREQ_32B[8] TCC_EA_RDREQ_DRAM_CREDIT_STALL[8] TCC_EA_RDREQ_GMI_CREDIT_STALL[8] TCC_EA_RDREQ[9] TCC_EA_RDREQ_32B[9] TCC_EA_RDREQ_DRAM_CREDIT_STALL[9] TCC_EA_RDREQ_GMI_CREDIT_STALL[9] TCC_EA_RDREQ[10] TCC_EA_RDREQ_32B[10] TCC_EA_RDREQ_DRAM_CREDIT_STALL[10] TCC_EA_RDREQ_GMI_CREDIT_STALL[10] TCC_EA_RDREQ[11] TCC_EA_RDREQ_32B[11] TCC_EA_RDREQ_DRAM_CREDIT_STALL[11] TCC_EA_RDREQ_GMI_CREDIT_STALL[11] TCC_EA_RDREQ[12] TCC_EA_RDREQ_32B[12] TCC_EA_RDREQ_DRAM_CREDIT_STALL[12] TCC_EA_RDREQ_GMI_CREDIT_STALL[12] TCC_EA_RDREQ[13] TCC_EA_RDREQ_32B[13] TCC_EA_RDREQ_DRAM_CREDIT_STALL[13] TCC_EA_RDREQ_GMI_CREDIT_STALL[13] TCC_EA_RDREQ[14] TCC_EA_RDREQ_32B[14] TCC_EA_RDREQ_DRAM_CREDIT_STALL[14] TCC_EA_RDREQ_GMI_CREDIT_STALL[14] TCC_EA_RDREQ[15] TCC_EA_RDREQ_32B[15] TCC_EA_RDREQ_DRAM_CREDIT_STALL[15] TCC_EA_RDREQ_GMI_CREDIT_STALL[15] TCC_EA_RDREQ[16] TCC_EA_RDREQ_32B[16] TCC_EA_RDREQ_DRAM_CREDIT_STALL[16] TCC_EA_RDREQ_GMI_CREDIT_STALL[16] TCC_EA_RDREQ[17] TCC_EA_RDREQ_32B[17] TCC_EA_RDREQ_DRAM_CREDIT_STALL[17] TCC_EA_RDREQ_GMI_CREDIT_STALL[17] TCC_EA_RDREQ[18] TCC_EA_RDREQ_32B[18] TCC_EA_RDREQ_DRAM_CREDIT_STALL[18] TCC_EA_RDREQ_GMI_CREDIT_STALL[18] TCC_EA_RDREQ[19] TCC_EA_RDREQ_32B[19] TCC_EA_RDREQ_DRAM_CREDIT_STALL[19] TCC_EA_RDREQ_GMI_CREDIT_STALL[19] TCC_EA_RDREQ[20] TCC_EA_RDREQ_32B[20] TCC_EA_RDREQ_DRAM_CREDIT_STALL[20] TCC_EA_RDREQ_GMI_CREDIT_STALL[20] TCC_EA_RDREQ[21] TCC_EA_RDREQ_32B[21] TCC_EA_RDREQ_DRAM_CREDIT_STALL[21] TCC_EA_RDREQ_GMI_CREDIT_STALL[21] TCC_EA_RDREQ[22] TCC_EA_RDREQ_32B[22] TCC_EA_RDREQ_DRAM_CREDIT_STALL[22] TCC_EA_RDREQ_GMI_CREDIT_STALL[22] TCC_EA_RDREQ[23] TCC_EA_RDREQ_32B[23] TCC_EA_RDREQ_DRAM_CREDIT_STALL[23] TCC_EA_RDREQ_GMI_CREDIT_STALL[23] TCC_EA_RDREQ[24] TCC_EA_RDREQ_32B[24] TCC_EA_RDREQ_DRAM_CREDIT_STALL[24] TCC_EA_RDREQ_GMI_CREDIT_STALL[24] TCC_EA_RDREQ[25] TCC_EA_RDREQ_32B[25] TCC_EA_RDREQ_DRAM_CREDIT_STALL[25] TCC_EA_RDREQ_GMI_CREDIT_STALL[25] TCC_EA_RDREQ[26] TCC_EA_RDREQ_32B[26] TCC_EA_RDREQ_DRAM_CREDIT_STALL[26] TCC_EA_RDREQ_GMI_CREDIT_STALL[26] TCC_EA_RDREQ[27] TCC_EA_RDREQ_32B[27] TCC_EA_RDREQ_DRAM_CREDIT_STALL[27] TCC_EA_RDREQ_GMI_CREDIT_STALL[27] TCC_EA_RDREQ[28] TCC_EA_RDREQ_32B[28] TCC_EA_RDREQ_DRAM_CREDIT_STALL[28] TCC_EA_RDREQ_GMI_CREDIT_STALL[28] TCC_EA_RDREQ[29] TCC_EA_RDREQ_32B[29] TCC_EA_RDREQ_DRAM_CREDIT_STALL[29] TCC_EA_RDREQ_GMI_CREDIT_STALL[29] TCC_EA_RDREQ[30] TCC_EA_RDREQ_32B[30] TCC_EA_RDREQ_DRAM_CREDIT_STALL[30] TCC_EA_RDREQ_GMI_CREDIT_STALL[30] TCC_EA_RDREQ[31] TCC_EA_RDREQ_32B[31] TCC_EA_RDREQ_DRAM_CREDIT_STALL[31] TCC_EA_RDREQ_GMI_CREDIT_STALL[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_EA_RDREQ_IO_CREDIT_STALL[0] TCC_EA_RDREQ_LEVEL[0] TCC_EA_WRREQ[0] TCC_EA_WRREQ_64B[0] TCC_EA_RDREQ_IO_CREDIT_STALL[1] TCC_EA_RDREQ_LEVEL[1] TCC_EA_WRREQ[1] TCC_EA_WRREQ_64B[1] TCC_EA_RDREQ_IO_CREDIT_STALL[2] TCC_EA_RDREQ_LEVEL[2] TCC_EA_WRREQ[2] TCC_EA_WRREQ_64B[2] TCC_EA_RDREQ_IO_CREDIT_STALL[3] TCC_EA_RDREQ_LEVEL[3] TCC_EA_WRREQ[3] TCC_EA_WRREQ_64B[3] TCC_EA_RDREQ_IO_CREDIT_STALL[4] TCC_EA_RDREQ_LEVEL[4] TCC_EA_WRREQ[4] TCC_EA_WRREQ_64B[4] TCC_EA_RDREQ_IO_CREDIT_STALL[5] TCC_EA_RDREQ_LEVEL[5] TCC_EA_WRREQ[5] TCC_EA_WRREQ_64B[5] TCC_EA_RDREQ_IO_CREDIT_STALL[6] TCC_EA_RDREQ_LEVEL[6] TCC_EA_WRREQ[6] TCC_EA_WRREQ_64B[6] TCC_EA_RDREQ_IO_CREDIT_STALL[7] TCC_EA_RDREQ_LEVEL[7] TCC_EA_WRREQ[7] TCC_EA_WRREQ_64B[7] TCC_EA_RDREQ_IO_CREDIT_STALL[8] TCC_EA_RDREQ_LEVEL[8] TCC_EA_WRREQ[8] TCC_EA_WRREQ_64B[8] TCC_EA_RDREQ_IO_CREDIT_STALL[9] TCC_EA_RDREQ_LEVEL[9] TCC_EA_WRREQ[9] TCC_EA_WRREQ_64B[9] TCC_EA_RDREQ_IO_CREDIT_STALL[10] TCC_EA_RDREQ_LEVEL[10] TCC_EA_WRREQ[10] TCC_EA_WRREQ_64B[10] TCC_EA_RDREQ_IO_CREDIT_STALL[11] TCC_EA_RDREQ_LEVEL[11] TCC_EA_WRREQ[11] TCC_EA_WRREQ_64B[11] TCC_EA_RDREQ_IO_CREDIT_STALL[12] TCC_EA_RDREQ_LEVEL[12] TCC_EA_WRREQ[12] TCC_EA_WRREQ_64B[12] TCC_EA_RDREQ_IO_CREDIT_STALL[13] TCC_EA_RDREQ_LEVEL[13] TCC_EA_WRREQ[13] TCC_EA_WRREQ_64B[13] TCC_EA_RDREQ_IO_CREDIT_STALL[14] TCC_EA_RDREQ_LEVEL[14] TCC_EA_WRREQ[14] TCC_EA_WRREQ_64B[14] TCC_EA_RDREQ_IO_CREDIT_STALL[15] TCC_EA_RDREQ_LEVEL[15] TCC_EA_WRREQ[15] TCC_EA_WRREQ_64B[15] TCC_EA_RDREQ_IO_CREDIT_STALL[16] TCC_EA_RDREQ_LEVEL[16] TCC_EA_WRREQ[16] TCC_EA_WRREQ_64B[16] TCC_EA_RDREQ_IO_CREDIT_STALL[17] TCC_EA_RDREQ_LEVEL[17] TCC_EA_WRREQ[17] TCC_EA_WRREQ_64B[17] TCC_EA_RDREQ_IO_CREDIT_STALL[18] TCC_EA_RDREQ_LEVEL[18] TCC_EA_WRREQ[18] TCC_EA_WRREQ_64B[18] TCC_EA_RDREQ_IO_CREDIT_STALL[19] TCC_EA_RDREQ_LEVEL[19] TCC_EA_WRREQ[19] TCC_EA_WRREQ_64B[19] TCC_EA_RDREQ_IO_CREDIT_STALL[20] TCC_EA_RDREQ_LEVEL[20] TCC_EA_WRREQ[20] TCC_EA_WRREQ_64B[20] TCC_EA_RDREQ_IO_CREDIT_STALL[21] TCC_EA_RDREQ_LEVEL[21] TCC_EA_WRREQ[21] TCC_EA_WRREQ_64B[21] TCC_EA_RDREQ_IO_CREDIT_STALL[22] TCC_EA_RDREQ_LEVEL[22] TCC_EA_WRREQ[22] TCC_EA_WRREQ_64B[22] TCC_EA_RDREQ_IO_CREDIT_STALL[23] TCC_EA_RDREQ_LEVEL[23] TCC_EA_WRREQ[23] TCC_EA_WRREQ_64B[23] TCC_EA_RDREQ_IO_CREDIT_STALL[24] TCC_EA_RDREQ_LEVEL[24] TCC_EA_WRREQ[24] TCC_EA_WRREQ_64B[24] TCC_EA_RDREQ_IO_CREDIT_STALL[25] TCC_EA_RDREQ_LEVEL[25] TCC_EA_WRREQ[25] TCC_EA_WRREQ_64B[25] TCC_EA_RDREQ_IO_CREDIT_STALL[26] TCC_EA_RDREQ_LEVEL[26] TCC_EA_WRREQ[26] TCC_EA_WRREQ_64B[26] TCC_EA_RDREQ_IO_CREDIT_STALL[27] TCC_EA_RDREQ_LEVEL[27] TCC_EA_WRREQ[27] TCC_EA_WRREQ_64B[27] TCC_EA_RDREQ_IO_CREDIT_STALL[28] TCC_EA_RDREQ_LEVEL[28] TCC_EA_WRREQ[28] TCC_EA_WRREQ_64B[28] TCC_EA_RDREQ_IO_CREDIT_STALL[29] TCC_EA_RDREQ_LEVEL[29] TCC_EA_WRREQ[29] TCC_EA_WRREQ_64B[29] TCC_EA_RDREQ_IO_CREDIT_STALL[30] TCC_EA_RDREQ_LEVEL[30] TCC_EA_WRREQ[30] TCC_EA_WRREQ_64B[30] TCC_EA_RDREQ_IO_CREDIT_STALL[31] TCC_EA_RDREQ_LEVEL[31] TCC_EA_WRREQ[31] TCC_EA_WRREQ_64B[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_EA_WRREQ_DRAM_CREDIT_STALL[0] TCC_EA_WRREQ_GMI_CREDIT_STALL[0] TCC_EA_WRREQ_IO_CREDIT_STALL[0] TCC_EA_WRREQ_LEVEL[0] TCC_EA_WRREQ_DRAM_CREDIT_STALL[1] TCC_EA_WRREQ_GMI_CREDIT_STALL[1] TCC_EA_WRREQ_IO_CREDIT_STALL[1] TCC_EA_WRREQ_LEVEL[1] TCC_EA_WRREQ_DRAM_CREDIT_STALL[2] TCC_EA_WRREQ_GMI_CREDIT_STALL[2] TCC_EA_WRREQ_IO_CREDIT_STALL[2] TCC_EA_WRREQ_LEVEL[2] TCC_EA_WRREQ_DRAM_CREDIT_STALL[3] TCC_EA_WRREQ_GMI_CREDIT_STALL[3] TCC_EA_WRREQ_IO_CREDIT_STALL[3] TCC_EA_WRREQ_LEVEL[3] TCC_EA_WRREQ_DRAM_CREDIT_STALL[4] TCC_EA_WRREQ_GMI_CREDIT_STALL[4] TCC_EA_WRREQ_IO_CREDIT_STALL[4] TCC_EA_WRREQ_LEVEL[4] TCC_EA_WRREQ_DRAM_CREDIT_STALL[5] TCC_EA_WRREQ_GMI_CREDIT_STALL[5] TCC_EA_WRREQ_IO_CREDIT_STALL[5] TCC_EA_WRREQ_LEVEL[5] TCC_EA_WRREQ_DRAM_CREDIT_STALL[6] TCC_EA_WRREQ_GMI_CREDIT_STALL[6] TCC_EA_WRREQ_IO_CREDIT_STALL[6] TCC_EA_WRREQ_LEVEL[6] TCC_EA_WRREQ_DRAM_CREDIT_STALL[7] TCC_EA_WRREQ_GMI_CREDIT_STALL[7] TCC_EA_WRREQ_IO_CREDIT_STALL[7] TCC_EA_WRREQ_LEVEL[7] TCC_EA_WRREQ_DRAM_CREDIT_STALL[8] TCC_EA_WRREQ_GMI_CREDIT_STALL[8] TCC_EA_WRREQ_IO_CREDIT_STALL[8] TCC_EA_WRREQ_LEVEL[8] TCC_EA_WRREQ_DRAM_CREDIT_STALL[9] TCC_EA_WRREQ_GMI_CREDIT_STALL[9] TCC_EA_WRREQ_IO_CREDIT_STALL[9] TCC_EA_WRREQ_LEVEL[9] TCC_EA_WRREQ_DRAM_CREDIT_STALL[10] TCC_EA_WRREQ_GMI_CREDIT_STALL[10] TCC_EA_WRREQ_IO_CREDIT_STALL[10] TCC_EA_WRREQ_LEVEL[10] TCC_EA_WRREQ_DRAM_CREDIT_STALL[11] TCC_EA_WRREQ_GMI_CREDIT_STALL[11] TCC_EA_WRREQ_IO_CREDIT_STALL[11] TCC_EA_WRREQ_LEVEL[11] TCC_EA_WRREQ_DRAM_CREDIT_STALL[12] TCC_EA_WRREQ_GMI_CREDIT_STALL[12] TCC_EA_WRREQ_IO_CREDIT_STALL[12] TCC_EA_WRREQ_LEVEL[12] TCC_EA_WRREQ_DRAM_CREDIT_STALL[13] TCC_EA_WRREQ_GMI_CREDIT_STALL[13] TCC_EA_WRREQ_IO_CREDIT_STALL[13] TCC_EA_WRREQ_LEVEL[13] TCC_EA_WRREQ_DRAM_CREDIT_STALL[14] TCC_EA_WRREQ_GMI_CREDIT_STALL[14] TCC_EA_WRREQ_IO_CREDIT_STALL[14] TCC_EA_WRREQ_LEVEL[14] TCC_EA_WRREQ_DRAM_CREDIT_STALL[15] TCC_EA_WRREQ_GMI_CREDIT_STALL[15] TCC_EA_WRREQ_IO_CREDIT_STALL[15] TCC_EA_WRREQ_LEVEL[15] TCC_EA_WRREQ_DRAM_CREDIT_STALL[16] TCC_EA_WRREQ_GMI_CREDIT_STALL[16] TCC_EA_WRREQ_IO_CREDIT_STALL[16] TCC_EA_WRREQ_LEVEL[16] TCC_EA_WRREQ_DRAM_CREDIT_STALL[17] TCC_EA_WRREQ_GMI_CREDIT_STALL[17] TCC_EA_WRREQ_IO_CREDIT_STALL[17] TCC_EA_WRREQ_LEVEL[17] TCC_EA_WRREQ_DRAM_CREDIT_STALL[18] TCC_EA_WRREQ_GMI_CREDIT_STALL[18] TCC_EA_WRREQ_IO_CREDIT_STALL[18] TCC_EA_WRREQ_LEVEL[18] TCC_EA_WRREQ_DRAM_CREDIT_STALL[19] TCC_EA_WRREQ_GMI_CREDIT_STALL[19] TCC_EA_WRREQ_IO_CREDIT_STALL[19] TCC_EA_WRREQ_LEVEL[19] TCC_EA_WRREQ_DRAM_CREDIT_STALL[20] TCC_EA_WRREQ_GMI_CREDIT_STALL[20] TCC_EA_WRREQ_IO_CREDIT_STALL[20] TCC_EA_WRREQ_LEVEL[20] TCC_EA_WRREQ_DRAM_CREDIT_STALL[21] TCC_EA_WRREQ_GMI_CREDIT_STALL[21] TCC_EA_WRREQ_IO_CREDIT_STALL[21] TCC_EA_WRREQ_LEVEL[21] TCC_EA_WRREQ_DRAM_CREDIT_STALL[22] TCC_EA_WRREQ_GMI_CREDIT_STALL[22] TCC_EA_WRREQ_IO_CREDIT_STALL[22] TCC_EA_WRREQ_LEVEL[22] TCC_EA_WRREQ_DRAM_CREDIT_STALL[23] TCC_EA_WRREQ_GMI_CREDIT_STALL[23] TCC_EA_WRREQ_IO_CREDIT_STALL[23] TCC_EA_WRREQ_LEVEL[23] TCC_EA_WRREQ_DRAM_CREDIT_STALL[24] TCC_EA_WRREQ_GMI_CREDIT_STALL[24] TCC_EA_WRREQ_IO_CREDIT_STALL[24] TCC_EA_WRREQ_LEVEL[24] TCC_EA_WRREQ_DRAM_CREDIT_STALL[25] TCC_EA_WRREQ_GMI_CREDIT_STALL[25] TCC_EA_WRREQ_IO_CREDIT_STALL[25] TCC_EA_WRREQ_LEVEL[25] TCC_EA_WRREQ_DRAM_CREDIT_STALL[26] TCC_EA_WRREQ_GMI_CREDIT_STALL[26] TCC_EA_WRREQ_IO_CREDIT_STALL[26] TCC_EA_WRREQ_LEVEL[26] TCC_EA_WRREQ_DRAM_CREDIT_STALL[27] TCC_EA_WRREQ_GMI_CREDIT_STALL[27] TCC_EA_WRREQ_IO_CREDIT_STALL[27] TCC_EA_WRREQ_LEVEL[27] TCC_EA_WRREQ_DRAM_CREDIT_STALL[28] TCC_EA_WRREQ_GMI_CREDIT_STALL[28] TCC_EA_WRREQ_IO_CREDIT_STALL[28] TCC_EA_WRREQ_LEVEL[28] TCC_EA_WRREQ_DRAM_CREDIT_STALL[29] TCC_EA_WRREQ_GMI_CREDIT_STALL[29] TCC_EA_WRREQ_IO_CREDIT_STALL[29] TCC_EA_WRREQ_LEVEL[29] TCC_EA_WRREQ_DRAM_CREDIT_STALL[30] TCC_EA_WRREQ_GMI_CREDIT_STALL[30] TCC_EA_WRREQ_IO_CREDIT_STALL[30] TCC_EA_WRREQ_LEVEL[30] TCC_EA_WRREQ_DRAM_CREDIT_STALL[31] TCC_EA_WRREQ_GMI_CREDIT_STALL[31] TCC_EA_WRREQ_IO_CREDIT_STALL[31] TCC_EA_WRREQ_LEVEL[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_HIT[0] TCC_MISS[0] TCC_READ[0] TCC_REQ[0] TCC_HIT[1] TCC_MISS[1] TCC_READ[1] TCC_REQ[1] TCC_HIT[2] TCC_MISS[2] TCC_READ[2] TCC_REQ[2] TCC_HIT[3] TCC_MISS[3] TCC_READ[3] TCC_REQ[3] TCC_HIT[4] TCC_MISS[4] TCC_READ[4] TCC_REQ[4] TCC_HIT[5] TCC_MISS[5] TCC_READ[5] TCC_REQ[5] TCC_HIT[6] TCC_MISS[6] TCC_READ[6] TCC_REQ[6] TCC_HIT[7] TCC_MISS[7] TCC_READ[7] TCC_REQ[7] TCC_HIT[8] TCC_MISS[8] TCC_READ[8] TCC_REQ[8] TCC_HIT[9] TCC_MISS[9] TCC_READ[9] TCC_REQ[9] TCC_HIT[10] TCC_MISS[10] TCC_READ[10] TCC_REQ[10] TCC_HIT[11] TCC_MISS[11] TCC_READ[11] TCC_REQ[11] TCC_HIT[12] TCC_MISS[12] TCC_READ[12] TCC_REQ[12] TCC_HIT[13] TCC_MISS[13] TCC_READ[13] TCC_REQ[13] TCC_HIT[14] TCC_MISS[14] TCC_READ[14] TCC_REQ[14] TCC_HIT[15] TCC_MISS[15] TCC_READ[15] TCC_REQ[15] TCC_HIT[16] TCC_MISS[16] TCC_READ[16] TCC_REQ[16] TCC_HIT[17] TCC_MISS[17] TCC_READ[17] TCC_REQ[17] TCC_HIT[18] TCC_MISS[18] TCC_READ[18] TCC_REQ[18] TCC_HIT[19] TCC_MISS[19] TCC_READ[19] TCC_REQ[19] TCC_HIT[20] TCC_MISS[20] TCC_READ[20] TCC_REQ[20] TCC_HIT[21] TCC_MISS[21] TCC_READ[21] TCC_REQ[21] TCC_HIT[22] TCC_MISS[22] TCC_READ[22] TCC_REQ[22] TCC_HIT[23] TCC_MISS[23] TCC_READ[23] TCC_REQ[23] TCC_HIT[24] TCC_MISS[24] TCC_READ[24] TCC_REQ[24] TCC_HIT[25] TCC_MISS[25] TCC_READ[25] TCC_REQ[25] TCC_HIT[26] TCC_MISS[26] TCC_READ[26] TCC_REQ[26] TCC_HIT[27] TCC_MISS[27] TCC_READ[27] TCC_REQ[27] TCC_HIT[28] TCC_MISS[28] TCC_READ[28] TCC_REQ[28] TCC_HIT[29] TCC_MISS[29] TCC_READ[29] TCC_REQ[29] TCC_HIT[30] TCC_MISS[30] TCC_READ[30] TCC_REQ[30] TCC_HIT[31] TCC_MISS[31] TCC_READ[31] TCC_REQ[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: TCC_RW_REQ[0] TCC_TOO_MANY_EA_WRREQS_STALL[0] TCC_WRITE[0] TCC_RW_REQ[1] TCC_TOO_MANY_EA_WRREQS_STALL[1] TCC_WRITE[1] TCC_RW_REQ[2] TCC_TOO_MANY_EA_WRREQS_STALL[2] TCC_WRITE[2] TCC_RW_REQ[3] TCC_TOO_MANY_EA_WRREQS_STALL[3] TCC_WRITE[3] TCC_RW_REQ[4] TCC_TOO_MANY_EA_WRREQS_STALL[4] TCC_WRITE[4] TCC_RW_REQ[5] TCC_TOO_MANY_EA_WRREQS_STALL[5] TCC_WRITE[5] TCC_RW_REQ[6] TCC_TOO_MANY_EA_WRREQS_STALL[6] TCC_WRITE[6] TCC_RW_REQ[7] TCC_TOO_MANY_EA_WRREQS_STALL[7] TCC_WRITE[7] TCC_RW_REQ[8] TCC_TOO_MANY_EA_WRREQS_STALL[8] TCC_WRITE[8] TCC_RW_REQ[9] TCC_TOO_MANY_EA_WRREQS_STALL[9] TCC_WRITE[9] TCC_RW_REQ[10] TCC_TOO_MANY_EA_WRREQS_STALL[10] TCC_WRITE[10] TCC_RW_REQ[11] TCC_TOO_MANY_EA_WRREQS_STALL[11] TCC_WRITE[11] TCC_RW_REQ[12] TCC_TOO_MANY_EA_WRREQS_STALL[12] TCC_WRITE[12] TCC_RW_REQ[13] TCC_TOO_MANY_EA_WRREQS_STALL[13] TCC_WRITE[13] TCC_RW_REQ[14] TCC_TOO_MANY_EA_WRREQS_STALL[14] TCC_WRITE[14] TCC_RW_REQ[15] TCC_TOO_MANY_EA_WRREQS_STALL[15] TCC_WRITE[15] TCC_RW_REQ[16] TCC_TOO_MANY_EA_WRREQS_STALL[16] TCC_WRITE[16] TCC_RW_REQ[17] TCC_TOO_MANY_EA_WRREQS_STALL[17] TCC_WRITE[17] TCC_RW_REQ[18] TCC_TOO_MANY_EA_WRREQS_STALL[18] TCC_WRITE[18] TCC_RW_REQ[19] TCC_TOO_MANY_EA_WRREQS_STALL[19] TCC_WRITE[19] TCC_RW_REQ[20] TCC_TOO_MANY_EA_WRREQS_STALL[20] TCC_WRITE[20] TCC_RW_REQ[21] TCC_TOO_MANY_EA_WRREQS_STALL[21] TCC_WRITE[21] TCC_RW_REQ[22] TCC_TOO_MANY_EA_WRREQS_STALL[22] TCC_WRITE[22] TCC_RW_REQ[23] TCC_TOO_MANY_EA_WRREQS_STALL[23] TCC_WRITE[23] TCC_RW_REQ[24] TCC_TOO_MANY_EA_WRREQS_STALL[24] TCC_WRITE[24] TCC_RW_REQ[25] TCC_TOO_MANY_EA_WRREQS_STALL[25] TCC_WRITE[25] TCC_RW_REQ[26] TCC_TOO_MANY_EA_WRREQS_STALL[26] TCC_WRITE[26] TCC_RW_REQ[27] TCC_TOO_MANY_EA_WRREQS_STALL[27] TCC_WRITE[27] TCC_RW_REQ[28] TCC_TOO_MANY_EA_WRREQS_STALL[28] TCC_WRITE[28] TCC_RW_REQ[29] TCC_TOO_MANY_EA_WRREQS_STALL[29] TCC_WRITE[29] TCC_RW_REQ[30] TCC_TOO_MANY_EA_WRREQS_STALL[30] TCC_WRITE[30] TCC_RW_REQ[31] TCC_TOO_MANY_EA_WRREQS_STALL[31] TCC_WRITE[31]
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VALU_MUL_F32 SQ_INSTS_VALU_FMA_F32 SQ_INSTS_VALU_TRANS_F32 SQ_INSTS_VALU_ADD_F64 SQ_INSTS_VALU_MUL_F64 SQ_INSTS_VALU_FMA_F64 SQ_INSTS_VALU_TRANS_F64 SQ_INSTS_VALU_INT32 TCP_VOLATILE_sum TCP_TOTAL_ACCESSES_sum TCP_TOTAL_READ_sum TCP_TOTAL_WRITE_sum TA_BUFFER_ATOMIC_WAVEFRONTS_sum TA_BUFFER_TOTAL_CYCLES_sum TD_ATOMIC_WAVEFRONT_sum TD_STORE_WAVEFRONT_sum SPI_RA_REQ_NO_ALLOC SPI_RA_REQ_NO_ALLOC_CSN CPC_CPC_STAT_STALL CPC_UTCL1_STALL_ON_TRANSLATION CPF_CPF_STAT_IDLE CPF_CPF_TCIU_IDLE TCC_REQ_sum TCC_STREAMING_REQ_sum TCC_HIT_sum TCC_MISS_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VALU_INT64 SQ_INSTS_SMEM SQ_INSTS_FLAT SQ_INSTS_LDS SQ_INSTS_GDS SQ_INSTS_EXP_GDS SQ_INSTS_BRANCH SQ_INSTS_SENDMSG TCP_TOTAL_ATOMIC_WITH_RET_sum TCP_TOTAL_ATOMIC_WITHOUT_RET_sum TCP_TOTAL_WRITEBACK_INVALIDATES_sum TCP_TOTAL_CACHE_ACCESSES_sum TA_BUFFER_COALESCED_READ_CYCLES_sum TA_BUFFER_COALESCED_WRITE_CYCLES_sum TD_COALESCABLE_WAVEFRONT_sum SPI_RA_RES_STALL_CSN SPI_RA_TMP_STALL_CSN CPC_CPC_UTCL2IU_BUSY CPC_CPC_UTCL2IU_IDLE CPF_CMP_UTCL1_STALL_ON_TRANSLATION TCC_READ_sum TCC_WRITE_sum TCC_ATOMIC_sum TCC_WRITEBACK_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_WAVE_CYCLES SQ_WAIT_ANY SQ_WAIT_INST_ANY SQ_ACTIVE_INST_ANY SQ_BUSY_CU_CYCLES SQ_ACTIVE_INST_VMEM SQ_ACTIVE_INST_LDS SQ_ACTIVE_INST_VALU TCP_UTCL1_TRANSLATION_MISS_sum TCP_UTCL1_TRANSLATION_HIT_sum TCP_UTCL1_PERMISSION_MISS_sum TCP_UTCL1_REQUEST_sum TA_ADDR_STALLED_BY_TC_CYCLES_sum TA_TOTAL_WAVEFRONTS_sum SPI_RA_WAVE_SIMD_FULL_CSN SPI_RA_VGPR_SIMD_FULL_CSN CPC_CPC_UTCL2IU_STALL CPC_ME1_BUSY_FOR_PACKET_DECODE TCC_EA_WRREQ_sum TCC_EA_WRREQ_64B_sum TCC_EA_WR_UNCACHED_32B_sum TCC_EA_WRREQ_DRAM_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_ACTIVE_INST_SCA SQ_ACTIVE_INST_EXP_GDS SQ_ACTIVE_INST_MISC SQ_ACTIVE_INST_FLAT SQ_INST_CYCLES_VMEM_WR SQ_INST_CYCLES_VMEM_RD SQ_INST_CYCLES_SMEM SQ_INST_CYCLES_SALU TCP_TCP_LATENCY_sum TCP_TCC_READ_REQ_LATENCY_sum TCP_TCC_WRITE_REQ_LATENCY_sum TCP_TCC_READ_REQ_sum TA_ADDR_STALLED_BY_TD_CYCLES_sum TA_DATA_STALLED_BY_TC_CYCLES_sum SPI_RA_SGPR_SIMD_FULL_CSN SPI_RA_LDS_CU_FULL_CSN CPC_ME1_DC0_SPI_BUSY TCC_EA_WRREQ_STALL_sum TCC_EA_RDREQ_sum TCC_EA_RDREQ_32B_sum TCC_EA_RD_UNCACHED_32B_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_THREAD_CYCLES_VALU SQ_IFETCH SQ_LDS_BANK_CONFLICT SQ_LDS_ADDR_CONFLICT SQ_LDS_UNALIGNED_STALL SQ_WAVES_EQ_64 SQ_WAVES_LT_64 SQ_WAVES_LT_48 TCP_TCC_WRITE_REQ_sum TCP_TCC_ATOMIC_WITH_RET_REQ_sum TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum TCP_TCC_NC_READ_REQ_sum TA_FLAT_WAVEFRONTS_sum TA_FLAT_READ_WAVEFRONTS_sum SPI_RA_BAR_CU_FULL_CSN SPI_RA_TGLIM_CU_FULL_CSN TCC_EA_RDREQ_DRAM_sum TCC_TAG_STALL_sum TCC_NORMAL_WRITEBACK_sum TCC_ALL_TC_OP_WB_WRITEBACK_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_WAVES_LT_32 SQ_WAVES_LT_16 SQ_ITEMS SQ_LDS_MEM_VIOLATIONS SQ_LDS_ATOMIC_RETURN SQ_LDS_IDX_ACTIVE SQ_WAVES_RESTORED SQ_WAVES_SAVED TCP_TCC_NC_WRITE_REQ_sum TCP_TCC_NC_ATOMIC_REQ_sum TCP_TCC_UC_READ_REQ_sum TCP_TCC_UC_WRITE_REQ_sum TA_FLAT_WRITE_WAVEFRONTS_sum TA_FLAT_ATOMIC_WAVEFRONTS_sum SPI_RA_WVLIM_STALL_CSN SPI_SWC_CSC_WR TCC_NORMAL_EVICT_sum TCC_ALL_TC_OP_INV_EVICT_sum TCC_TOO_MANY_EA_WRREQS_STALL_sum TCC_EA_ATOMIC_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_SMEM_NORM SQ_INSTS_MFMA SQ_INSTS_VALU_MFMA_I8 SQ_INSTS_VALU_MFMA_F16 SQ_INSTS_VALU_MFMA_BF16 SQ_INSTS_VALU_MFMA_F32 SQ_INSTS_VALU_MFMA_F64 SQ_VALU_MFMA_BUSY_CYCLES TCP_TCC_UC_ATOMIC_REQ_sum TCP_TCC_CC_READ_REQ_sum TCP_TCC_CC_WRITE_REQ_sum TCP_TCC_CC_ATOMIC_REQ_sum SPI_VWC_CSC_WR SPI_RA_BULKY_CU_FULL_CSN TCC_EA_RDREQ_LEVEL_sum TCC_EA_WRREQ_LEVEL_sum TCC_EA_ATOMIC_LEVEL_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_FLAT_LDS_ONLY SQ_INSTS_VALU_MFMA_MOPS_I8 SQ_INSTS_VALU_MFMA_MOPS_F16 SQ_INSTS_VALU_MFMA_MOPS_BF16 SQ_INSTS_VALU_MFMA_MOPS_F32 SQ_INSTS_VALU_MFMA_MOPS_F64 SQC_TC_INST_REQ SQC_TC_DATA_READ_REQ TCP_TCC_RW_READ_REQ_sum TCP_TCC_RW_WRITE_REQ_sum TCP_TCC_RW_ATOMIC_REQ_sum TCP_PENDING_STALL_CYCLES_sum
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,5 @@
pmc:
gpu:
range:
kernel: vecCopy
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2
1 Dispatch_ID Kernel_Name GPU_ID
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2
File diff suppressed because one or more lines are too long
@@ -0,0 +1,5 @@
device,HBMBw,HBMBwLow,hbmBwHigh,L2Bw,L2BwLow,L2BwHigh,L1Bw,L1BwLow,L1BwHigh,LDSBw,LDSBwLow,LDSBwHigh,FP32Flops,FP32FlopsLow,FP32FlopsHigh,FP64Flops,FP64FlopsLow,FP64FlopsHigh,MFMABF16Flops,MFMABF16FlopsLow,MFMABF16FlopsHigh,MFMAF16Flops,MFMAF16FlopsLow,MFMAF16FlopsHigh,MFMAF32Flops,MFMAF32FlopsLow,MFMAF32FlopsHigh,MFMAF64Flops,MFMAF64FlopsLow,MFMAF64FlopsHigh,MFMAI8Ops,MFMAFI8OpsLow,MFMAI8OpsHigh
0,1388.8003,1388.2018,1389.3988,5018.0034,5014.9653,5021.0415,9227.3828,9226.832,9227.9336,17715.787,17712.695,17718.879,20942.057,20881.729,21002.385,20273.387,20272.754,20274.02,170490.59,170486.83,170494.36,164905.91,164902.36,164909.45,41435.109,41434.566,41435.652,41490.23,41489.523,41490.938,166390.95,165895.59,166886.31
1,1389.1022,1388.6135,1389.5908,5028.1416,5026.2451,5030.0381,9235.0059,9234.3955,9235.6162,18231.234,18229.113,18233.355,20975.01,20907.771,21042.248,20293.785,20293.287,20294.283,170865.64,170860.17,170871.11,165134.81,165131.06,165138.56,41491.012,41490.277,41491.746,41546.773,41546.109,41547.438,166891.17,166888.8,166893.55
2,1388.7727,1388.2306,1389.3148,5035.6616,5033.3149,5038.0083,9261.2324,9260.5088,9261.9561,18217.076,18216.133,18218.02,21049.338,21048.953,21049.723,20351.391,20350.875,20351.906,171111.95,171108.86,171115.05,165494.22,165491.36,165497.08,41585.855,41584.934,41586.777,41643.219,41642.5,41643.938,166984.53,166391.91,167577.16
3,1389.2549,1388.5906,1389.9192,5032.6069,5031.0239,5034.1899,9239.5078,9238.6982,9240.3174,18329.676,18328.783,18330.568,21036.42,21036.109,21036.73,20313.297,20312.844,20313.75,171033,171029.09,171036.91,165165.02,165160.69,165169.34,41426.988,41272.484,41581.492,41573.762,41572.508,41575.016,166972.52,166968.34,166976.69
1 device HBMBw HBMBwLow hbmBwHigh L2Bw L2BwLow L2BwHigh L1Bw L1BwLow L1BwHigh LDSBw LDSBwLow LDSBwHigh FP32Flops FP32FlopsLow FP32FlopsHigh FP64Flops FP64FlopsLow FP64FlopsHigh MFMABF16Flops MFMABF16FlopsLow MFMABF16FlopsHigh MFMAF16Flops MFMAF16FlopsLow MFMAF16FlopsHigh MFMAF32Flops MFMAF32FlopsLow MFMAF32FlopsHigh MFMAF64Flops MFMAF64FlopsLow MFMAF64FlopsHigh MFMAI8Ops MFMAFI8OpsLow MFMAI8OpsHigh
2 0 1388.8003 1388.2018 1389.3988 5018.0034 5014.9653 5021.0415 9227.3828 9226.832 9227.9336 17715.787 17712.695 17718.879 20942.057 20881.729 21002.385 20273.387 20272.754 20274.02 170490.59 170486.83 170494.36 164905.91 164902.36 164909.45 41435.109 41434.566 41435.652 41490.23 41489.523 41490.938 166390.95 165895.59 166886.31
3 1 1389.1022 1388.6135 1389.5908 5028.1416 5026.2451 5030.0381 9235.0059 9234.3955 9235.6162 18231.234 18229.113 18233.355 20975.01 20907.771 21042.248 20293.785 20293.287 20294.283 170865.64 170860.17 170871.11 165134.81 165131.06 165138.56 41491.012 41490.277 41491.746 41546.773 41546.109 41547.438 166891.17 166888.8 166893.55
4 2 1388.7727 1388.2306 1389.3148 5035.6616 5033.3149 5038.0083 9261.2324 9260.5088 9261.9561 18217.076 18216.133 18218.02 21049.338 21048.953 21049.723 20351.391 20350.875 20351.906 171111.95 171108.86 171115.05 165494.22 165491.36 165497.08 41585.855 41584.934 41586.777 41643.219 41642.5 41643.938 166984.53 166391.91 167577.16
5 3 1389.2549 1388.5906 1389.9192 5032.6069 5031.0239 5034.1899 9239.5078 9238.6982 9240.3174 18329.676 18328.783 18330.568 21036.42 21036.109 21036.73 20313.297 20312.844 20313.75 171033 171029.09 171036.91 165165.02 165160.69 165169.34 41426.988 41272.484 41581.492 41573.762 41572.508 41575.016 166972.52 166968.34 166976.69
@@ -0,0 +1,2 @@
workload_name,command,ip_blocks,timestamp,version,hostname,cpu_model,sbios,linux_distro,linux_kernel_version,amd_gpu_kernel_version,cpu_memory,gpu_memory,rocm_version,vbios,compute_partition,memory_partition,gpu_model,gpu_arch,gpu_l1,gpu_l2,cu_per_gpu,simd_per_cu,se_per_gpu,wave_size,workgroup_max_size,max_waves_per_cu,max_sclk,max_mclk,cur_sclk,cur_mclk,total_l2_chan,lds_banks_per_cu,sqc_per_gpu,pipes_per_gpu,hbm_bw,num_xcd,num_hbm_channels
path,./tests/vcopy -n 1048576 -b 256 -i 3,SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF|roofline,Thu 21 Mar 2024 04:16:46 PM (CDT),2,t007-002.hpcfund,AMD EPYC 7V13 64-Core Processor,American Megatrends Inc.0602,Rocky Linux 9.1 (Blue Onyx),5.14.0-162.18.1.el9_1.x86_64,,527650760,,6.0.2-115,113-D67301-059,NA,NA,MI200,gfx90a,16,8192,104,4,8,64,1024,32,1700,1600,1700,1600,32,32,56,4,1638.4,1,32
1 workload_name command ip_blocks timestamp version hostname cpu_model sbios linux_distro linux_kernel_version amd_gpu_kernel_version cpu_memory gpu_memory rocm_version vbios compute_partition memory_partition gpu_model gpu_arch gpu_l1 gpu_l2 cu_per_gpu simd_per_cu se_per_gpu wave_size workgroup_max_size max_waves_per_cu max_sclk max_mclk cur_sclk cur_mclk total_l2_chan lds_banks_per_cu sqc_per_gpu pipes_per_gpu hbm_bw num_xcd num_hbm_channels
2 path ./tests/vcopy -n 1048576 -b 256 -i 3 SQ|LDS|SQC|TA|TD|TCP|TCC|SPI|CPC|CPF|roofline Thu 21 Mar 2024 04:16:46 PM (CDT) 2 t007-002.hpcfund AMD EPYC 7V13 64-Core Processor American Megatrends Inc.0602 Rocky Linux 9.1 (Blue Onyx) 5.14.0-162.18.1.el9_1.x86_64 527650760 6.0.2-115 113-D67301-059 NA NA MI200 gfx90a 16 8192 104 4 8 64 1024 32 1700 1600 1700 1600 32 32 56 4 1638.4 1 32
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID,queue-id,queue-index,pid,tid,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,wave_size,sig,obj,DispatchNs,Start_Timestamp,End_Timestamp,CompleteNs
0,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,0,1463,1463,1048576,256,0,0,8,0,16,64,0x0,0x7f09a4408ec0,1414445656634195,1414445656659785,1414445656680105,1414445656694409
1,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,2,1463,1463,1048576,256,0,0,8,0,16,64,0x0,0x7f09a4408ec0,1414445656691093,1414445656699625,1414445656715145,1414445656771244
2,"vecCopy(double*, double*, double*, int, int) [clone .kd]",2,0,4,1463,1463,1048576,256,0,0,8,0,16,64,0x0,0x7f09a4408ec0,1414445656725698,1414445656774345,1414445656790665,1414445656791793
1 Dispatch_ID Kernel_Name GPU_ID queue-id queue-index pid tid Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR wave_size sig obj DispatchNs Start_Timestamp End_Timestamp CompleteNs
2 0 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 0 1463 1463 1048576 256 0 0 8 0 16 64 0x0 0x7f09a4408ec0 1414445656634195 1414445656659785 1414445656680105 1414445656694409
3 1 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 2 1463 1463 1048576 256 0 0 8 0 16 64 0x0 0x7f09a4408ec0 1414445656691093 1414445656699625 1414445656715145 1414445656771244
4 2 vecCopy(double*, double*, double*, int, int) [clone .kd] 2 0 4 1463 1463 1048576 256 0 0 8 0 16 64 0x0 0x7f09a4408ec0 1414445656725698 1414445656774345 1414445656790665 1414445656791793
@@ -0,0 +1,4 @@
Dispatch_ID,GPU_ID,Queue_ID,PID,TID,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,Wave_Size,Kernel_Name,Start_Timestamp,End_Timestamp,Correlation_ID,GRBM_COUNT,GRBM_GUI_ACTIVE,SQ_WAVES,SQ_IFETCH,SQ_IFETCH_LEVEL,SQ_ACCUM_PREV_HIRES
0,11995,1,150565,150565,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741709003,73994741716935,0,162848.0,162848.0,16384.0,65536.0,23275.0,1854956.0
1,11995,1,150565,150565,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741753751,73994741759760,0,183427.0,183427.0,16384.0,65536.0,13062.0,1048728.0
2,11995,1,150565,150565,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741733040,73994741738809,0,169147.0,169147.0,16384.0,65536.0,13160.0,1049428.0
1 Dispatch_ID GPU_ID Queue_ID PID TID Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR Wave_Size Kernel_Name Start_Timestamp End_Timestamp Correlation_ID GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES
2 0 11995 1 150565 150565 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741709003 73994741716935 0 162848.0 162848.0 16384.0 65536.0 23275.0 1854956.0
3 1 11995 1 150565 150565 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741753751 73994741759760 0 183427.0 183427.0 16384.0 65536.0 13062.0 1048728.0
4 2 11995 1 150565 150565 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741733040 73994741738809 0 169147.0 169147.0 16384.0 65536.0 13160.0 1049428.0
@@ -0,0 +1,4 @@
Dispatch_ID,GPU_ID,Queue_ID,PID,TID,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,Wave_Size,Kernel_Name,Start_Timestamp,End_Timestamp,Correlation_ID,SQ_INSTS_LDS,SQ_INST_LEVEL_LDS,SQ_ACCUM_PREV_HIRES
0,11995,1,150579,150579,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741709003,73994741716935,0,0.0,0.0,0.0
1,11995,1,150579,150579,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741753751,73994741759760,0,0.0,0.0,0.0
2,11995,1,150579,150579,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741733040,73994741738809,0,0.0,0.0,0.0
1 Dispatch_ID GPU_ID Queue_ID PID TID Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR Wave_Size Kernel_Name Start_Timestamp End_Timestamp Correlation_ID SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES
2 0 11995 1 150579 150579 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741709003 73994741716935 0 0.0 0.0 0.0
3 1 11995 1 150579 150579 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741753751 73994741759760 0 0.0 0.0 0.0
4 2 11995 1 150579 150579 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741733040 73994741738809 0 0.0 0.0 0.0
@@ -0,0 +1,4 @@
Dispatch_ID,GPU_ID,Queue_ID,PID,TID,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,Wave_Size,Kernel_Name,Start_Timestamp,End_Timestamp,Correlation_ID,SQ_INSTS_SMEM,SQ_INST_LEVEL_SMEM,SQ_ACCUM_PREV_HIRES
0,11995,1,150591,150591,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741709003,73994741716935,0,65536.0,278636.0,22348096.0
1,11995,1,150591,150591,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741753751,73994741759760,0,65536.0,275446.0,22108808.0
2,11995,1,150591,150591,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741733040,73994741738809,0,65536.0,199560.0,15923520.0
1 Dispatch_ID GPU_ID Queue_ID PID TID Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR Wave_Size Kernel_Name Start_Timestamp End_Timestamp Correlation_ID SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES
2 0 11995 1 150591 150591 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741709003 73994741716935 0 65536.0 278636.0 22348096.0
3 1 11995 1 150591 150591 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741753751 73994741759760 0 65536.0 275446.0 22108808.0
4 2 11995 1 150591 150591 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741733040 73994741738809 0 65536.0 199560.0 15923520.0
@@ -0,0 +1,4 @@
Dispatch_ID,GPU_ID,Queue_ID,PID,TID,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,Wave_Size,Kernel_Name,Start_Timestamp,End_Timestamp,Correlation_ID,SQ_INSTS_VMEM,SQ_INST_LEVEL_VMEM,SQ_ACCUM_PREV_HIRES
0,11995,1,150603,150603,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741709003,73994741716935,0,32768.0,535005.0,42794664.0
1,11995,1,150603,150603,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741753751,73994741759760,0,32768.0,421277.0,33695608.0
2,11995,1,150603,150603,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741733040,73994741738809,0,32768.0,421240.0,33685740.0
1 Dispatch_ID GPU_ID Queue_ID PID TID Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR Wave_Size Kernel_Name Start_Timestamp End_Timestamp Correlation_ID SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES
2 0 11995 1 150603 150603 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741709003 73994741716935 0 32768.0 535005.0 42794664.0
3 1 11995 1 150603 150603 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741753751 73994741759760 0 32768.0 421277.0 33695608.0
4 2 11995 1 150603 150603 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741733040 73994741738809 0 32768.0 421240.0 33685740.0
@@ -0,0 +1,4 @@
Dispatch_ID,GPU_ID,Queue_ID,PID,TID,Grid_Size,Workgroup_Size,LDS_Per_Workgroup,Scratch_Per_Workitem,Arch_VGPR,Accum_VGPR,SGPR,Wave_Size,Kernel_Name,Start_Timestamp,End_Timestamp,Correlation_ID,GRBM_COUNT,GRBM_GUI_ACTIVE,CPC_ME1_BUSY_FOR_PACKET_DECODE,SQ_CYCLES,SQ_WAVES,SQ_WAVE_CYCLES,SQ_BUSY_CYCLES,SQ_LEVEL_WAVES,SQ_ACCUM_PREV_HIRES
0,11995,1,150615,150615,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741709003,73994741716935,0,210164.0,210164.0,118323.0,840656.0,16384.0,12983237.0,245238.0,0.0,52328424.0
1,11995,1,150615,150615,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741753751,73994741759760,0,192152.0,192152.0,107629.0,768608.0,16384.0,11427426.0,206240.0,0.0,46110964.0
2,11995,1,150615,150615,1048576,256,0,0,4,4,16,64,"vecCopy(double*, double*, double*, int, int) (.kd)",73994741733040,73994741738809,0,173684.0,173684.0,93734.0,694736.0,16384.0,9922380.0,188269.0,0.0,40094032.0
1 Dispatch_ID GPU_ID Queue_ID PID TID Grid_Size Workgroup_Size LDS_Per_Workgroup Scratch_Per_Workitem Arch_VGPR Accum_VGPR SGPR Wave_Size Kernel_Name Start_Timestamp End_Timestamp Correlation_ID GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES
2 0 11995 1 150615 150615 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741709003 73994741716935 0 210164.0 210164.0 118323.0 840656.0 16384.0 12983237.0 245238.0 0.0 52328424.0
3 1 11995 1 150615 150615 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741753751 73994741759760 0 192152.0 192152.0 107629.0 768608.0 16384.0 11427426.0 206240.0 0.0 46110964.0
4 2 11995 1 150615 150615 1048576 256 0 0 4 4 16 64 vecCopy(double*, double*, double*, int, int) (.kd) 73994741733040 73994741738809 0 173684.0 173684.0 93734.0 694736.0 16384.0 9922380.0 188269.0 0.0 40094032.0
@@ -0,0 +1,215 @@
Omniperf version: 2.0.0
Profiler choice: rocprofv2
Path: /home/colramos/omniperf/tests/workloads/kernel/MI300A_A1
Target: MI300A_A1
Command: ./tests/vcopy -n 1048576 -b 256 -i 3
Kernel Selection: ['"vecCopy(double*,', 'double*,', 'double*,', 'int,', 'int)', '[clone', '.kd]"']
Dispatch Selection: None
Hardware Blocks: All
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Collecting Performance Counters
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/SQ_IFETCH_LEVEL.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - GRBM_COUNT
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/SQ_INST_LEVEL_LDS.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_LDS
|-> [/opt/rocm/bin/rocprofv2] - SQ_INST_LEVEL_LDS
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACCUM_PREV_HIRES
|-> [/opt/rocm/bin/rocprofv2] Enabling Counter Collection
|-> [/opt/rocm/bin/rocprofv2] vcopy testing on GCD 0
|-> [/opt/rocm/bin/rocprofv2] Finished allocating vectors on the CPU
|-> [/opt/rocm/bin/rocprofv2] Finished allocating vectors on the GPU
|-> [/opt/rocm/bin/rocprofv2] Finished copying vectors to the GPU
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/SQ_INST_LEVEL_SMEM.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_SMEM
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/SQ_INST_LEVEL_VMEM.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VMEM
|-> [/opt/rocm/bin/rocprofv2] - SQ_INST_LEVEL_VMEM
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACCUM_PREV_HIRES
|-> [/opt/rocm/bin/rocprofv2] Enabling Counter Collection
|-> [/opt/rocm/bin/rocprofv2] vcopy testing on GCD 0
|-> [/opt/rocm/bin/rocprofv2] Finished allocating vectors on the CPU
|-> [/opt/rocm/bin/rocprofv2] Finished allocating vectors on the GPU
|-> [/opt/rocm/bin/rocprofv2] Finished copying vectors to the GPU
|-> [/opt/rocm/bin/rocprofv2] sw thinks it moved 1.000000 KB per wave
|-> [/opt/rocm/bin/rocprofv2] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [/opt/rocm/bin/rocprofv2] Launching the kernel on the GPU
|-> [/opt/rocm/bin/rocprofv2] Finished executing kernel
|-> [/opt/rocm/bin/rocprofv2] Finished executing kernel
|-> [/opt/rocm/bin/rocprofv2] Finished executing kernel
|-> [/opt/rocm/bin/rocprofv2] Finished copying the output vector from the GPU to the CPU
|-> [/opt/rocm/bin/rocprofv2] Releasing GPU memory
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/SQ_LEVEL_WAVES.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - GRBM_COUNT
|-> [/opt/rocm/bin/rocprofv2] - GRBM_GUI_ACTIVE
|-> [/opt/rocm/bin/rocprofv2] - CPC_ME1_BUSY_FOR_PACKET_DECODE
|-> [/opt/rocm/bin/rocprofv2] - SQ_CYCLES
|-> [/opt/rocm/bin/rocprofv2] - SQ_WAVES
|-> [/opt/rocm/bin/rocprofv2] - SQ_WAVE_CYCLES
|-> [/opt/rocm/bin/rocprofv2] - SQ_BUSY_CYCLES
|-> [/opt/rocm/bin/rocprofv2] - SQ_LEVEL_WAVES
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_0.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_CYCLES
|-> [/opt/rocm/bin/rocprofv2] - SQ_BUSY_CYCLES
|-> [/opt/rocm/bin/rocprofv2] - SQ_BUSY_CU_CYCLES
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_1.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VMEM
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_10.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQC_TC_DATA_ATOMIC_REQ
|-> [/opt/rocm/bin/rocprofv2] - SQC_TC_STALL
|-> [/opt/rocm/bin/rocprofv2] - SQC_TC_REQ
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_REQ_READ_16
|-> [/opt/rocm/bin/rocprofv2] - SQC_ICACHE_REQ
|-> [/opt/rocm/bin/rocprofv2] - SQC_ICACHE_HITS
|-> [/opt/rocm/bin/rocprofv2] - SQC_ICACHE_MISSES
|-> [/opt/rocm/bin/rocprofv2] - SQC_ICACHE_MISSES_DUPLICATE
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_11.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_INPUT_VALID_READYB
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_ATOMIC
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_REQ_READ_8
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_REQ
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_HITS
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_MISSES
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_MISSES_DUPLICATE
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_REQ_READ_1
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_12.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQC_DCACHE_REQ_READ_2
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_13.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - TCC_ATOMIC[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_BUBBLE[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_CYCLE[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_ATOMIC[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_ATOMIC[1]
|-> [/opt/rocm/bin/rocprofv2] - TCC_BUBBLE[1]
|-> [/opt/rocm/bin/rocprofv2] - TCC_CYCLE[1]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_ATOMIC[1]
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_14.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_ATOMIC_LEVEL[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_RDREQ[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_RDREQ_32B[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_RDREQ_LEVEL[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_ATOMIC_LEVEL[1]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_RDREQ[1]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_RDREQ_32B[1]
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_15.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_WRREQ[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_WRREQ_64B[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_WRREQ_LEVEL[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_HIT[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_WRREQ[1]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_WRREQ_64B[1]
|-> [/opt/rocm/bin/rocprofv2] - TCC_EA0_WRREQ_LEVEL[1]
|-> [/opt/rocm/bin/rocprofv2] - TCC_HIT[1]
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_16.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - TCC_MISS[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_READ[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_REQ[0]
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_17.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - TCC_TAG_STALL[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_TOO_MANY_EA_WRREQS_STALL[0]
|-> [/opt/rocm/bin/rocprofv2] - TCC_WRITE[0]
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_2.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_TRANS_F16
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_ADD_F32
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_MUL_F32
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_FMA_F32
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_TRANS_F32
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_ADD_F64
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_MUL_F64
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_3.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_TRANS_F64
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_INT32
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_4.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_BRANCH
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_SENDMSG
|-> [/opt/rocm/bin/rocprofv2] - SQ_WAIT_ANY
|-> [/opt/rocm/bin/rocprofv2] - SQ_WAIT_INST_ANY
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACTIVE_INST_ANY
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACTIVE_INST_VMEM
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACTIVE_INST_LDS
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_5.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACTIVE_INST_SCA
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACTIVE_INST_EXP_GDS
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACTIVE_INST_MISC
|-> [/opt/rocm/bin/rocprofv2] - SQ_ACTIVE_INST_FLAT
|-> [/opt/rocm/bin/rocprofv2] - SQ_INST_CYCLES_VMEM_WR
|-> [/opt/rocm/bin/rocprofv2] - SQ_INST_CYCLES_VMEM_RD
|-> [/opt/rocm/bin/rocprofv2] - SQ_INST_CYCLES_SMEM
|-> [/opt/rocm/bin/rocprofv2] - SQ_INST_CYCLES_SALU
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_6.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_THREAD_CYCLES_VALU
|-> [/opt/rocm/bin/rocprofv2] - SQ_IFETCH
|-> [/opt/rocm/bin/rocprofv2] - SQ_LDS_BANK_CONFLICT
|-> [/opt/rocm/bin/rocprofv2] - SQ_LDS_ADDR_CONFLICT
|-> [/opt/rocm/bin/rocprofv2] - SQ_LDS_UNALIGNED_STALL
|-> [/opt/rocm/bin/rocprofv2] - SQ_WAVES_EQ_64
|-> [/opt/rocm/bin/rocprofv2] - SQ_WAVES_LT_64
|-> [/opt/rocm/bin/rocprofv2] - SQ_WAVES_LT_48
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_7.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_WAVES_LT_32
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_8.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_SMEM_NORM
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_MFMA
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_MFMA_I8
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/pmc_perf_9.txt
|-> [/opt/rocm/bin/rocprofv2] ROCProfilerV2: Collecting the following counters:
|-> [/opt/rocm/bin/rocprofv2] - SQ_INSTS_VALU_MFMA_MOPS_I8
[profiling] Current input file: tests/workloads/kernel/MI300A_A1/perfmon/timestamps.txt
|-> [/opt/rocm/bin/rocprofv2] vcopy testing on GCD 0
|-> [/opt/rocm/bin/rocprofv2] Finished allocating vectors on the CPU
|-> [/opt/rocm/bin/rocprofv2] Finished allocating vectors on the GPU
|-> [/opt/rocm/bin/rocprofv2] Finished copying vectors to the GPU
|-> [/opt/rocm/bin/rocprofv2] sw thinks it moved 1.000000 KB per wave
|-> [/opt/rocm/bin/rocprofv2] Total threads: 1048576, Grid Size: 4096 block Size:256, Wavefronts:16384:
|-> [/opt/rocm/bin/rocprofv2] Launching the kernel on the GPU
|-> [/opt/rocm/bin/rocprofv2] Finished executing kernel
|-> [/opt/rocm/bin/rocprofv2] Finished executing kernel
[roofline] Roofline temporarily disabled in MI300
@@ -0,0 +1,5 @@
pmc: GRBM_COUNT GRBM_GUI_ACTIVE SQ_WAVES SQ_IFETCH SQ_IFETCH_LEVEL SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_LDS SQ_INST_LEVEL_LDS SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_SMEM SQ_INST_LEVEL_SMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VMEM SQ_INST_LEVEL_VMEM SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: GRBM_COUNT GRBM_GUI_ACTIVE CPC_ME1_BUSY_FOR_PACKET_DECODE SQ_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_BUSY_CYCLES SQ_LEVEL_WAVES SQ_ACCUM_PREV_HIRES
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_CYCLES SQ_BUSY_CYCLES SQ_BUSY_CU_CYCLES SQ_WAVES SQ_WAVE_CYCLES SQ_INSTS_VALU_CVT SQ_INSTS_VMEM_WR SQ_INSTS_VMEM_RD GRBM_COUNT GRBM_GUI_ACTIVE TCP_GATE_EN1_sum TCP_GATE_EN2_sum TCP_TD_TCP_STALL_CYCLES_sum TCP_TCR_TCP_STALL_CYCLES_sum TA_TA_BUSY_sum TA_BUFFER_WAVEFRONTS_sum TD_TD_BUSY_sum TD_TC_STALL_sum SPI_CSN_WINDOW_VALID SPI_CSN_BUSY CPC_CPC_STAT_BUSY CPC_CPC_STAT_IDLE CPF_CPF_STAT_BUSY CPF_CPF_STAT_STALL TCC_CYCLE_sum TCC_BUSY_sum TCC_PROBE_sum TCC_PROBE_ALL_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VMEM SQ_INSTS_SALU SQ_INSTS_VSKIPPED SQ_INSTS SQ_INSTS_VALU SQ_INSTS_VALU_ADD_F16 SQ_INSTS_VALU_MUL_F16 SQ_INSTS_VALU_FMA_F16 GRBM_SPI_BUSY TCP_READ_TAGCONFLICT_STALL_CYCLES_sum TCP_WRITE_TAGCONFLICT_STALL_CYCLES_sum TCP_ATOMIC_TAGCONFLICT_STALL_CYCLES_sum TCP_TA_TCP_STATE_READ_sum TA_BUFFER_READ_WAVEFRONTS_sum TA_BUFFER_WRITE_WAVEFRONTS_sum TD_SPI_STALL_sum TD_LOAD_WAVEFRONT_sum SPI_CSN_NUM_THREADGROUPS SPI_CSN_WAVE CPC_CPC_TCIU_BUSY CPC_CPC_TCIU_IDLE CPF_CPF_TCIU_BUSY CPF_CPF_TCIU_STALL TCC_NC_REQ_sum TCC_UC_REQ_sum TCC_CC_REQ_sum TCC_RW_REQ_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQC_TC_DATA_ATOMIC_REQ SQC_TC_STALL SQC_TC_REQ SQC_DCACHE_REQ_READ_16 SQC_ICACHE_REQ SQC_ICACHE_HITS SQC_ICACHE_MISSES SQC_ICACHE_MISSES_DUPLICATE
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQC_DCACHE_INPUT_VALID_READYB SQC_DCACHE_ATOMIC SQC_DCACHE_REQ_READ_8 SQC_DCACHE_REQ SQC_DCACHE_HITS SQC_DCACHE_MISSES SQC_DCACHE_MISSES_DUPLICATE SQC_DCACHE_REQ_READ_1
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQC_DCACHE_REQ_READ_2 SQC_DCACHE_REQ_READ_4
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: TCC_ATOMIC[0] TCC_BUBBLE[0] TCC_CYCLE[0] TCC_EA0_ATOMIC[0] TCC_ATOMIC[1] TCC_BUBBLE[1] TCC_CYCLE[1] TCC_EA0_ATOMIC[1] TCC_ATOMIC[2] TCC_BUBBLE[2] TCC_CYCLE[2] TCC_EA0_ATOMIC[2] TCC_ATOMIC[3] TCC_BUBBLE[3] TCC_CYCLE[3] TCC_EA0_ATOMIC[3] TCC_ATOMIC[4] TCC_BUBBLE[4] TCC_CYCLE[4] TCC_EA0_ATOMIC[4] TCC_ATOMIC[5] TCC_BUBBLE[5] TCC_CYCLE[5] TCC_EA0_ATOMIC[5] TCC_ATOMIC[6] TCC_BUBBLE[6] TCC_CYCLE[6] TCC_EA0_ATOMIC[6] TCC_ATOMIC[7] TCC_BUBBLE[7] TCC_CYCLE[7] TCC_EA0_ATOMIC[7] TCC_ATOMIC[8] TCC_BUBBLE[8] TCC_CYCLE[8] TCC_EA0_ATOMIC[8] TCC_ATOMIC[9] TCC_BUBBLE[9] TCC_CYCLE[9] TCC_EA0_ATOMIC[9] TCC_ATOMIC[10] TCC_BUBBLE[10] TCC_CYCLE[10] TCC_EA0_ATOMIC[10] TCC_ATOMIC[11] TCC_BUBBLE[11] TCC_CYCLE[11] TCC_EA0_ATOMIC[11] TCC_ATOMIC[12] TCC_BUBBLE[12] TCC_CYCLE[12] TCC_EA0_ATOMIC[12] TCC_ATOMIC[13] TCC_BUBBLE[13] TCC_CYCLE[13] TCC_EA0_ATOMIC[13] TCC_ATOMIC[14] TCC_BUBBLE[14] TCC_CYCLE[14] TCC_EA0_ATOMIC[14] TCC_ATOMIC[15] TCC_BUBBLE[15] TCC_CYCLE[15] TCC_EA0_ATOMIC[15]
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: TCC_EA0_ATOMIC_LEVEL[0] TCC_EA0_RDREQ[0] TCC_EA0_RDREQ_32B[0] TCC_EA0_RDREQ_LEVEL[0] TCC_EA0_ATOMIC_LEVEL[1] TCC_EA0_RDREQ[1] TCC_EA0_RDREQ_32B[1] TCC_EA0_RDREQ_LEVEL[1] TCC_EA0_ATOMIC_LEVEL[2] TCC_EA0_RDREQ[2] TCC_EA0_RDREQ_32B[2] TCC_EA0_RDREQ_LEVEL[2] TCC_EA0_ATOMIC_LEVEL[3] TCC_EA0_RDREQ[3] TCC_EA0_RDREQ_32B[3] TCC_EA0_RDREQ_LEVEL[3] TCC_EA0_ATOMIC_LEVEL[4] TCC_EA0_RDREQ[4] TCC_EA0_RDREQ_32B[4] TCC_EA0_RDREQ_LEVEL[4] TCC_EA0_ATOMIC_LEVEL[5] TCC_EA0_RDREQ[5] TCC_EA0_RDREQ_32B[5] TCC_EA0_RDREQ_LEVEL[5] TCC_EA0_ATOMIC_LEVEL[6] TCC_EA0_RDREQ[6] TCC_EA0_RDREQ_32B[6] TCC_EA0_RDREQ_LEVEL[6] TCC_EA0_ATOMIC_LEVEL[7] TCC_EA0_RDREQ[7] TCC_EA0_RDREQ_32B[7] TCC_EA0_RDREQ_LEVEL[7] TCC_EA0_ATOMIC_LEVEL[8] TCC_EA0_RDREQ[8] TCC_EA0_RDREQ_32B[8] TCC_EA0_RDREQ_LEVEL[8] TCC_EA0_ATOMIC_LEVEL[9] TCC_EA0_RDREQ[9] TCC_EA0_RDREQ_32B[9] TCC_EA0_RDREQ_LEVEL[9] TCC_EA0_ATOMIC_LEVEL[10] TCC_EA0_RDREQ[10] TCC_EA0_RDREQ_32B[10] TCC_EA0_RDREQ_LEVEL[10] TCC_EA0_ATOMIC_LEVEL[11] TCC_EA0_RDREQ[11] TCC_EA0_RDREQ_32B[11] TCC_EA0_RDREQ_LEVEL[11] TCC_EA0_ATOMIC_LEVEL[12] TCC_EA0_RDREQ[12] TCC_EA0_RDREQ_32B[12] TCC_EA0_RDREQ_LEVEL[12] TCC_EA0_ATOMIC_LEVEL[13] TCC_EA0_RDREQ[13] TCC_EA0_RDREQ_32B[13] TCC_EA0_RDREQ_LEVEL[13] TCC_EA0_ATOMIC_LEVEL[14] TCC_EA0_RDREQ[14] TCC_EA0_RDREQ_32B[14] TCC_EA0_RDREQ_LEVEL[14] TCC_EA0_ATOMIC_LEVEL[15] TCC_EA0_RDREQ[15] TCC_EA0_RDREQ_32B[15] TCC_EA0_RDREQ_LEVEL[15]
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: TCC_EA0_WRREQ[0] TCC_EA0_WRREQ_64B[0] TCC_EA0_WRREQ_LEVEL[0] TCC_HIT[0] TCC_EA0_WRREQ[1] TCC_EA0_WRREQ_64B[1] TCC_EA0_WRREQ_LEVEL[1] TCC_HIT[1] TCC_EA0_WRREQ[2] TCC_EA0_WRREQ_64B[2] TCC_EA0_WRREQ_LEVEL[2] TCC_HIT[2] TCC_EA0_WRREQ[3] TCC_EA0_WRREQ_64B[3] TCC_EA0_WRREQ_LEVEL[3] TCC_HIT[3] TCC_EA0_WRREQ[4] TCC_EA0_WRREQ_64B[4] TCC_EA0_WRREQ_LEVEL[4] TCC_HIT[4] TCC_EA0_WRREQ[5] TCC_EA0_WRREQ_64B[5] TCC_EA0_WRREQ_LEVEL[5] TCC_HIT[5] TCC_EA0_WRREQ[6] TCC_EA0_WRREQ_64B[6] TCC_EA0_WRREQ_LEVEL[6] TCC_HIT[6] TCC_EA0_WRREQ[7] TCC_EA0_WRREQ_64B[7] TCC_EA0_WRREQ_LEVEL[7] TCC_HIT[7] TCC_EA0_WRREQ[8] TCC_EA0_WRREQ_64B[8] TCC_EA0_WRREQ_LEVEL[8] TCC_HIT[8] TCC_EA0_WRREQ[9] TCC_EA0_WRREQ_64B[9] TCC_EA0_WRREQ_LEVEL[9] TCC_HIT[9] TCC_EA0_WRREQ[10] TCC_EA0_WRREQ_64B[10] TCC_EA0_WRREQ_LEVEL[10] TCC_HIT[10] TCC_EA0_WRREQ[11] TCC_EA0_WRREQ_64B[11] TCC_EA0_WRREQ_LEVEL[11] TCC_HIT[11] TCC_EA0_WRREQ[12] TCC_EA0_WRREQ_64B[12] TCC_EA0_WRREQ_LEVEL[12] TCC_HIT[12] TCC_EA0_WRREQ[13] TCC_EA0_WRREQ_64B[13] TCC_EA0_WRREQ_LEVEL[13] TCC_HIT[13] TCC_EA0_WRREQ[14] TCC_EA0_WRREQ_64B[14] TCC_EA0_WRREQ_LEVEL[14] TCC_HIT[14] TCC_EA0_WRREQ[15] TCC_EA0_WRREQ_64B[15] TCC_EA0_WRREQ_LEVEL[15] TCC_HIT[15]
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: TCC_MISS[0] TCC_READ[0] TCC_REQ[0] TCC_RW_REQ[0] TCC_MISS[1] TCC_READ[1] TCC_REQ[1] TCC_RW_REQ[1] TCC_MISS[2] TCC_READ[2] TCC_REQ[2] TCC_RW_REQ[2] TCC_MISS[3] TCC_READ[3] TCC_REQ[3] TCC_RW_REQ[3] TCC_MISS[4] TCC_READ[4] TCC_REQ[4] TCC_RW_REQ[4] TCC_MISS[5] TCC_READ[5] TCC_REQ[5] TCC_RW_REQ[5] TCC_MISS[6] TCC_READ[6] TCC_REQ[6] TCC_RW_REQ[6] TCC_MISS[7] TCC_READ[7] TCC_REQ[7] TCC_RW_REQ[7] TCC_MISS[8] TCC_READ[8] TCC_REQ[8] TCC_RW_REQ[8] TCC_MISS[9] TCC_READ[9] TCC_REQ[9] TCC_RW_REQ[9] TCC_MISS[10] TCC_READ[10] TCC_REQ[10] TCC_RW_REQ[10] TCC_MISS[11] TCC_READ[11] TCC_REQ[11] TCC_RW_REQ[11] TCC_MISS[12] TCC_READ[12] TCC_REQ[12] TCC_RW_REQ[12] TCC_MISS[13] TCC_READ[13] TCC_REQ[13] TCC_RW_REQ[13] TCC_MISS[14] TCC_READ[14] TCC_REQ[14] TCC_RW_REQ[14] TCC_MISS[15] TCC_READ[15] TCC_REQ[15] TCC_RW_REQ[15]
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: TCC_TAG_STALL[0] TCC_TOO_MANY_EA_WRREQS_STALL[0] TCC_WRITE[0] TCC_TAG_STALL[1] TCC_TOO_MANY_EA_WRREQS_STALL[1] TCC_WRITE[1] TCC_TAG_STALL[2] TCC_TOO_MANY_EA_WRREQS_STALL[2] TCC_WRITE[2] TCC_TAG_STALL[3] TCC_TOO_MANY_EA_WRREQS_STALL[3] TCC_WRITE[3] TCC_TAG_STALL[4] TCC_TOO_MANY_EA_WRREQS_STALL[4] TCC_WRITE[4] TCC_TAG_STALL[5] TCC_TOO_MANY_EA_WRREQS_STALL[5] TCC_WRITE[5] TCC_TAG_STALL[6] TCC_TOO_MANY_EA_WRREQS_STALL[6] TCC_WRITE[6] TCC_TAG_STALL[7] TCC_TOO_MANY_EA_WRREQS_STALL[7] TCC_WRITE[7] TCC_TAG_STALL[8] TCC_TOO_MANY_EA_WRREQS_STALL[8] TCC_WRITE[8] TCC_TAG_STALL[9] TCC_TOO_MANY_EA_WRREQS_STALL[9] TCC_WRITE[9] TCC_TAG_STALL[10] TCC_TOO_MANY_EA_WRREQS_STALL[10] TCC_WRITE[10] TCC_TAG_STALL[11] TCC_TOO_MANY_EA_WRREQS_STALL[11] TCC_WRITE[11] TCC_TAG_STALL[12] TCC_TOO_MANY_EA_WRREQS_STALL[12] TCC_WRITE[12] TCC_TAG_STALL[13] TCC_TOO_MANY_EA_WRREQS_STALL[13] TCC_WRITE[13] TCC_TAG_STALL[14] TCC_TOO_MANY_EA_WRREQS_STALL[14] TCC_WRITE[14] TCC_TAG_STALL[15] TCC_TOO_MANY_EA_WRREQS_STALL[15] TCC_WRITE[15]
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VALU_TRANS_F16 SQ_INSTS_VALU_ADD_F32 SQ_INSTS_VALU_MUL_F32 SQ_INSTS_VALU_FMA_F32 SQ_INSTS_VALU_TRANS_F32 SQ_INSTS_VALU_ADD_F64 SQ_INSTS_VALU_MUL_F64 SQ_INSTS_VALU_FMA_F64 TCP_VOLATILE_sum TCP_TOTAL_ACCESSES_sum TCP_TOTAL_READ_sum TCP_TOTAL_WRITE_sum TA_BUFFER_ATOMIC_WAVEFRONTS_sum TA_BUFFER_TOTAL_CYCLES_sum TD_ATOMIC_WAVEFRONT_sum TD_STORE_WAVEFRONT_sum SPI_RA_REQ_NO_ALLOC SPI_RA_REQ_NO_ALLOC_CSN CPC_CPC_STAT_STALL CPC_UTCL1_STALL_ON_TRANSLATION CPF_CPF_STAT_IDLE CPF_CPF_TCIU_IDLE TCC_REQ_sum TCC_STREAMING_REQ_sum TCC_HIT_sum TCC_MISS_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VALU_TRANS_F64 SQ_INSTS_VALU_INT32 SQ_INSTS_VALU_INT64 SQ_INSTS_SMEM SQ_INSTS_FLAT SQ_INSTS_LDS SQ_INSTS_GDS SQ_INSTS_EXP_GDS TCP_TOTAL_ATOMIC_WITH_RET_sum TCP_TOTAL_ATOMIC_WITHOUT_RET_sum TCP_TOTAL_WRITEBACK_INVALIDATES_sum TCP_TOTAL_CACHE_ACCESSES_sum TA_BUFFER_COALESCED_READ_CYCLES_sum TA_BUFFER_COALESCED_WRITE_CYCLES_sum TD_COALESCABLE_WAVEFRONT_sum SPI_RA_RES_STALL_CSN SPI_RA_TMP_STALL_CSN CPC_CPC_UTCL2IU_BUSY CPC_CPC_UTCL2IU_IDLE CPF_CMP_UTCL1_STALL_ON_TRANSLATION TCC_READ_sum TCC_WRITE_sum TCC_ATOMIC_sum TCC_WRITEBACK_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_BRANCH SQ_INSTS_SENDMSG SQ_WAIT_ANY SQ_WAIT_INST_ANY SQ_ACTIVE_INST_ANY SQ_ACTIVE_INST_VMEM SQ_ACTIVE_INST_LDS SQ_ACTIVE_INST_VALU TCP_UTCL1_TRANSLATION_MISS_sum TCP_UTCL1_TRANSLATION_HIT_sum TCP_UTCL1_PERMISSION_MISS_sum TCP_UTCL1_REQUEST_sum TA_ADDR_STALLED_BY_TC_CYCLES_sum TA_TOTAL_WAVEFRONTS_sum SPI_RA_WAVE_SIMD_FULL_CSN SPI_RA_VGPR_SIMD_FULL_CSN CPC_CPC_UTCL2IU_STALL CPC_ME1_BUSY_FOR_PACKET_DECODE TCC_EA0_WRREQ_sum TCC_EA0_WRREQ_64B_sum TCC_EA0_WR_UNCACHED_32B_sum TCC_EA0_WRREQ_DRAM_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_ACTIVE_INST_SCA SQ_ACTIVE_INST_EXP_GDS SQ_ACTIVE_INST_MISC SQ_ACTIVE_INST_FLAT SQ_INST_CYCLES_VMEM_WR SQ_INST_CYCLES_VMEM_RD SQ_INST_CYCLES_SMEM SQ_INST_CYCLES_SALU TCP_TCC_READ_REQ_sum TCP_TCC_WRITE_REQ_sum TCP_TCC_ATOMIC_WITH_RET_REQ_sum TCP_TCC_ATOMIC_WITHOUT_RET_REQ_sum TA_ADDR_STALLED_BY_TD_CYCLES_sum TA_DATA_STALLED_BY_TC_CYCLES_sum SPI_RA_SGPR_SIMD_FULL_CSN SPI_RA_LDS_CU_FULL_CSN CPC_ME1_DC0_SPI_BUSY TCC_EA0_RDREQ_sum TCC_EA0_RDREQ_32B_sum TCC_BUBBLE_sum TCC_EA0_RD_UNCACHED_32B_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_THREAD_CYCLES_VALU SQ_IFETCH SQ_LDS_BANK_CONFLICT SQ_LDS_ADDR_CONFLICT SQ_LDS_UNALIGNED_STALL SQ_WAVES_EQ_64 SQ_WAVES_LT_64 SQ_WAVES_LT_48 TCP_TCC_NC_READ_REQ_sum TCP_TCC_NC_WRITE_REQ_sum TCP_TCC_NC_ATOMIC_REQ_sum TCP_TCC_UC_READ_REQ_sum TA_FLAT_WAVEFRONTS_sum TA_FLAT_READ_WAVEFRONTS_sum SPI_RA_BAR_CU_FULL_CSN SPI_RA_TGLIM_CU_FULL_CSN TCC_EA0_RDREQ_DRAM_sum TCC_TAG_STALL_sum TCC_NORMAL_WRITEBACK_sum TCC_ALL_TC_OP_WB_WRITEBACK_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_WAVES_LT_32 SQ_WAVES_LT_16 SQ_ITEMS SQ_LDS_MEM_VIOLATIONS SQ_LDS_ATOMIC_RETURN SQ_LDS_IDX_ACTIVE SQ_WAVES_RESTORED SQ_WAVES_SAVED TCP_TCC_UC_WRITE_REQ_sum TCP_TCC_UC_ATOMIC_REQ_sum TCP_TCC_CC_READ_REQ_sum TCP_TCC_CC_WRITE_REQ_sum TA_FLAT_WRITE_WAVEFRONTS_sum TA_FLAT_ATOMIC_WAVEFRONTS_sum SPI_RA_WVLIM_STALL_CSN SPI_SWC_CSC_WR TCC_NORMAL_EVICT_sum TCC_ALL_TC_OP_INV_EVICT_sum TCC_TOO_MANY_EA_WRREQS_STALL_sum TCC_EA0_ATOMIC_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_SMEM_NORM SQ_INSTS_MFMA SQ_INSTS_VALU_MFMA_I8 SQ_INSTS_VALU_MFMA_F16 SQ_INSTS_VALU_MFMA_BF16 SQ_INSTS_VALU_MFMA_F32 SQ_INSTS_VALU_MFMA_F64 SQ_VALU_MFMA_BUSY_CYCLES TCP_TCC_CC_ATOMIC_REQ_sum TCP_TCC_RW_READ_REQ_sum TCP_TCC_RW_WRITE_REQ_sum TCP_TCC_RW_ATOMIC_REQ_sum SPI_VWC_CSC_WR SPI_RA_BULKY_CU_FULL_CSN TCC_EA0_RDREQ_LEVEL_sum TCC_EA0_WRREQ_LEVEL_sum TCC_EA0_ATOMIC_LEVEL_sum TCC_EA0_WRREQ_STALL_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc: SQ_INSTS_VALU_MFMA_MOPS_I8 SQ_INSTS_VALU_MFMA_MOPS_F16 SQ_INSTS_VALU_MFMA_MOPS_BF16 SQ_INSTS_VALU_MFMA_MOPS_F32 SQ_INSTS_VALU_MFMA_MOPS_F64 SQC_TC_INST_REQ SQC_TC_DATA_READ_REQ SQC_TC_DATA_WRITE_REQ TCP_PENDING_STALL_CYCLES_sum
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,5 @@
pmc:
gpu:
range:
kernel: "vecCopy(double*,,double*,,double*,,int,,int),[clone,.kd]"
@@ -0,0 +1,4 @@
Dispatch_ID,Kernel_Name,GPU_ID
0,"vecCopy(double*, double*, double*, int, int) (.kd)",11995
1,"vecCopy(double*, double*, double*, int, int) (.kd)",11995
2,"vecCopy(double*, double*, double*, int, int) (.kd)",11995
1 Dispatch_ID Kernel_Name GPU_ID
2 0 vecCopy(double*, double*, double*, int, int) (.kd) 11995
3 1 vecCopy(double*, double*, double*, int, int) (.kd) 11995
4 2 vecCopy(double*, double*, double*, int, int) (.kd) 11995

Some files were not shown because too many files have changed in this diff Show More