-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathncu_async8_4096_details.csv
More file actions
We can make this file beautiful and searchable if this error is corrected: It looks like row 2 should actually have 20 columns, instead of 16 in line 1.
116 lines (116 loc) · 31 KB
/
Copy pathncu_async8_4096_details.csv
File metadata and controls
116 lines (116 loc) · 31 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
"ID","Process ID","Process Name","Host Name","Kernel Name","Context","Stream","Block Size","Grid Size","Device","CC","Section Name","Metric Name","Metric Unit","Metric Value","Rule Name","Rule Type","Rule Description","Estimated Speedup Type","Estimated Speedup"
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","gpu__compute_memory_throughput.avg.pct_of_peak_sustained_elapsed","%","39.26",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","gpu__dram_throughput.avg.pct_of_peak_sustained_elapsed","%","9.56",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","gpu__time_duration.sum","ms","35.61",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","l1tex__throughput.avg.pct_of_peak_sustained_active","%","39.97",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","launch__occupancy_limit_registers","block","8",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","launch__occupancy_limit_shared_mem","block","5",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","launch__occupancy_limit_warps","block","6",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","launch__occupancy_per_register_count","","4,648",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","launch__occupancy_per_shared_mem_size","","3,096",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","launch__registers_per_thread","register/thread","30",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","launch__shared_mem_per_block","Kbyte/block","17.41",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","lts__throughput.avg.pct_of_peak_sustained_elapsed","%","14.53",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","sm__throughput.avg.pct_of_peak_sustained_elapsed","%","33.60",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Command line profiler metrics","sm__warps_active.avg.pct_of_peak_sustained_active","%","82.98",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","DRAM Frequency","Ghz","12.79",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","SM Frequency","Ghz","2.32",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","Elapsed Cycles","cycle","82,494,459",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","Memory Throughput","%","39.26",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","DRAM Throughput","%","9.56",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","Duration","ms","35.61",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","L1/TEX Cache Throughput","%","39.97",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","L2 Cache Throughput","%","14.53",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","SM Active Cycles","cycle","80,982,398.27",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU Speed Of Light Throughput","Compute (SM) Throughput","%","33.60",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","SpeedOfLight","","","","SOLBottleneck","OPT","This workload exhibits low compute throughput and memory bandwidth utilization relative to the peak performance of this device. Achieved compute throughput and/or memory bandwidth below 60.0% of peak typically indicate latency issues. Look at Scheduler Statistics and Warp State Statistics for potential reasons.","",""
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","SpeedOfLight_RooflineChart","","","","SOLFPRoofline","INF","The ratio of peak float (FP32) to double (FP64) performance on this device is 64:1. The workload achieved 0% of this device's FP32 peak performance and 0% of its FP64 peak performance. See the Profiling Guide (https://docs.nvidia.com/nsight-compute/ProfilingGuide/index.html#roofline) for more details on roofline analysis.","",""
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","PM Sampling","Maximum Buffer Size","Mbyte","30.74",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","PM Sampling","Maximum Sampling Interval","us","12",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","PM Sampling","# Pass Groups","","2",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Compute Workload Analysis","Executed Ipc Active","inst/cycle","1.37",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Compute Workload Analysis","Executed Ipc Elapsed","inst/cycle","1.34",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Compute Workload Analysis","Issue Slots Busy","%","33.60",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Compute Workload Analysis","Issued Ipc Active","inst/cycle","1.37",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Compute Workload Analysis","SM Busy","%","33.60",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","ComputeWorkloadAnalysis","","","","HighPipeUtilization","INF","ALU is the highest-utilized pipeline (20.2%) based on elapsed cycles in the workload, taking into account the rates of its different instructions. It executes integer and logic operations. It is well-utilized, but should not be a bottleneck.","",""
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","Local Memory Spilling Requests","","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","Local Memory Spilling Request Overhead","%","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","L2 Sector Promotion Misses","%","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","Shared Memory Spilling Requests","","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","Shared Memory Spilling Request Overhead","%","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","Memory Throughput","Gbyte/s","39.14",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","Mem Busy","%","39.26",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","Max Bandwidth","%","25.81",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","L1/TEX Hit Rate","%","75.74",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","L2 Persisting Size","Mbyte","4.72",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","L2 Compression Success Rate","%","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","L2 Compression Ratio","","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","L2 Compression Input Sectors","sector","8",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","L2 Hit Rate","%","67.89",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Memory Workload Analysis","Mem Pipes Busy","%","25.81",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","MemoryWorkloadAnalysis_Tables","","","","SharedMemoryConflicts","OPT","The memory access pattern for shared loads might not be optimal and causes on average a 8.1 - way bank conflict across all 33554432 shared load requests.This results in 136211555 bank conflicts, which represent 50.37% of the overall 270424299 wavefronts for shared loads. Check the Source Counters section for uncoalesced shared loads.","global","20.13"
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Scheduler Statistics","One or More Eligible","%","34.31",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Scheduler Statistics","Issued Warp Per Scheduler","","0.34",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Scheduler Statistics","No Eligible","%","65.69",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Scheduler Statistics","Active Warps Per Scheduler","warp","9.97",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Scheduler Statistics","Eligible Warps Per Scheduler","warp","0.63",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","SchedulerStats","","","","IssueSlotUtilization","OPT","Every scheduler is capable of issuing one instruction per cycle, but for this workload each scheduler only issues an instruction every 2.9 cycles. This might leave hardware resources underutilized and may lead to less optimal performance. Out of the maximum of 12 warps per scheduler, this workload allocates an average of 9.97 active warps per scheduler, but only an average of 0.63 warps were eligible per cycle. Eligible warps are the subset of active warps that are ready to issue their next instruction. Every cycle with no eligible warp results in no instruction being issued and the issue slot remains unused. To increase the number of eligible warps, avoid possible load imbalances due to highly different execution durations per warp. Reducing stalls indicated on the Warp State Statistics and Source Counters sections can help, too.","local","60.74"
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Warp State Statistics","Warp Cycles Per Issued Instruction","cycle","29.07",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Warp State Statistics","Warp Cycles Per Executed Instruction","cycle","29.07",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Warp State Statistics","Avg. Active Threads Per Warp","","32.00",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Warp State Statistics","Avg. Not Predicated Off Threads Per Warp","","31.35",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","WarpStateStats","","","","CPIStall","OPT","On average, each warp of this workload spends 20.6 cycles being stalled waiting for a scoreboard dependency on a L1TEX (local, global, surface, texture) operation. Find the instruction producing the data being waited upon to identify the culprit. To reduce the number of cycles waiting on L1TEX data accesses verify the memory access patterns are optimal for the target architecture, attempt to increase cache hit rates by increasing data locality (coalescing), or by changing the cache configuration. Consider moving frequently used data to shared memory. This stall type represents about 70.8% of the total average of 29.1 cycles between issuing two instructions.","global","60.74"
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","WarpStateStats","","","","CPIStall","INF","Check the Warp Stall Sampling (All Samples) table for the top stall locations in your source based on sampling data. The Profiling Guide (https://docs.nvidia.com/nsight-compute/ProfilingGuide/index.html#metrics-reference) provides more details on each stall reason.","",""
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Instruction Statistics","Local Memory Spilling Requests","byte","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Instruction Statistics","Shared Memory Spilling Requests","byte","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Instruction Statistics","Avg. Executed Instructions Per Scheduler","inst","27,702,723.99",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Instruction Statistics","Executed Instructions","inst","3,324,326,879",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Instruction Statistics","Avg. Issued Instructions Per Scheduler","inst","27,702,833.77",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Instruction Statistics","Issued Instructions","inst","3,324,340,052",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Block Size","","256",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Cluster Scheduling Policy","","PolicySpread",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Cluster Size","","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Function Cache Configuration","","CachePreferNone",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Grid Size","","8,192",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Preferred Cluster Size","","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Registers Per Thread","register/thread","30",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Shared Memory Configuration Size","Kbyte","102.40",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Driver Shared Memory Per Block","Kbyte/block","1.02",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Dynamic Shared Memory Per Block","Kbyte/block","16.38",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Static Shared Memory Per Block","byte/block","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","# SMs","SM","30",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Stack Size","","1,024",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Threads","thread","2,097,152",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","# TPCs","","15",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Enabled TPC IDs","","all",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Uses Green Context","","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Launch Statistics","Waves Per SM","","54.61",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Max Active Clusters","cluster","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Max Cluster Size","block","8",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Overall GPU Occupancy","%","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Cluster Occupancy","%","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Block Limit Barriers","block","24",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Block Limit SM","block","24",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Block Limit Registers","block","8",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Block Limit Shared Mem","block","5",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Block Limit Warps","block","6",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Theoretical Active Warps per SM","warp","40",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Theoretical Occupancy","%","83.33",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Achieved Occupancy","%","82.98",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Occupancy","Achieved Active Warps Per SM","warp","39.83",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Average DRAM Active Cycles","cycle","43,562,720",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Total DRAM Elapsed Cycles","cycle","1,822,346,240",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Average L1 Active Cycles","cycle","80,982,398.27",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Total L1 Elapsed Cycles","cycle","2,473,583,640",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Average L2 Active Cycles","cycle","73,554,667.42",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Total L2 Elapsed Cycles","cycle","891,819,840",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Average SM Active Cycles","cycle","80,982,398.27",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Total SM Elapsed Cycles","cycle","2,473,583,640",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Average SMSP Active Cycles","cycle","80,741,333.16",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","GPU and Memory Workload Distribution","Total SMSP Elapsed Cycles","cycle","9,894,334,560",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Source Counters","Branch Instructions Ratio","%","0.06",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Source Counters","Branch Instructions","inst","201,528,987",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Source Counters","Branch Efficiency","%","100",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","Source Counters","Avg. Divergent Branches","branches","0",
"0","49808","profile_async8_4096.exe","127.0.0.1","matrixMul_wmma_async(const __half *, const __half *, float *, int, int, int)","1","7","(32, 8, 1)","(256, 32, 1)","0","12.0","SourceCounters","","","","UncoalescedSharedAccess","OPT","This workload has uncoalesced shared accesses resulting in a total of 134217728 excessive wavefronts (33% of the total 402653184 wavefronts). Check the L1 Wavefronts Shared Excessive table for the primary source locations. The CUDA Best Practices Guide (https://docs.nvidia.com/cuda/cuda-c-best-practices-guide/index.html#shared-memory-in-matrix-multiplication-c-ab) has an example on optimizing shared memory accesses.","global","32.74"