/home/tgrogers/github/purdue-aalp/gpgpu-sim_simulations/util/correlation/../../benchmarks/bin/9.1/release/transpose Starting...

> Device 0: "Quadro GV100"
> SM Capability 7.0 detected:
> [Quadro GV100] has 80 MP(s) x -1 (Cores/MP) = -80 (Cores)
> Compute performance scaling factor = 1.00
> MatrixSize X = 128
> MatrixSize Y = 128

Matrix size: 128x128 (8x8 tiles), tile size: 16x16, block size: 16x16


transpose-Outer-simple copy       , Throughput = 4.3153 GB/s, Time = 0.02829 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
transpose-Inner-simple copy       , Throughput = 4.3104 GB/s, Time = 0.02832 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256

transpose-Outer-shared memory copy, Throughput = 4.9671 GB/s, Time = 0.02458 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
transpose-Inner-shared memory copy, Throughput = 4.7037 GB/s, Time = 0.02595 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256

transpose-Outer-naive             , Throughput = 5.2472 GB/s, Time = 0.02326 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
transpose-Inner-naive             , Throughput = 4.7803 GB/s, Time = 0.02554 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256

transpose-Outer-coalesced         , Throughput = 5.4496 GB/s, Time = 0.02240 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
transpose-Inner-coalesced         , Throughput = 4.6864 GB/s, Time = 0.02605 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256

transpose-Outer-optimized         , Throughput = 5.4186 GB/s, Time = 0.02253 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
transpose-Inner-optimized         , Throughput = 4.8226 GB/s, Time = 0.02531 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256

transpose-Outer-coarse-grained    , Throughput = 5.2762 GB/s, Time = 0.02314 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
transpose-Inner-coarse-grained    , Throughput = 4.9222 GB/s, Time = 0.02480 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256

transpose-Outer-fine-grained      , Throughput = 5.6264 GB/s, Time = 0.02170 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
transpose-Inner-fine-grained      , Throughput = 4.8287 GB/s, Time = 0.02528 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256

transpose-Outer-diagonal          , Throughput = 5.5689 GB/s, Time = 0.02192 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
transpose-Inner-diagonal          , Throughput = 4.2012 GB/s, Time = 0.02906 s, Size = 16384 fp32 elements, NumDevsUsed = 1, Workgroup = 256
