GPU Kernel Information Aggregated by Name
kernel_name | kernel_count | kernel_duration (us) | model_duration_percentage | kernel_flops | kernel_dram_read_bytes | kernel_dram_write_bytes | kernel_achieved_occupancy (%) | kernel_arithmetic_intensity (flops/byte) | kernel_arithmetic_throughput (GFlops) | kernel_memory_bound |
---|
kernel_name | kernel_count | kernel_duration (us) | model_duration_percentage | kernel_flops | kernel_dram_read_bytes | kernel_dram_write_bytes | kernel_achieved_occupancy (%) | kernel_arithmetic_intensity (flops/byte) | kernel_arithmetic_throughput (GFlops) | kernel_memory_bound |
---|---|---|---|---|---|---|---|---|---|---|
cudnn::maxwell::gemm::computeOffsetsKernel(cudnn::maxwell::gemm::ComputeOffsetsParams) | 5 | 25.67 | 0.05 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
maxwell_scudnn_128x128_relu_interior_nn | 0 | 467.33 | 0.91 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
maxwell_scudnn_128x128_relu_small_nn | 0 | 3442.33 | 6.67 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
maxwell_scudnn_128x32_relu_small_nn | 0 | 79.33 | 0.15 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
maxwell_scudnn_128x64_relu_interior_nn | 1 | 1156.00 | 2.24 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
maxwell_scudnn_128x64_relu_small_nn | 0 | 3912.00 | 7.58 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
maxwell_scudnn_winograd_128x128_ldg1_ldg4_tile148n_nt | 9 | 30490.67 | 59.08 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
void cudnn::detail::bn_fw_inf_1C11_kernel_NCHW<float, float, true, 1>(float, float, cudnnTensorStruct, float const*, cudnnTensorStruct, float*, cudnnTensorStruct, float const*, float const*, float const*, float const*, float) | 14 | 3463.34 | 6.71 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
void cudnn::detail::pooling_fw_4d_kernel<float, float, cudnn::detail::averpooling_func<float>, 1, false>(cudnnTensorStruct, float const*, cudnnTensorStruct, float*, cudnnPoolingStruct, float, float, int, cudnn::reduced_divisor, cudnn::reduced_divisor) | 0 | 118.67 | 0.23 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
void cudnn::winograd::generateWinogradTilesKernel<0, float, float>(cudnn::winograd::GenerateWinogradTilesParams<float, float>) | 9 | 1300.00 | 2.52 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
void gemv2T_kernel_val<int, int, float, float, float, 128, 16, 2, 2, false, cublasGemvParams<cublasGemvTensor<float const>, cublasGemvTensor<float>, float> >(cublasGemvParams<cublasGemvTensor<float const>, cublasGemvTensor<float>, float>, float, float) | 0 | 10.00 | 0.02 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
void mshadow::cuda::MapPlanKernel<mshadow::sv::plusto, 8, mshadow::expr::Plan<mshadow::Tensor<mshadow::gpu, 2, float>, float>, mshadow::expr::Plan<mshadow::expr::Broadcast1DExp<mshadow::Tensor<mshadow::gpu, 1, float>, float, 2, 1>, float> >(mshadow::expr::Plan<mshadow::Tensor<mshadow::gpu, 2, float>, float>, int, mshadow::Shape<2>, mshadow::expr::Plan<mshadow::expr::Broadcast1DExp<mshadow::Tensor<mshadow::gpu, 1, float>, float, 2, 1>, float>) | 0 | 3.00 | 0.01 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
void mshadow::cuda::MapPlanKernel<mshadow::sv::saveto, 8, mshadow::expr::Plan<mshadow::Tensor<mshadow::gpu, 1, float>, float>, mshadow::expr::Plan<mshadow::expr::ScalarExp<float>, float> >(mshadow::expr::Plan<mshadow::Tensor<mshadow::gpu, 1, float>, float>, int, mshadow::Shape<2>, mshadow::expr::Plan<mshadow::expr::ScalarExp<float>, float>) | 0 | 3.00 | 0.01 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
void mxnet::op::mxnet_op::mxnet_generic_kernel<mxnet::op::mxnet_op::op_with_req<mxnet::op::mshadow_op::plus, 1>, float*, float*, float*>(int, float*, float*, float*) | 5 | 2261.67 | 4.38 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
void op_generic_tensor_kernel<2, float, float, float, 256, (cudnnGenericOp_t)8, (cudnnNanPropagation_t)0, (cudnnDimOrder_t)0, 1>(cudnnTensorStruct, float*, cudnnTensorStruct, float const*, cudnnTensorStruct, float const*, float, float, float, float, dimArray, reducedDivisorArray, bool) | 12 | 3307.00 | 6.41 | 0 | 0.00 | 0.00 | 0.00 | 0.00 | 0.00 | true |
Showing 1 to 15 of 15 entries