
NOTICE: Existing SQLite export found: /report/profiles/node_5/nsys/d0.sqlite
        It is assumed file was previously exported from: /report/profiles/node_5/nsys/d0.nsys-rep
        Consider using --force-export=true if needed.

Processing [/report/profiles/node_5/nsys/d0.sqlite] with [/opt/nvidia/nsight-systems-cli/2026.4.1/target-linux-x64/reports/cuda_gpu_kern_sum.py]... 

 ** CUDA GPU Kernel Summary (cuda_gpu_kern_sum):

 Time (%)  Total Time (ns)  Instances   Avg (ns)     Med (ns)    Min (ns)   Max (ns)   StdDev (ns)                                                  Name                                                
 --------  ---------------  ---------  -----------  -----------  --------  ----------  -----------  ----------------------------------------------------------------------------------------------------
     49.8   14,590,525,348      4,590  3,178,763.7  3,891,418.0   284,937   3,949,675  1,438,067.1  _fwd_kernel                                                                                         
     31.9    9,340,217,936     30,600    305,235.9    288,426.0   215,750  52,394,336    348,000.6  ncclDevKernel_AllReduce_Sum_bf16_RING_LL(ncclDevKernelArgsStorage<(unsigned long)4096>)             
      3.4      995,286,483     28,152     35,354.0     32,577.0    12,864      73,090     11,589.5  void cutlass::device_kernel<cutlass::gemm::kernel::GemmUniversal<cutlass::gemm::GroupProblemShape<c…
      2.9      846,418,308     14,076     60,132.0     60,066.0    57,634      73,538        928.6  void cutlass::Kernel2<cutlass_80_tensorop_s16816gemm_bf16_256x64_32x4_tn_align8>(T1::Params)        
      2.6      767,151,972     33,048     23,213.3      6,433.0     4,000      50,722     20,092.6  void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_16x16_128x2_tn_align8>(T1::Par…
      1.8      532,711,449        459  1,160,591.4     89,891.0    53,761  10,032,449  2,298,458.7  ncclDevKernel_AllGather_RING_LL(ncclDevKernelArgsStorage<(unsigned long)4096>)                      
      1.4      423,285,388     10,557     40,095.2     40,129.0    38,850      41,377        418.0  fused_sigmoid_gating_delta_rule_update_kernel                                                       
      1.3      379,086,218     26,622     14,239.6     17,761.0     5,152      27,009      7,252.8  void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x2_tn_align8>(T1::Para…
      0.6      186,309,304     29,376      6,342.2      6,240.0     4,416       8,192        606.2  _score_kernel                                                                                       
      0.4      116,686,329     14,076      8,289.7      8,225.0     5,409      13,345      1,028.3  void tensorrt_llm::kernels::cutlass_kernels::finalizeMoeRoutingKernel<__nv_bfloat16, __nv_bfloat16,…
      0.3      100,782,270      3,672     27,446.2     27,457.0    25,761      29,889        586.0  kernel_cutlass__dsv3_fused_a_gemm_kernel_tensorptri32gmemo2112358435841_tensorptri32gmemo1635843584…
      0.3       78,720,700        153    514,514.4    514,447.0   512,207     531,088      1,661.3  void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x1_tn_align8>(T1::Par…
      0.2       71,900,977     18,666      3,852.0      4,128.0     2,368       4,832        653.2  void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x1_tn_align8>(T1::Para…
      0.2       71,014,349     27,234      2,607.6      2,720.0     1,568       5,504        503.4  void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, __nv_bfloat16, __nv_bfloat16, float, __nv…
      0.2       61,252,462     29,376      2,085.1      2,112.0       896       3,328        522.9  _combine_kernel                                                                                     
      0.2       60,376,010     28,917      2,087.9      2,049.0     1,888       5,952        216.5  kernel_cutlass_kernel_flashinfernormkernelsrmsnormRMSNormKernel_object_at__tensorptrbf16gmemalign12…
      0.2       56,038,934     10,557      5,308.2      5,344.0     4,704       6,784        219.6  _causal_conv1d_update_kernel                                                                        
      0.2       54,128,829     14,076      3,845.5      3,840.0     3,616       4,863         81.1  void tensorrt_llm::kernels::cutlass_kernels::doActivationKernel<__nv_fp8_e4m3, __nv_bfloat16, __nv_…
      0.2       49,309,425     14,076      3,503.1      3,488.0     3,296       3,969         88.6  void tensorrt_llm::kernels::cutlass_kernels::expandInputRowsKernel<__nv_fp8_e4m3, __nv_fp8_e4m3, (t…
      0.1       37,420,001     14,076      2,658.4      2,656.0     2,304       3,233        110.0  void sglang::route_quant_fused_kernel<(bool)1, float, __nv_bfloat16>(sglang::RouteQuantFusedParams) 
      0.1       33,274,101     14,076      2,363.9      2,368.0     2,208       2,784         53.7  void tensorrt_llm::kernels::cutlass_kernels::blockExpertPrefixSumKernel<(int)256>(const int *, int …
      0.1       32,261,405     14,076      2,291.9      2,272.0     2,176       2,752         58.5  void tensorrt_llm::kernels::cutlass_kernels::computeStridesTmaWarpSpecializedKernel<__nv_fp8_e4m3, …
      0.1       29,070,517     14,076      2,065.3      2,048.0     1,920       4,608         70.8  kernel_cutlass_kernel_flashinfernormkernelsrmsnormRMSNormKernel_object_at__tensorptrbf16gmemalign12…
      0.1       27,968,788     10,557      2,649.3      2,624.0     2,304       3,456         92.2  void gemmSN_TN_kernel<float, (int)128, (int)16, (int)2, (int)4, (int)8, (int)9, (bool)0, cublasGemv…
      0.1       26,748,027     14,076      1,900.3      1,888.0     1,792       2,208         39.8  void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::n…
      0.1       26,510,955     14,076      1,883.4      1,887.0     1,791       2,272         45.4  void tensorrt_llm::kernels::quantize_with_block_size<(tensorrt_llm::BlockScaleQuantizationType)2, _…
      0.1       25,641,069     13,770      1,862.1      1,856.0     1,408       5,728        409.2  void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::n…
      0.1       25,617,459     14,076      1,819.9      1,824.0     1,728       2,049         36.0  void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, float, float, float, float, (bool)0, floa…
      0.1       25,021,158     14,076      1,777.6      1,760.0     1,632       2,144         50.1  void sglang::situ_and_mul_kernel<float, __nv_bfloat16, (bool)1, (bool)1>(sglang::SituAndMulParams)  
      0.1       23,003,024     10,557      2,178.9      2,176.0     2,048       2,497         56.6  layer_norm_gated_fwd_kernel                                                                         
      0.1       22,467,866      3,672      6,118.7      6,112.0     5,824       6,592        113.9  void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void …
      0.1       20,225,634     14,076      1,436.9      1,408.0     1,248       2,080        109.1  void sglang::add3_kernel<(bool)1, (bool)1>(sglang::Add3Params)                                      
      0.1       19,553,652        306     63,900.8     63,922.5    61,858      79,522      1,506.0  void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_256x64_32x4_tn_align8>(T1::Para…
      0.1       18,358,363     14,076      1,304.2      1,280.0     1,184       1,760         66.8  tensorrt_llm::kernels::cutlass_kernels::mergeExpertPrefixSumKernel(const int *, const int *, const …
      0.1       17,067,934      3,825      4,462.2      3,456.0     3,104      27,457      4,532.6  void cutlass::Kernel2<cutlass_80_wmma_tensorop_s161616gemm_bf16_32x32_64x1_tn_align8>(T1::Params)   
      0.1       16,291,644     14,076      1,157.4      1,152.0     1,056       1,568         46.6  void tensorrt_llm::kernels::cutlass_kernels::globalExpertPrefixSumKernel<(int)256>(const int *, int…
      0.0       12,752,121     13,158        969.2        960.0       768       1,376         36.9  void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<c10::BFloat16>, …
      0.0       12,160,760      7,344      1,655.9      1,648.5     1,472       2,048         98.9  void at::native::<unnamed>::CatArrayBatchedCopy<at::native::<unnamed>::OpaqueType<(unsigned int)2>,…
      0.0       10,939,300      3,672      2,979.1      2,944.0     2,048       4,864        297.0  kernel_cutlass_kernel_flashinfernormkernelsrmsnormRMSNormKernel_object_at__tensorptrbf16gmemalign16…
      0.0        7,683,600      3,825      2,008.8      1,984.0     1,856       2,624        119.9  void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, float, __nv_bfloat16, float, __nv_bfloat1…
      0.0        5,207,022      3,672      1,418.0      1,408.0     1,344       1,760         42.9  void sglang::mla_output_gate_kernel<(int)256, (bool)1>(sglang::MlaOutputGateParams)                 
      0.0        5,069,771        459     11,045.3     15,041.0     2,816      16,864      5,774.6  create_flashinfer_kv_indices_triton                                                                 
      0.0        4,776,945      3,672      1,300.9      1,248.0     1,088       3,584        292.6  kernel_cutlass_kernel_flashinfernormkernelsrmsnormRMSNormKernel_object_at__tensorptrbf16gmemalign16…
      0.0        4,260,889      1,836      2,320.7      2,304.0     2,144       2,656         58.7  kernel_cutlass_kernel_flashinfernormkernelsfused_add_rmsnormFusedAddRMSNormKernel_object_at__tensor…
      0.0        3,022,079        153     19,752.2     19,744.0    18,944      21,505        331.4  void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x2_nn_align8>(T1::Para…
      0.0        2,772,279        918      3,019.9      3,040.0     2,784       3,137         51.1  void sglang::act_and_mul_kernel<__nv_bfloat16, (sglang::ActivationKind)0, (bool)1, (bool)0, (bool)0…
      0.0        2,164,385        153     14,146.3     14,144.0    13,217      17,088        446.0  _fused_mamba_state_scatter_with_mask_kernel                                                         
      0.0        1,458,665        153      9,533.8      9,536.0     9,312       9,792         87.5  void at::native::reduce_kernel<(int)512, (int)1, at::native::ReduceOp<c10::BFloat16, at::native::Ma…
      0.0        1,405,361        320      4,391.8      2,464.0     1,280      12,000      2,715.6  void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void …
      0.0        1,230,757        153      8,044.2      8,032.0     7,872       8,384         95.2  void at::native::reduce_kernel<(int)512, (int)1, at::native::ReduceOp<float, at::native::ArgMaxOps<…
      0.0        1,139,234        918      1,241.0      1,216.0     1,120       1,728         80.8  _table_qk_norm_rope_kernel                                                                          
      0.0        1,050,014        153      6,862.8      6,849.0     6,688       7,104         64.1  void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIterator…
      0.0        1,022,938        918      1,114.3      1,088.0     1,024       1,888        126.9  void sglang::store_kvcache<(long)256, (long)256, (int)1, (bool)1, long>(sglang::StoreKVCacheParams) 
      0.0          956,987        765      1,251.0      1,056.0       960       1,824        263.4  void at_cuda_detail::cub::detail::scan::DeviceScanKernel<at_cuda_detail::cub::detail::scan::policy_…
      0.0          949,915        918      1,034.8        944.5       832       1,664        224.3  set_kv_buffer_prefix_valid_tiled                                                                    
      0.0          919,532        893      1,029.7      1,088.0       736       1,697        178.4  void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIterator…
      0.0          754,770        322      2,344.0      2,336.0     1,312       3,009        308.7  void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void …
      0.0          699,890        483      1,449.0      1,440.0     1,184       2,176        126.0  void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIterator…
      0.0          623,732        592      1,053.6      1,056.0       832       1,472        129.7  void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std…
      0.0          603,445        153      3,944.1      3,904.0     3,520       5,089        249.4  _fused_conv_window_scatter_multi_kernel                                                             
      0.0          575,030        153      3,758.4      3,744.0     3,648       4,192         88.2  void at::native::reduce_kernel<(int)512, (int)1, at::native::ReduceOp<c10::BFloat16, at::native::Ar…
      0.0          535,246        153      3,498.3      3,488.0     3,360       3,648         43.9  void sglang::situ_and_mul_kernel<__nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1>(sglang::SituAndMul…
      0.0          472,175        765        617.2        544.0       480       1,024        121.2  void at_cuda_detail::cub::detail::scan::DeviceScanInitKernel<at_cuda_detail::cub::ScanTileState<lon…
      0.0          428,936        306      1,401.8      1,408.0     1,216       1,664         98.8  void at::native::<unnamed>::multi_tensor_apply_kernel<at::native::<unnamed>::TensorListMetadata<(in…
      0.0          378,666        153      2,474.9      2,432.0     2,272       3,072        143.2  void at::native::<unnamed>::CatArrayBatchedCopy_vectorized<at::native::<unnamed>::OpaqueType<(unsig…
      0.0          370,094        153      2,418.9      2,432.0     2,240       2,656         81.6  _dflash_accept_bonus_contig_kernel                                                                  
      0.0          348,362         80      4,354.5      4,832.0     1,344       5,728      1,448.7  alloc_extend_kernel                                                                                 
      0.0          332,681        313      1,062.9        928.0       800       2,240        239.5  _vocab_parallel_embedding_kernel                                                                    
      0.0          321,512        306      1,050.7      1,056.0       992       1,344         46.1  void at::native::vectorized_elementwise_kernel<(int)2, at::native::AUnaryFunctor<long, long, long, …
      0.0          261,832        153      1,711.3      1,696.0     1,600       2,048         75.2  _fused_norm_rope_kernel_stacked                                                                     
      0.0          248,327        153      1,623.1      1,600.0     1,536       1,856         63.0  void at::native::_scatter_gather_elementwise_kernel<(int)128, (int)8, void at::native::_cuda_scatte…
      0.0          226,628        160      1,416.4      1,408.0     1,152       1,664         90.9  _prepare_dflash_draft_block_contig_kernel                                                           
      0.0          216,363        306        707.1        704.0       640         928         37.1  void <unnamed>::elementwise_kernel_with_index<int, at::native::arange_cuda_out(const c10::Scalar &,…
      0.0          181,541        153      1,186.5      1,184.0     1,120       1,344         43.2  _fused_replay_state_indices_kernel                                                                  
      0.0          161,766         32      5,055.2      4,848.5     4,640       5,888        413.5  void at_cuda_detail::cub::detail::radix_sort::DeviceRadixSortOnesweepKernel<at_cuda_detail::cub::de…
      0.0          158,632        153      1,036.8      1,024.0       992       1,216         35.5  void at::native::vectorized_elementwise_kernel<(int)2, at::native::AUnaryFunctor<long, long, long, …
      0.0          153,253        153      1,001.7        992.0       929       1,216         38.8  void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctor_add<long>, std::arra…
      0.0          138,342        153        904.2        896.0       832       1,120         38.4  void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<c10::BFloat16, c10…
      0.0          120,260        153        786.0        768.0       736         992         37.8  void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<c…
      0.0          101,381         64      1,584.1      1,568.0     1,472       1,792         80.7  void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::n…
      0.0           81,764         64      1,277.6      1,152.0     1,088       2,400        385.8  void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void …
      0.0           31,745          8      3,968.1      3,936.5     3,712       4,256        190.5  void at_cuda_detail::cub::detail::select::DeviceSelectSweepKernel<at_cuda_detail::cub::detail::sele…
      0.0           23,169         16      1,448.1      1,424.0     1,280       1,632        118.3  assign_req_to_token_pool                                                                            
      0.0           21,730         16      1,358.1      1,360.0     1,152       1,696        166.6  get_last_loc_kernel                                                                                 
      0.0           18,240          8      2,280.0      2,272.0     2,240       2,368         37.3  void at_cuda_detail::cub::detail::radix_sort::DeviceRadixSortHistogramKernel<at_cuda_detail::cub::d…
      0.0           16,992          8      2,124.0      2,112.0     2,048       2,272         66.1  void at::native::vectorized_elementwise_kernel<(int)4, at::native::BUnaryFunctor<int, int, int, at:…
      0.0            9,440          8      1,180.0      1,168.0     1,120       1,248         43.4  void at::native::<unnamed>::CatArrayBatchedCopy_alignedK_contig<at::native::<unnamed>::OpaqueType<(…
      0.0            8,225          8      1,028.1      1,024.0       992       1,120         43.3  void at_cuda_detail::cub::detail::radix_sort::DeviceRadixSortExclusiveSumKernel<at_cuda_detail::cub…
      0.0            5,888          8        736.0        736.0       704         768         24.2  void at_cuda_detail::cub::detail::scan::DeviceCompactInitKernel<at_cuda_detail::cub::ScanTileState<…

Processing [/report/profiles/node_5/nsys/d0.sqlite] with [/opt/nvidia/nsight-systems-cli/2026.4.1/target-linux-x64/reports/nvtx_sum.py]... 

 ** NVTX Range Summary (nvtx_sum):

 Time (%)  Total Time (ns)  Instances    Avg (ns)     Med (ns)    Min (ns)    Max (ns)    StdDev (ns)    Style                   Range                
 --------  ---------------  ---------  ------------  -----------  ---------  -----------  ------------  --------  ------------------------------------
     66.4   14,731,201,668     78,873     186,771.2    269,479.0        484  106,046,379     779,117.6  StartEnd  NCCL:AllReduce                      
     13.0    2,890,729,983        160  18,067,062.4  5,237,441.5  4,817,856  304,143,864  54,432,245.2  PushPop   :dflash_target_verify               
     10.0    2,207,342,791      1,440   1,532,876.9     88,978.5        759   10,030,305   2,577,710.2  StartEnd  NCCL:AllGather                      
      6.7    1,481,928,797        160   9,262,055.0  2,657,347.0  2,085,029  192,311,281  27,973,693.1  PushPop   :dflash_draft_prepare_forward_sample
      1.0      231,076,159        160   1,444,226.0  1,231,009.0    908,153    3,671,689     490,453.7  PushPop   :dflash_materialize_draft_kv        
      0.9      206,649,966     32,480       6,362.4      5,120.0      1,418    1,267,086      19,524.2  StartEnd  NCCL:GroupRuntime                   
      0.8      173,590,142         16  10,849,383.9     85,687.5     51,325  172,281,867  43,048,669.5  StartEnd  NCCL:CommInit Collection            
      0.4       83,715,348        160     523,220.9    456,385.0    324,574    1,188,730     174,248.6  PushPop   :dflash_replayssm_commit            
      0.3       68,016,805        160     425,105.0    369,313.0    298,881    1,001,480     136,790.6  PushPop   :dflash_accept                      
      0.2       35,803,080        800      44,753.8     39,169.5     23,489      166,215      18,141.6  PushPop   CCCL:cub::DeviceScan::InclusiveScan 
      0.1       23,106,075        160     144,413.0    115,591.5     89,611      401,881      68,365.7  StartEnd  NCCL:API Group                      
      0.1       19,587,358        160     122,421.0     99,108.5     72,056      351,450      57,966.4  StartEnd  NCCL:GroupLaunch                    
      0.1       18,657,059          8   2,332,132.4  2,351,962.0  2,105,165    2,502,099     134,544.4  PushPop   CCCL:cub::DeviceRadixSort           
      0.0        5,960,181        160      37,251.1     27,998.5     19,843      123,127      20,804.4  StartEnd  NCCL:KernelLaunch                   
      0.0          530,446          8      66,305.8     62,777.0     41,062      119,196      24,108.4  PushPop   CCCL:cub::DeviceSelect::Unique      

Processing [/report/profiles/node_5/nsys/d0.sqlite] with [/opt/nvidia/nsight-systems-cli/2026.4.1/target-linux-x64/reports/nvtx_gpu_proj_sum.py]... 

 ** NVTX GPU Projection Summary (nvtx_gpu_proj_sum):

                Range                   Style    Total Proj Time (ns)  Total Range Time (ns)  Range Instances  Proj Avg (ns)  Proj Med (ns)  Proj Min (ns)  Proj Max (ns)  Proj StdDev (ns)  Total GPU Ops  Avg GPU Ops  Avg Range Lvl  Avg Num Child
 ------------------------------------  --------  --------------------  ---------------------  ---------------  -------------  -------------  -------------  -------------  ----------------  -------------  -----------  -------------  -------------
 :dflash_target_verify                 PushPop         28,286,105,196          2,852,637,482              153  184,876,504.5  183,519,609.0    180,949,744    192,032,157       3,296,693.3            459          3.0            0.0            0.0
 :dflash_draft_prepare_forward_sample  PushPop          1,310,186,765          1,481,928,797              160    8,188,667.3    6,112,744.5          4,960    240,344,638      18,843,179.5          3,992         24.9            0.0            0.0
 NCCL:API Group                        StartEnd           128,169,619             22,367,893              153      837,709.9      328,905.0        216,455     12,461,880       2,182,886.6            153          1.0            0.0            1.0
 NCCL:GroupLaunch                      StartEnd           128,169,619             18,945,082              153      837,709.9      328,905.0        216,455     12,461,880       2,182,886.6            153          1.0            1.0            1.0
 NCCL:KernelLaunch                     StartEnd           128,169,619              5,795,224              153      837,709.9      328,905.0        216,455     12,461,880       2,182,886.6            153          1.0            2.0            0.0
 :dflash_materialize_draft_kv          PushPop             87,608,199            222,205,495              153      572,602.6      572,305.0        567,124        588,626           2,418.4          1,836         12.0            0.0            0.0
 :dflash_replayssm_commit              PushPop              4,910,863             80,548,047              153       32,097.1       32,449.0         28,641         35,681             992.8            765          5.0            0.0            0.0
 CCCL:cub::DeviceScan::InclusiveScan   PushPop              2,630,000             34,473,275              765        3,437.9        3,104.0          2,976         79,875           3,163.9          1,530          2.0            0.0            0.0
 :dflash_accept                        PushPop              2,463,661             65,354,814              153       16,102.4       15,872.0         13,889         24,289           1,682.7            459          3.0            0.0            0.0
 CCCL:cub::DeviceRadixSort             PushPop                521,809             18,657,059                8       65,226.1       65,970.0         56,674         68,130           3,628.4             96         12.0            0.0            0.0
 CCCL:cub::DeviceSelect::Unique        PushPop                 48,002                530,446                8        6,000.3        5,952.5          5,760          6,304             191.1             16          2.0            0.0            0.0

Processing [/report/profiles/node_5/nsys/d0.sqlite] with [/opt/nvidia/nsight-systems-cli/2026.4.1/target-linux-x64/reports/cuda_api_sum.py]... 

 ** CUDA API Summary (cuda_api_sum):

 Time (%)  Total Time (ns)  Num Calls    Avg (ns)       Med (ns)     Min (ns)   Max (ns)    StdDev (ns)                  Name                
 --------  ---------------  ---------  -------------  -------------  --------  -----------  ------------  -----------------------------------
     82.9   24,339,915,697        159  153,081,230.8  168,293,622.0     9,971  233,933,563  50,404,828.9  cudaEventSynchronize               
      9.8    2,886,748,801        320    9,021,090.0    2,969,324.5   232,892  303,625,849  39,358,940.3  cudaGraphLaunch                    
      5.1    1,500,800,764        424    3,539,624.4        9,488.0     4,929  187,136,136  25,465,360.8  cudaStreamSynchronize              
      1.0      290,912,342        932      312,137.7        1,066.0       337  145,452,863   6,694,242.3  cudaThreadExchangeStreamCaptureMode
      0.4      124,266,257      5,816       21,366.3       13,003.0     5,665    2,165,877      73,994.7  cudaLaunchKernel                   
      0.3       82,200,378      2,944       27,921.3       20,489.5     7,697    2,053,090      49,464.5  cudaMemcpyAsync                    
      0.1       42,849,315      2,832       15,130.4       11,695.5     6,114      103,714      10,281.8  cuLaunchKernelEx                   
      0.1       28,395,935      6,804        4,173.4        2,674.5       750       76,514       4,809.7  cudaEventQuery                     
      0.0        9,264,477      1,646        5,628.5        4,954.0       846       53,181       4,468.1  cudaEventRecordWithFlags           
      0.0        7,849,046        480       16,352.2       14,336.0     7,190       72,283       8,410.7  cudaLaunchKernelExC                
      0.0        5,597,154        208       26,909.4       25,857.0     3,981       84,131      14,636.8  cudaMemsetAsync                    
      0.0        5,527,685          6      921,280.8      892,572.0   841,781    1,074,944      91,114.7  cudaSetDevice                      
      0.0        5,310,946        968        5,486.5        5,159.0     1,144       39,237       3,450.0  cudaStreamWaitEvent                
      0.0        5,277,551      3,702        1,425.6        1,085.5       362       42,406       1,933.2  cudaStreamIsCapturing              
      0.0        4,780,979      5,976          800.0          480.5       157       35,068       1,321.9  cuKernelGetName                    
      0.0        4,716,514        320       14,739.1       12,800.0     7,430       50,752       6,496.7  cuLaunchKernel                     
      0.0        3,089,232        648        4,767.3        4,055.0     1,976       28,911       2,562.9  cudaEventCreateWithFlags           
      0.0        2,750,711        656        4,193.2        2,432.5     1,088       57,955       4,512.4  cudaEventRecord                    
      0.0        1,829,159        647        2,827.1        2,506.0     1,571       15,604       1,128.0  cudaEventDestroy                   
      0.0        1,067,741        160        6,673.4        4,764.5     3,627       41,102       5,688.6  cudaGetFuncBySymbol                
      0.0        1,007,825        320        3,149.5        2,469.5     1,115       25,034       3,341.4  cuKernelGetFunction                
      0.0          607,949        320        1,899.8        1,874.5       356       34,920       2,383.3  cudaStreamGetCaptureInfo           
      0.0          177,089         24        7,378.7        7,460.5       880       30,146       6,206.9  cuLibraryGetKernel                 
      0.0           90,950          1       90,950.0       90,950.0    90,950       90,950           0.0  cuProfilerStart                    
      0.0           85,778          6       14,296.3       13,874.0    12,396       16,997       1,630.7  cuDevicePrimaryCtxGetState         
      0.0            5,827          6          971.2          933.5       614        1,420         287.6  cudaGetLastError                   

