diff --git a/CURRENT_STATE.md b/CURRENT_STATE.md index d91b1fe..77c9678 100644 --- a/CURRENT_STATE.md +++ b/CURRENT_STATE.md @@ -1,6 +1,6 @@ # H3 Runtime Current State -Status date: 2026-08-24 +Status date: 2026-08-25 This document is the canonical snapshot of implemented scope and remaining work. Historical handoffs in `PLAN.md` and `PARITY.md` may describe older states. @@ -160,6 +160,18 @@ video and audio tensors. Spark enables the path with `SAGE2_BLACKWELL_DESIGN.md`, and `benchmarks/gb10-post-optimization-profile-summary.json`. +The isolated FC2 cuBLASLt scheduling study is complete. The production +heuristic's `_stream_k` kernel requests the same `25.664 GB` of operands as the +retained public split-K-1 schedule, but its L2 hit rate is only `53.32%` versus +`91.10%`; it incurs `9.853 GB` more L2 read misses and spends heavily in +synchronization polling. Algorithm 70, tile 20, stages 37, split-K 1 is +byte-exact with zero workspace. It improves complete blocks 0, 24, and 49 by +`8.16-8.88%`, the two-step trajectory by `7.50%`, and the canonical 12-step +trajectory from `278.201 s` to `255.371 s` (`8.21%`) with exact video and audio +latents. This remains a research-retained integration candidate: production +dispatch and configuration are unchanged. See +`research/fc2_nvfp4_scheduling/RESULTS.md`. + The Spark hot runtime was rebuilt and recreated with image `sha256:5f879c43374bcedb89745971d7d95d82afc8fcf9c41d30f11b257c2863b9fe28`. Health and startup warmup pass with modulation fusion enabled. A resident real diff --git a/PERFORMANCE_ROADMAP.md b/PERFORMANCE_ROADMAP.md index 9390832..f29de8c 100644 --- a/PERFORMANCE_ROADMAP.md +++ b/PERFORMANCE_ROADMAP.md @@ -200,7 +200,8 @@ another's result. ### Phase 3: NVFP4 Fused Projection Prototype -Status: active next component on GB10/SM121. +Status: active on GB10/SM121; isolated FC2 library scheduling is complete and +awaits production integration. The first profile also exposed and fixed a native activation-packer layout bug: the previous width-specific block-scale swizzle failed at the attention output's @@ -383,6 +384,24 @@ off-chip traffic bottleneck, the NVFP4 GEMMs. Do not select a new kernel from the old profile. See `benchmarks/gb10-fully-fused-fresh-nsight-summary.json`. +The FC2 follow-up resolves that projection's `16.1x` off-chip amplification. +The production heuristic launches an undocumented-sentinel `_stream_k` kernel; +the retained documented configuration is cuBLASLt algorithm 70, tile 20, +stages 37, public split-K 1, reduction scheme 0, and zero workspace. Both +schedules request the same `25.664 GB` of operands, but split-K 1 raises L2 hit +rate from `53.32%` to `91.10%`, removes `9.853 GB` of L2 read misses, and raises +tensor-pipe activity from `25.01%` to `82.42%`. There is no material global +partial-accumulator or output-reduction traffic; the baseline instead spends +heavily in Stream-K synchronization polling and loses traversal locality. + +The candidate is byte-exact and improves FC2 p50 from `52.521 ms` to +`15.636 ms`. Complete blocks 0, 24, and 49 improve by `8.16-8.88%`; two-step +and canonical 12-step trajectories improve by `7.50%` and `8.21%`, with exact +video and audio latents. The custom persistent-kernel branch is therefore +closed. Production integration remains separate work, so current dispatch is +unchanged. See `research/fc2_nvfp4_scheduling/RESULTS.md` and +`benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json`. + The follow-up real block-24 Sage2 decomposition now selects the next exact kernel experiment. Manual preparation plus the existing prequantized mainloop is byte-exact against public SageAttention 2.2.0. Uninstrumented median phase diff --git a/benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv b/benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv new file mode 100644 index 0000000..1b17848 --- /dev/null +++ b/benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv @@ -0,0 +1,4 @@ +"ID","Process ID","Process Name","Host Name","Kernel Name","Context","Stream","Block Size","Grid Size","Device","CC","c2clink__enabled_mask","c2clink__present","derived__avg_thread_executed","derived__avg_thread_executed_true","derived__avg_thread_unexecuted_true","derived__derivative_avg_thread_executed_true","derived__l1tex__lsu_writeback_bytes_mem_lgds.sum.peak_sustained","derived__l1tex__lsu_writeback_bytes_mem_lgds.sum.per_second","derived__local_spilling_requests","derived__local_spilling_requests_pct","derived__lts__lts2xbar_bytes.sum.peak_sustained","derived__lts__lts2xbar_bytes.sum.per_second","derived__memory_l1_conflicts_shared_nway","derived__memory_l1_wavefronts_shared_excessive","derived__memory_l2_theoretical_sectors_global_excessive","derived__pct_occupancy_per_barrier_count","derived__pct_occupancy_per_block_size","derived__pct_occupancy_per_register_count","derived__pct_occupancy_per_shared_mem_size","derived__sm__sass_thread_inst_executed_op_dfma_pred_on_x2","derived__sm__sass_thread_inst_executed_op_ffma_pred_on_x2","derived__sm__sass_thread_inst_executed_op_hfma_pred_on_x4","derived__smsp__inst_executed_op_branch_pct","derived__smsp__sass_thread_inst_executed_op_dfma_pred_on_x2","derived__smsp__sass_thread_inst_executed_op_ffma_pred_on_x2","derived__smsp__sass_thread_inst_executed_op_hadd_pred_on_x2","derived__smsp__sass_thread_inst_executed_op_hfma_pred_on_x4","derived__smsp__sass_thread_inst_executed_op_hmul_pred_on_x2","derived_tempMetric0","derived_tempMetric1","derived_tempMetric2","derived_tempMetric3","derived_tempMetric4","derived_tempMetric5","device__attribute_architecture","device__attribute_async_engine_count","device__attribute_can_flush_remote_writes","device__attribute_can_map_host_memory","device__attribute_can_tex2d_gather","device__attribute_can_use_64_bit_stream_mem_ops","device__attribute_can_use_64_bit_stream_mem_ops_v1","device__attribute_can_use_host_pointer_for_registered_mem","device__attribute_can_use_stream_mem_ops_v1","device__attribute_can_use_stream_wait_value_nor","device__attribute_can_use_stream_wait_value_nor_v1","device__attribute_chip","device__attribute_clock_rate","device__attribute_cluster_launch","device__attribute_compute_capability_major","device__attribute_compute_capability_minor","device__attribute_compute_mode","device__attribute_compute_preemption_supported","device__attribute_concurrent_kernels","device__attribute_concurrent_managed_access","device__attribute_confidential_computing_mode","device__attribute_cooperative_launch","device__attribute_cooperative_multi_device_launch","device__attribute_deferred_mapping_cuda_array_supported","device__attribute_device_index","device__attribute_direct_managed_mem_access_from_host","device__attribute_display_name","device__attribute_dma_buf_supported","device__attribute_ecc_enabled","device__attribute_fb_bus_width","device__attribute_fbp_count","device__attribute_generic_compression_supported","device__attribute_global_l1_cache_supported","device__attribute_global_memory_bus_width","device__attribute_gpu_direct_rdma_flush_writes_options","device__attribute_gpu_direct_rdma_supported","device__attribute_gpu_direct_rdma_with_cuda_vmm_supported","device__attribute_gpu_direct_rdma_writes_ordering","device__attribute_gpu_overlap","device__attribute_gpu_pci_device_id","device__attribute_gpu_pci_ext_device_id","device__attribute_gpu_pci_ext_downstream_link_rate","device__attribute_gpu_pci_ext_downstream_link_width","device__attribute_gpu_pci_ext_gen","device__attribute_gpu_pci_ext_gpu_gen","device__attribute_gpu_pci_ext_gpu_link_rate","device__attribute_gpu_pci_ext_gpu_link_width","device__attribute_gpu_pci_revision_id","device__attribute_gpu_pci_sub_system_id","device__attribute_handle_type_fabric_supported","device__attribute_handle_type_posix_file_descriptor_supported","device__attribute_handle_type_win32_handle_supported","device__attribute_handle_type_win32_kmt_handle_supported","device__attribute_host_native_atomic_supported","device__attribute_host_numa_id","device__attribute_host_register_supported","device__attribute_implementation","device__attribute_integrated","device__attribute_ipc_event_supported","device__attribute_kernel_exec_timeout","device__attribute_l2_cache_size","device__attribute_l2s_count","device__attribute_limits_max_cta_per_sm","device__attribute_limits_num_tpcs","device__attribute_local_l1_cache_supported","device__attribute_managed_memory","device__attribute_max_access_policy_window_size","device__attribute_max_block_dim_x","device__attribute_max_block_dim_y","device__attribute_max_block_dim_z","device__attribute_max_blocks_per_multiprocessor","device__attribute_max_gpu_frequency_khz","device__attribute_max_grid_dim_x","device__attribute_max_grid_dim_y","device__attribute_max_grid_dim_z","device__attribute_max_ipc_per_multiprocessor","device__attribute_max_ipc_per_scheduler","device__attribute_max_mem_frequency_khz","device__attribute_max_persisting_l2_cache_size","device__attribute_max_pitch","device__attribute_max_registers_per_block","device__attribute_max_registers_per_multiprocessor","device__attribute_max_registers_per_thread","device__attribute_max_shared_memory_per_block","device__attribute_max_shared_memory_per_block_optin","device__attribute_max_shared_memory_per_multiprocessor","device__attribute_max_threads_per_block","device__attribute_max_threads_per_multiprocessor","device__attribute_max_warps_per_multiprocessor","device__attribute_max_warps_per_scheduler","device__attribute_maximum_surface1d_layered_layers","device__attribute_maximum_surface1d_layered_width","device__attribute_maximum_surface1d_width","device__attribute_maximum_surface2d_height","device__attribute_maximum_surface2d_layered_height","device__attribute_maximum_surface2d_layered_layers","device__attribute_maximum_surface2d_layered_width","device__attribute_maximum_surface2d_width","device__attribute_maximum_surface3d_depth","device__attribute_maximum_surface3d_height","device__attribute_maximum_surface3d_width","device__attribute_maximum_surfacecubemap_layered_layers","device__attribute_maximum_surfacecubemap_layered_width","device__attribute_maximum_surfacecubemap_width","device__attribute_maximum_texture1d_layered_layers","device__attribute_maximum_texture1d_layered_width","device__attribute_maximum_texture1d_linear_width","device__attribute_maximum_texture1d_mipmapped_width","device__attribute_maximum_texture1d_width","device__attribute_maximum_texture2d_gather_height","device__attribute_maximum_texture2d_gather_width","device__attribute_maximum_texture2d_height","device__attribute_maximum_texture2d_layered_height","device__attribute_maximum_texture2d_layered_layers","device__attribute_maximum_texture2d_layered_width","device__attribute_maximum_texture2d_linear_height","device__attribute_maximum_texture2d_linear_pitch","device__attribute_maximum_texture2d_linear_width","device__attribute_maximum_texture2d_mipmapped_height","device__attribute_maximum_texture2d_mipmapped_width","device__attribute_maximum_texture2d_width","device__attribute_maximum_texture3d_depth","device__attribute_maximum_texture3d_depth_alternate","device__attribute_maximum_texture3d_height","device__attribute_maximum_texture3d_height_alternate","device__attribute_maximum_texture3d_width","device__attribute_maximum_texture3d_width_alternate","device__attribute_maximum_texturecubemap_layered_layers","device__attribute_maximum_texturecubemap_layered_width","device__attribute_maximum_texturecubemap_width","device__attribute_mem_sync_domain_count","device__attribute_memory_clock_rate","device__attribute_memory_pools_supported","device__attribute_mempool_supported_handle_types","device__attribute_mps_enabled","device__attribute_multi_gpu_board","device__attribute_multi_gpu_board_group_id","device__attribute_multicast_supported","device__attribute_multiprocessor_count","device__attribute_num_l2s_per_fbp","device__attribute_num_schedulers_per_multiprocessor","device__attribute_num_tex_per_multiprocessor","device__attribute_numa_config","device__attribute_pageable_memory_access","device__attribute_pageable_memory_access_uses_host_page_tables","device__attribute_pci_bus_id","device__attribute_pci_device_id","device__attribute_pci_domain_id","device__attribute_ram_location","device__attribute_ram_type","device__attribute_reserved_shared_memory_per_block","device__attribute_sass_level","device__attribute_single_to_double_precision_perf_ratio","device__attribute_sparse_cuda_array_supported","device__attribute_stream_priorities_supported","device__attribute_surface_alignment","device__attribute_tcc_driver","device__attribute_tensor_map_access_supported","device__attribute_texture_alignment","device__attribute_texture_pitch_alignment","device__attribute_total_constant_memory","device__attribute_total_memory","device__attribute_unified_addressing","device__attribute_unified_function_pointers","device__attribute_virtual_address_management_supported","device__attribute_warp_size","gcc__cache_requests_type_constant.sum","gcc__cache_requests_type_constant.sum.pct_of_peak_sustained_elapsed","gcc__cache_requests_type_instruction.sum","gcc__cache_requests_type_instruction.sum.pct_of_peak_sustained_elapsed","gcc__xbar2gcc_sectors.sum","gcc__xbar2gcc_sectors.sum.pct_of_peak_sustained_elapsed","gpc__cycles_elapsed.avg","gpc__cycles_elapsed.avg.per_second","gpc__cycles_elapsed.max","gpc__cycles_elapsed.max.per_second","gpc__cycles_elapsed.min","gpc__cycles_elapsed.min.per_second","gpc__cycles_elapsed.sum","gpc__cycles_elapsed.sum.per_second","gpu__compute_memory_access_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.sum.pct_of_peak_sustained_elapsed","gpu__time_duration.avg","gpu__time_duration.max","gpu__time_duration.min","gpu__time_duration.sum","gr__workids_granted.avg","gr__workids_granted.max","gr__workids_granted.min","gr__workids_granted.sum","gr__workids_granted_as_ctas.avg","gr__workids_granted_as_ctas.max","gr__workids_granted_as_ctas.min","gr__workids_granted_as_ctas.sum","gr__workids_requested.avg","gr__workids_requested.max","gr__workids_requested.min","gr__workids_requested.sum","idc__request_cycles_active.avg.pct_of_peak_sustained_elapsed","idc__request_cycles_active.max.pct_of_peak_sustained_elapsed","idc__request_cycles_active.min.pct_of_peak_sustained_elapsed","idc__request_cycles_active.sum.pct_of_peak_sustained_elapsed","idc__request_hit_rate.pct","idc__requests.sum","idc__requests.sum.pct_of_peak_sustained_elapsed","inst_executed","l1tex__cycles_active.avg","l1tex__cycles_active.max","l1tex__cycles_active.min","l1tex__cycles_active.sum","l1tex__cycles_elapsed.avg","l1tex__cycles_elapsed.avg.per_second","l1tex__cycles_elapsed.max","l1tex__cycles_elapsed.max.per_second","l1tex__cycles_elapsed.min","l1tex__cycles_elapsed.min.per_second","l1tex__cycles_elapsed.sum","l1tex__cycles_elapsed.sum.per_second","l1tex__data_bank_conflicts_pipe_lsu_mem_shared.sum","l1tex__data_bank_conflicts_pipe_lsu_mem_shared_op_atom.sum","l1tex__data_bank_conflicts_pipe_lsu_mem_shared_op_ld.sum","l1tex__data_bank_conflicts_pipe_lsu_mem_shared_op_ldgsts.sum","l1tex__data_bank_conflicts_pipe_lsu_mem_shared_op_st.sum","l1tex__data_bank_reads.avg.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.max.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.min.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.sum.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.avg.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.max.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.min.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts_mem_shared.sum","l1tex__data_pipe_lsu_wavefronts_mem_shared.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_atom.sum","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_ld.sum","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_st.sum","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.avg.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.max.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.min.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.sum.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active_mem_lgds.sum","l1tex__lsu_writeback_active_mem_lgds.sum.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active_mem_lgds.sum.peak_sustained","l1tex__lsu_writeback_active_mem_lgds.sum.per_second","l1tex__lsuin_requests.avg.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.max.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.min.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.avg.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.max.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.min.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes.sum","l1tex__m_l1tex2xbar_write_bytes.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes.sum.per_second","l1tex__m_l1tex2xbar_write_bytes_mem_dshared.sum","l1tex__m_l1tex2xbar_write_bytes_mem_dshared.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes_mem_dshared.sum.per_second","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_red.sum","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_red.sum.per_second","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_st.sum","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_st.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_atom.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_atom.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_ld.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_ld.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_redas.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_redas.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_redas.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_st.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_tma_red.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_tma_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_tma_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_tma_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_atom.sum","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_red.sum","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_red.sum","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_red.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_st.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_lg_op_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_lg_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_atom.sum","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_red.sum","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_bytes.sum","l1tex__m_xbar2l1tex_read_bytes.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_bytes.sum.per_second","l1tex__m_xbar2l1tex_read_bytes_mem_dshared.sum","l1tex__m_xbar2l1tex_read_bytes_mem_dshared.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_bytes_mem_dshared.sum.per_second","l1tex__m_xbar2l1tex_read_bytes_mem_global_op_tma_ld.sum","l1tex__m_xbar2l1tex_read_bytes_mem_global_op_tma_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_bytes_mem_global_op_tma_ld.sum.per_second","l1tex__m_xbar2l1tex_read_sectors.avg.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.max.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.min.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_atom.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_atom.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_ld.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_ld.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_redas.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_redas.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_redas.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_st.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_st.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_tma_red.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_tma_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_tma_st.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_tma_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_atom.sum","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_tma_ld.sum","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_tma_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_tma_ld.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_lg_op_ld.sum","l1tex__m_xbar2l1tex_read_sectors_mem_lg_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_surface_op_atom.sum","l1tex__m_xbar2l1tex_read_sectors_mem_surface_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_surface_op_ld.sum","l1tex__m_xbar2l1tex_read_sectors_mem_surface_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_texture.sum","l1tex__m_xbar2l1tex_read_sectors_mem_texture.sum.pct_of_peak_sustained_elapsed","l1tex__t_bytes_pipe_lsu_mem_global_op_ldgsts_cache_access.sum","l1tex__t_bytes_pipe_lsu_mem_global_op_ldgsts_cache_access.sum.pct_of_peak_sustained_elapsed","l1tex__t_bytes_pipe_lsu_mem_global_op_ldgsts_cache_access.sum.per_second","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_atom.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_ld.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_redas.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_redas.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_st.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_atom.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_ld.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_red.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_st.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_local_op_ld.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_local_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_local_op_st.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_local_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_atom.sum","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_ld.sum","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_red.sum","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_st.sum","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_texture.sum","l1tex__t_output_wavefronts_pipe_tex_mem_texture.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_dshared_op_atom.sum","l1tex__t_requests_pipe_lsu_mem_dshared_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_dshared_op_ld.sum","l1tex__t_requests_pipe_lsu_mem_dshared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_dshared_op_redas.sum","l1tex__t_requests_pipe_lsu_mem_dshared_op_redas.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_dshared_op_st.sum","l1tex__t_requests_pipe_lsu_mem_dshared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_global_op_atom.sum","l1tex__t_requests_pipe_lsu_mem_global_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_global_op_ld.sum","l1tex__t_requests_pipe_lsu_mem_global_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_global_op_red.sum","l1tex__t_requests_pipe_lsu_mem_global_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_global_op_st.sum","l1tex__t_requests_pipe_lsu_mem_global_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_local_op_ld.sum","l1tex__t_requests_pipe_lsu_mem_local_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_local_op_st.sum","l1tex__t_requests_pipe_lsu_mem_local_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_surface_op_atom.sum","l1tex__t_requests_pipe_tex_mem_surface_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_surface_op_ld.sum","l1tex__t_requests_pipe_tex_mem_surface_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_surface_op_red.sum","l1tex__t_requests_pipe_tex_mem_surface_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_surface_op_st.sum","l1tex__t_requests_pipe_tex_mem_surface_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_texture.sum","l1tex__t_requests_pipe_tex_mem_texture.sum.pct_of_peak_sustained_elapsed","l1tex__t_sector_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_global_op_atom_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_global_op_ld_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_global_op_red_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_global_op_st_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_local_op_ld_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_local_op_st_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_surface_op_atom_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_surface_op_ld_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_surface_op_red_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_surface_op_st_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_texture_op_tex_hit_rate.pct","l1tex__t_sectors_pipe_lsu_mem_dshared_op_atom.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_atom_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_ld.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_ld_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_redas.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_redas_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_st.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_st_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_atom.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_atom_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_atom_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_ld.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_ld_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_ld_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_red.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_red_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_red_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_st.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_st_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_st_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_ld.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_ld_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_ld_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_st.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_st_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_st_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_atom.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_atom_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_atom_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_ld.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_ld_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_ld_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_red.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_red_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_red_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_st.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_st_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_st_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_texture.sum","l1tex__t_sectors_pipe_tex_mem_texture_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_texture_lookup_miss.sum","l1tex__tex_writeback_active.avg.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.max.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.min.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.sum","l1tex__tex_writeback_active.sum.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.avg.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.max.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.min.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.sum.pct_of_peak_sustained_elapsed","l1tex__throughput.avg.pct_of_peak_sustained_active","l1tex__throughput.avg.pct_of_peak_sustained_elapsed","l1tex__throughput.max.pct_of_peak_sustained_active","l1tex__throughput.max.pct_of_peak_sustained_elapsed","l1tex__throughput.min.pct_of_peak_sustained_active","l1tex__throughput.min.pct_of_peak_sustained_elapsed","l1tex__throughput.sum.pct_of_peak_sustained_active","l1tex__throughput.sum.pct_of_peak_sustained_elapsed","launch__barrier_count","launch__block_dim_x","launch__block_dim_y","launch__block_dim_z","launch__block_size","launch__cluster_dim_x","launch__cluster_dim_y","launch__cluster_dim_z","launch__cluster_max_active","launch__cluster_max_potential_size","launch__cluster_scheduling_policy","launch__cluster_size","launch__context_id","launch__device_id","launch__func_cache_config","launch__function_pcs","launch__grid_dim_x","launch__grid_dim_y","launch__grid_dim_z","launch__grid_size","launch__kernel_name","launch__occupancy_cluster_gpu_pct","launch__occupancy_cluster_pct","launch__occupancy_limit_barriers","launch__occupancy_limit_blocks","launch__occupancy_limit_registers","launch__occupancy_limit_shared_mem","launch__occupancy_limit_warps","launch__occupancy_per_barrier_count","launch__occupancy_per_block_size","launch__occupancy_per_cluster_size","launch__occupancy_per_register_count","launch__occupancy_per_shared_mem_size","launch__persisting_l2_cache_size","launch__preferred_cluster_dim_x","launch__preferred_cluster_dim_y","launch__preferred_cluster_dim_z","launch__preferred_cluster_size","launch__registers_per_thread","launch__registers_per_thread_allocated","launch__shared_mem_config_size","launch__shared_mem_per_block","launch__shared_mem_per_block_allocated","launch__shared_mem_per_block_driver","launch__shared_mem_per_block_dynamic","launch__shared_mem_per_block_static","launch__sm_count","launch__stack_size","launch__stream_id","launch__thread_count","launch__tpc_count","launch__tpc_enabled","launch__uses_cdp","launch__uses_green_context","launch__uses_mps","launch__uses_nvlink_centric_scheduling","launch__uses_vgpu","launch__waves_per_multiprocessor","lrc__average_ilc_input_sector_success_rate.pct","lrc__ilc_input_sectors.sum","lts__average_gcomp_input_sector_success_rate.pct","lts__average_gcomp_output_sector_compression_achieved_rate.ratio","lts__average_t_sectors_requested_srcunit_tex_op_read_vs_returned_sectors_realtime.ratio","lts__average_xcomp_gxc_sector_input_compression_rate.ratio","lts__cycles_active.avg","lts__cycles_active.max","lts__cycles_active.min","lts__cycles_active.sum","lts__cycles_elapsed.avg","lts__cycles_elapsed.avg.per_second","lts__cycles_elapsed.max","lts__cycles_elapsed.max.per_second","lts__cycles_elapsed.min","lts__cycles_elapsed.min.per_second","lts__cycles_elapsed.sum","lts__cycles_elapsed.sum.per_second","lts__d_atomic_input_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.max.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.min.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.sum.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.avg.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.max.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.min.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.sum.pct_of_peak_sustained_elapsed","lts__d_sectors.avg.pct_of_peak_sustained_elapsed","lts__d_sectors.max.pct_of_peak_sustained_elapsed","lts__d_sectors.min.pct_of_peak_sustained_elapsed","lts__d_sectors.sum.pct_of_peak_sustained_elapsed","lts__d_sectors_fill_sysmem.sum","lts__d_sectors_fill_sysmem.sum.pct_of_peak_sustained_elapsed","lts__d_sectors_fill_sysmem.sum.per_second","lts__gcomp_input_sectors.avg","lts__gcomp_input_sectors.max","lts__gcomp_input_sectors.min","lts__gcomp_input_sectors.sum","lts__lts2xbar_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.avg.peak_sustained","lts__lts2xbar_cycles_active.avg.per_second","lts__lts2xbar_cycles_active.max.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.max.peak_sustained","lts__lts2xbar_cycles_active.max.per_second","lts__lts2xbar_cycles_active.min.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.min.peak_sustained","lts__lts2xbar_cycles_active.min.per_second","lts__lts2xbar_cycles_active.sum.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.sum.peak_sustained","lts__lts2xbar_cycles_active.sum.per_second","lts__t_requests.sum","lts__t_requests_srcunit_gcc.sum","lts__t_requests_srcunit_tex.sum","lts__t_requests_srcunit_tex_op_atom_dot_alu.sum","lts__t_requests_srcunit_tex_op_atom_dot_cas.sum","lts__t_requests_srcunit_tex_op_read.sum","lts__t_requests_srcunit_tex_op_red.sum","lts__t_requests_srcunit_tex_op_write.sum","lts__t_sector_hit_rate.pct","lts__t_sector_op_read_hit_rate.pct","lts__t_sector_op_write_hit_rate.pct","lts__t_sectors.avg","lts__t_sectors.avg.pct_of_peak_sustained_elapsed","lts__t_sectors.avg.peak_sustained","lts__t_sectors.avg.per_cycle_elapsed","lts__t_sectors.max","lts__t_sectors.max.pct_of_peak_sustained_elapsed","lts__t_sectors.min","lts__t_sectors.min.pct_of_peak_sustained_elapsed","lts__t_sectors.sum","lts__t_sectors.sum.pct_of_peak_sustained_elapsed","lts__t_sectors.sum.per_second","lts__t_sectors_aperture_sysmem_op_write.sum","lts__t_sectors_aperture_sysmem_op_write.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_aperture_sysmem_op_write.sum.per_second","lts__t_sectors_data_ecc.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_data_ecc.avg.peak_sustained","lts__t_sectors_data_ecc.avg.per_cycle_elapsed","lts__t_sectors_data_ecc.sum","lts__t_sectors_data_ecc.sum.per_second","lts__t_sectors_evict_first_lookup_hit.sum","lts__t_sectors_evict_first_lookup_miss.sum","lts__t_sectors_evict_last_lookup_hit.sum","lts__t_sectors_evict_last_lookup_miss.sum","lts__t_sectors_evict_normal_demote_lookup_hit.sum","lts__t_sectors_evict_normal_demote_lookup_miss.sum","lts__t_sectors_evict_normal_lookup_hit.sum","lts__t_sectors_evict_normal_lookup_miss.sum","lts__t_sectors_lookup_hit.sum","lts__t_sectors_lookup_miss.sum","lts__t_sectors_requested_srcunit_tex_op_read_realtime.sum","lts__t_sectors_requested_srcunit_tex_op_read_realtime.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_requested_srcunit_tex_op_read_realtime.sum.per_second","lts__t_sectors_srcunit_gcc.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_gcc.avg.peak_sustained","lts__t_sectors_srcunit_gcc.avg.per_cycle_elapsed","lts__t_sectors_srcunit_gcc.sum","lts__t_sectors_srcunit_gcc.sum.per_second","lts__t_sectors_srcunit_gcc_lookup_hit.sum","lts__t_sectors_srcunit_gcc_lookup_miss.sum","lts__t_sectors_srcunit_tex.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex.avg.peak_sustained","lts__t_sectors_srcunit_tex.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex.max.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex.min.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex.sum","lts__t_sectors_srcunit_tex.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex.sum.per_second","lts__t_sectors_srcunit_tex_aperture_peer_lookup_miss.avg","lts__t_sectors_srcunit_tex_aperture_peer_lookup_miss.max","lts__t_sectors_srcunit_tex_aperture_peer_lookup_miss.min","lts__t_sectors_srcunit_tex_aperture_peer_lookup_miss.sum","lts__t_sectors_srcunit_tex_aperture_sysmem_lookup_miss.avg","lts__t_sectors_srcunit_tex_aperture_sysmem_lookup_miss.max","lts__t_sectors_srcunit_tex_aperture_sysmem_lookup_miss.min","lts__t_sectors_srcunit_tex_aperture_sysmem_lookup_miss.sum","lts__t_sectors_srcunit_tex_evict_first_lookup_hit.sum","lts__t_sectors_srcunit_tex_evict_first_lookup_miss.sum","lts__t_sectors_srcunit_tex_evict_last_lookup_hit.sum","lts__t_sectors_srcunit_tex_evict_last_lookup_miss.sum","lts__t_sectors_srcunit_tex_evict_normal_demote_lookup_hit.sum","lts__t_sectors_srcunit_tex_evict_normal_demote_lookup_miss.sum","lts__t_sectors_srcunit_tex_evict_normal_lookup_hit.sum","lts__t_sectors_srcunit_tex_evict_normal_lookup_miss.sum","lts__t_sectors_srcunit_tex_lookup_hit.sum","lts__t_sectors_srcunit_tex_lookup_miss.avg","lts__t_sectors_srcunit_tex_lookup_miss.max","lts__t_sectors_srcunit_tex_lookup_miss.min","lts__t_sectors_srcunit_tex_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom.sum","lts__t_sectors_srcunit_tex_op_atom.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_atom.sum.per_second","lts__t_sectors_srcunit_tex_op_atom_dot_alu.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_atom_dot_alu.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_atom_dot_alu.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_atom_dot_alu.sum","lts__t_sectors_srcunit_tex_op_atom_dot_alu.sum.per_second","lts__t_sectors_srcunit_tex_op_atom_dot_alu_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_dot_alu_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom_dot_cas.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_atom_dot_cas.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_atom_dot_cas.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_atom_dot_cas.sum","lts__t_sectors_srcunit_tex_op_atom_dot_cas.sum.per_second","lts__t_sectors_srcunit_tex_op_atom_dot_cas_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_dot_cas_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom_evict_first_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_evict_first_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom_evict_last_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_evict_last_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom_evict_normal_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_evict_normal_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_read.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_read.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_read.sum","lts__t_sectors_srcunit_tex_op_read.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_read.sum.per_second","lts__t_sectors_srcunit_tex_op_read_evict_first_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_evict_first_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read_evict_last_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_evict_last_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read_evict_normal_demote_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_evict_normal_demote_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read_evict_normal_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_evict_normal_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_red.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_red.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_red.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_red.sum","lts__t_sectors_srcunit_tex_op_red.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_red.sum.per_second","lts__t_sectors_srcunit_tex_op_red_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_red_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_write.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_write.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_write.sum","lts__t_sectors_srcunit_tex_op_write.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_write.sum.per_second","lts__t_sectors_srcunit_tex_op_write_evict_first_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_evict_first_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write_evict_last_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_evict_last_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write_evict_normal_demote_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_evict_normal_demote_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write_evict_normal_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_evict_normal_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_lookup_miss.sum","lts__t_tag_requests.avg.pct_of_peak_sustained_elapsed","lts__t_tag_requests.max.pct_of_peak_sustained_elapsed","lts__t_tag_requests.min.pct_of_peak_sustained_elapsed","lts__t_tag_requests.sum.pct_of_peak_sustained_elapsed","lts__throughput.avg.pct_of_peak_sustained_elapsed","lts__throughput.max.pct_of_peak_sustained_elapsed","lts__throughput.min.pct_of_peak_sustained_elapsed","lts__throughput.sum.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.max.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.min.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.sum.pct_of_peak_sustained_elapsed","lts__xcomp_gxc_sectors_input.sum","lts__xcomp_gxc_sectors_input.sum.pct_of_peak_sustained_elapsed","lts__xcomp_gxc_sectors_input.sum.per_second","lts__xcomp_gxc_sectors_output.sum","lts__xcomp_gxc_sectors_output.sum.pct_of_peak_sustained_elapsed","lts__xcomp_gxc_sectors_output.sum.per_second","memory_access_size_type","memory_access_type","memory_l1_tag_requests_global","memory_l1_wavefronts_shared","memory_l1_wavefronts_shared_ideal","memory_l2_theoretical_sectors_global","memory_l2_theoretical_sectors_global_ideal","memory_l2_theoretical_sectors_local","memory_type","numa__cpu_affinity","numa__dev_display_name_all","numa__id_cpu","numa__id_memory","nvlink__bandwidth","nvlink__count_logical","nvlink__count_physical","nvlink__destination_ports","nvlink__dev0Id","nvlink__dev0type","nvlink__dev1Id","nvlink__dev1type","nvlink__dev_display_name_all","nvlink__enabled_mask","nvlink__is_direct_link","nvlink__is_nvswitch_connected","nvlink__max_count","nvlink__peer_access","nvlink__peer_atomic","nvlink__source_ports","nvlink__system_access","nvlink__system_atomic","pmsampling:gr__ctas_launched_queue_sync_realtime.sum","pmsampling:l1tex__data_pipe_lsu_wavefronts.avg","pmsampling:l1tex__lsu_writeback_active.avg","pmsampling:l1tex__t_sector_hit_rate.pct","pmsampling:lts__average_t_sector_hit_rate_realtime.pct","pmsampling:sm__average_thread_inst_executed_pred_on_per_inst_executed_realtime.pct","pmsampling:sm__cycles_active.avg","pmsampling:sm__inst_executed_pipe_alu_realtime.avg.pct_of_peak_sustained_elapsed","pmsampling:sm__inst_executed_realtime.avg.per_cycle_active","pmsampling:sm__pipe_tensor_cycles_active_realtime.avg.pct_of_peak_sustained_elapsed","pmsampling:smsp__inst_executed_pipe_fmaheavy.avg.pct_of_peak_sustained_elapsed","pmsampling:smsp__inst_executed_pipe_fmalite.avg.pct_of_peak_sustained_elapsed","pmsampling:tpc__warps_active_realtime.avg.per_cycle_active","pmsampling:tpc__warps_active_realtime.sum.per_cycle_active","profiler__perfworks_session_reuse","profiler__pmsampler_buffer_size_bytes","profiler__pmsampler_ctxsw_0","profiler__pmsampler_ctxsw_1","profiler__pmsampler_interval_time","profiler__pmsampler_merged_samples","profiler__pmsampler_pass_groups","profiler__replayer_bytes_mem_accessible.avg","profiler__replayer_bytes_mem_accessible.max","profiler__replayer_bytes_mem_accessible.min","profiler__replayer_bytes_mem_accessible.sum","profiler__replayer_bytes_mem_backed_up.avg","profiler__replayer_bytes_mem_backed_up.max","profiler__replayer_bytes_mem_backed_up.min","profiler__replayer_bytes_mem_backed_up.sum","profiler__replayer_passes","profiler__replayer_passes_type_warmup","profiler__timestamp_workload_end_0","profiler__timestamp_workload_end_1","profiler__timestamp_workload_start_0","profiler__timestamp_workload_start_1","sass__inst_executed_global_loads","sass__inst_executed_global_stores","sass__inst_executed_local_loads","sass__inst_executed_local_stores","sass__inst_executed_per_opcode","sass__inst_executed_per_opcode_category","sass__inst_executed_per_opcode_with_modifier_all","sass__inst_executed_per_opcode_with_modifier_selective","sass__inst_executed_register_spilling","sass__inst_executed_register_spilling_mem_local","sass__inst_executed_register_spilling_mem_shared","sass__inst_executed_register_spilling_op_read","sass__inst_executed_register_spilling_op_write","sass__inst_executed_shared_loads","sass__inst_executed_shared_stores","sass__thread_inst_executed_per_opcode_category","sass__thread_inst_executed_true_per_opcode","sass__thread_inst_executed_true_per_opcode_with_modifier_all","sass__thread_inst_executed_true_per_opcode_with_modifier_selective","sm__cycles_active.avg","sm__cycles_active.max","sm__cycles_active.min","sm__cycles_active.sum","sm__cycles_elapsed.avg","sm__cycles_elapsed.avg.per_second","sm__cycles_elapsed.max","sm__cycles_elapsed.max.per_second","sm__cycles_elapsed.min","sm__cycles_elapsed.min.per_second","sm__cycles_elapsed.sum","sm__cycles_elapsed.sum.per_second","sm__dcc_request_hit_rate.pct","sm__dcc_requests.sum","sm__dcc_requests.sum.pct_of_peak_sustained_elapsed","sm__dcc_requests_lookup_hit.sum","sm__icc_request_hit_rate.pct","sm__icc_requests.sum","sm__icc_requests.sum.pct_of_peak_sustained_elapsed","sm__inst_executed.avg.pct_of_peak_sustained_elapsed","sm__inst_executed.avg.per_cycle_active","sm__inst_executed.avg.per_cycle_elapsed","sm__inst_executed.max.pct_of_peak_sustained_elapsed","sm__inst_executed.max.per_cycle_active","sm__inst_executed.max.per_cycle_elapsed","sm__inst_executed.min.pct_of_peak_sustained_elapsed","sm__inst_executed.min.per_cycle_active","sm__inst_executed.min.per_cycle_elapsed","sm__inst_executed.sum.pct_of_peak_sustained_elapsed","sm__inst_executed.sum.per_cycle_active","sm__inst_executed.sum.per_cycle_elapsed","sm__inst_executed_pipe_adu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_adu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_adu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_adu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_adu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_alu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_alu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_alu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_alu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_alu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_alu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_alu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_alu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_cbu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_cbu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_cbu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_cbu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma_type_fp16.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma_type_fp16.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_dmma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_dmma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_dmma.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_dmma.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_dmma.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_dmma.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_dmma.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_dmma.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_fp64.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_fp64.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_fp64.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_fp64.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_fp64.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_fp64.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_fp64.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_fp64.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_lsu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_lsu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_lsu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_lsu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tensor_subpipe_hmma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_tensor_subpipe_hmma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tensor_subpipe_imma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_tensor_subpipe_imma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_tex.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_tex.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_tex.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_tex.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_tma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tma.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_tma.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tma.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_tma.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tma.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_tma.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_uniform.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_uniform.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_uniform.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_uniform.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_workid.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_workid.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_workid.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_workid.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_workid.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_workid.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_workid.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_workid.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_xu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_xu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_xu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_xu.sum.pct_of_peak_sustained_elapsed","sm__inst_issued.avg.pct_of_peak_sustained_elapsed","sm__inst_issued.avg.per_cycle_active","sm__inst_issued.max.pct_of_peak_sustained_elapsed","sm__inst_issued.max.per_cycle_active","sm__inst_issued.min.pct_of_peak_sustained_elapsed","sm__inst_issued.min.per_cycle_active","sm__inst_issued.sum.pct_of_peak_sustained_elapsed","sm__inst_issued.sum.per_cycle_active","sm__instruction_throughput.avg.pct_of_peak_sustained_elapsed","sm__instruction_throughput.max.pct_of_peak_sustained_elapsed","sm__instruction_throughput.min.pct_of_peak_sustained_elapsed","sm__instruction_throughput.sum.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","sm__issue_active.avg.pct_of_peak_sustained_elapsed","sm__issue_active.max.pct_of_peak_sustained_elapsed","sm__issue_active.min.pct_of_peak_sustained_elapsed","sm__issue_active.sum.pct_of_peak_sustained_elapsed","sm__maximum_warps_avg_per_active_cycle","sm__maximum_warps_per_active_cycle_pct","sm__memory_throughput.avg.pct_of_peak_sustained_elapsed","sm__memory_throughput.max.pct_of_peak_sustained_elapsed","sm__memory_throughput.min.pct_of_peak_sustained_elapsed","sm__memory_throughput.sum.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.avg.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.max.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.min.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.sum.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.avg.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.max.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.min.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.sum.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.max.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.min.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.max.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.min.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.avg","sm__ops_path_tensor_src_bf16_dst_fp32.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.avg.peak_sustained","sm__ops_path_tensor_src_bf16_dst_fp32.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.avg.per_cycle_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.avg.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.max","sm__ops_path_tensor_src_bf16_dst_fp32.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.max.peak_sustained","sm__ops_path_tensor_src_bf16_dst_fp32.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.max.per_cycle_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.max.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.min","sm__ops_path_tensor_src_bf16_dst_fp32.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.min.peak_sustained","sm__ops_path_tensor_src_bf16_dst_fp32.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.min.per_cycle_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.min.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.sum","sm__ops_path_tensor_src_bf16_dst_fp32.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.sum.peak_sustained","sm__ops_path_tensor_src_bf16_dst_fp32.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.sum.per_cycle_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.sum.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.avg","sm__ops_path_tensor_src_fp16_dst_fp16.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.avg.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp16.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.avg.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.max","sm__ops_path_tensor_src_fp16_dst_fp16.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.max.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp16.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.max.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.min","sm__ops_path_tensor_src_fp16_dst_fp16.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.min.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp16.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.min.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.sum","sm__ops_path_tensor_src_fp16_dst_fp16.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.sum.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp16.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.sum.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.avg","sm__ops_path_tensor_src_fp16_dst_fp32.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.avg.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp32.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.avg.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.max","sm__ops_path_tensor_src_fp16_dst_fp32.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.max.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp32.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.max.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.min","sm__ops_path_tensor_src_fp16_dst_fp32.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.min.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp32.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.min.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.sum","sm__ops_path_tensor_src_fp16_dst_fp32.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.sum.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp32.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.sum.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_src_fp64.avg","sm__ops_path_tensor_src_fp64.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp64.avg.peak_sustained","sm__ops_path_tensor_src_fp64.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp64.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp64.avg.per_second","sm__ops_path_tensor_src_fp64.max","sm__ops_path_tensor_src_fp64.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp64.max.peak_sustained","sm__ops_path_tensor_src_fp64.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp64.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp64.max.per_second","sm__ops_path_tensor_src_fp64.min","sm__ops_path_tensor_src_fp64.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp64.min.peak_sustained","sm__ops_path_tensor_src_fp64.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp64.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp64.min.per_second","sm__ops_path_tensor_src_fp64.sum","sm__ops_path_tensor_src_fp64.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp64.sum.peak_sustained","sm__ops_path_tensor_src_fp64.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp64.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp64.sum.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_src_int8.avg","sm__ops_path_tensor_src_int8.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_int8.avg.peak_sustained","sm__ops_path_tensor_src_int8.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_int8.avg.per_cycle_elapsed","sm__ops_path_tensor_src_int8.avg.per_second","sm__ops_path_tensor_src_int8.max","sm__ops_path_tensor_src_int8.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_int8.max.peak_sustained","sm__ops_path_tensor_src_int8.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_int8.max.per_cycle_elapsed","sm__ops_path_tensor_src_int8.max.per_second","sm__ops_path_tensor_src_int8.min","sm__ops_path_tensor_src_int8.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_int8.min.peak_sustained","sm__ops_path_tensor_src_int8.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_int8.min.per_cycle_elapsed","sm__ops_path_tensor_src_int8.min.per_second","sm__ops_path_tensor_src_int8.sum","sm__ops_path_tensor_src_int8.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_int8.sum.peak_sustained","sm__ops_path_tensor_src_int8.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_int8.sum.per_cycle_elapsed","sm__ops_path_tensor_src_int8.sum.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.avg","sm__ops_path_tensor_src_tf32_dst_fp32.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.avg.peak_sustained","sm__ops_path_tensor_src_tf32_dst_fp32.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.avg.per_cycle_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.avg.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.max","sm__ops_path_tensor_src_tf32_dst_fp32.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.max.peak_sustained","sm__ops_path_tensor_src_tf32_dst_fp32.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.max.per_cycle_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.max.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.min","sm__ops_path_tensor_src_tf32_dst_fp32.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.min.peak_sustained","sm__ops_path_tensor_src_tf32_dst_fp32.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.min.per_cycle_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.min.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.sum","sm__ops_path_tensor_src_tf32_dst_fp32.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.sum.peak_sustained","sm__ops_path_tensor_src_tf32_dst_fp32.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.sum.per_cycle_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.sum.per_second","sm__pipe_alu_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_alu_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_alu_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_alu_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_alu_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_alu_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_alu_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_alu_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_fma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_fma_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_fma_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_fma_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_fp64_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_fp64_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_fp64_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_fp64_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_tensor_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_tensor_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_tensor_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_tensor_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_tensor_subpipe_hmma_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_tensor_subpipe_hmma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tensor_subpipe_imma_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_tensor_subpipe_imma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_tma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_tma_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_tma_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_tma_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__sass_inst_executed_op_ldgsts_cache_access.sum","sm__sass_inst_executed_op_ldgsts_cache_bypass.sum","sm__sass_l1tex_m_xbar2l1tex_read_bytes_mem_global_op_ldgsts_cache_bypass.sum","sm__sass_l1tex_m_xbar2l1tex_read_bytes_mem_global_op_ldgsts_cache_bypass.sum.pct_of_peak_sustained_elapsed","sm__sass_l1tex_m_xbar2l1tex_read_bytes_mem_global_op_ldgsts_cache_bypass.sum.per_second","sm__sass_l1tex_t_requests_pipe_lsu_mem_global_op_ldgsts.sum","sm__sass_l1tex_t_requests_pipe_lsu_mem_global_op_ldgsts.sum.pct_of_peak_sustained_elapsed","sm__sass_l1tex_t_requests_pipe_lsu_mem_global_op_ldgsts_cache_access.sum","sm__sass_l1tex_t_requests_pipe_lsu_mem_global_op_ldgsts_cache_bypass.sum","sm__sass_l1tex_t_sectors_pipe_lsu_mem_global_op_ldgsts_cache_access.sum","sm__sass_l1tex_t_sectors_pipe_lsu_mem_global_op_ldgsts_cache_access.sum.pct_of_peak_sustained_elapsed","sm__sass_l1tex_t_sectors_pipe_lsu_mem_global_op_ldgsts_cache_bypass.sum","sm__sass_l1tex_t_sectors_pipe_lsu_mem_global_op_ldgsts_cache_bypass.sum.pct_of_peak_sustained_elapsed","sm__sass_thread_inst_executed_op_dfma_pred_on.avg.peak_sustained","sm__sass_thread_inst_executed_op_dfma_pred_on.max.peak_sustained","sm__sass_thread_inst_executed_op_dfma_pred_on.min.peak_sustained","sm__sass_thread_inst_executed_op_dfma_pred_on.sum.peak_sustained","sm__sass_thread_inst_executed_op_ffma_pred_on.avg.peak_sustained","sm__sass_thread_inst_executed_op_ffma_pred_on.max.peak_sustained","sm__sass_thread_inst_executed_op_ffma_pred_on.min.peak_sustained","sm__sass_thread_inst_executed_op_ffma_pred_on.sum.peak_sustained","sm__sass_thread_inst_executed_op_hfma_pred_on.avg.peak_sustained","sm__sass_thread_inst_executed_op_hfma_pred_on.max.peak_sustained","sm__sass_thread_inst_executed_op_hfma_pred_on.min.peak_sustained","sm__sass_thread_inst_executed_op_hfma_pred_on.sum.peak_sustained","sm__throughput.avg.pct_of_peak_sustained_elapsed","sm__throughput.max.pct_of_peak_sustained_elapsed","sm__throughput.min.pct_of_peak_sustained_elapsed","sm__throughput.sum.pct_of_peak_sustained_elapsed","sm__warps_active.avg.pct_of_peak_sustained_active","sm__warps_active.avg.per_cycle_active","sm__warps_active.max.pct_of_peak_sustained_active","sm__warps_active.max.per_cycle_active","sm__warps_active.min.pct_of_peak_sustained_active","sm__warps_active.min.per_cycle_active","sm__warps_active.sum.pct_of_peak_sustained_active","sm__warps_active.sum.per_cycle_active","smsp__average_warp_latency_per_inst_issued.ratio","smsp__average_warps_active_per_inst_executed.ratio","smsp__average_warps_issue_stalled_barrier_per_issue_active.ratio","smsp__average_warps_issue_stalled_branch_resolving_per_issue_active.ratio","smsp__average_warps_issue_stalled_dispatch_stall_per_issue_active.ratio","smsp__average_warps_issue_stalled_drain_per_issue_active.ratio","smsp__average_warps_issue_stalled_lg_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_long_scoreboard_per_issue_active.ratio","smsp__average_warps_issue_stalled_math_pipe_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_membar_per_issue_active.ratio","smsp__average_warps_issue_stalled_mio_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_misc_per_issue_active.ratio","smsp__average_warps_issue_stalled_no_instruction_per_issue_active.ratio","smsp__average_warps_issue_stalled_not_selected_per_issue_active.ratio","smsp__average_warps_issue_stalled_selected_per_issue_active.ratio","smsp__average_warps_issue_stalled_short_scoreboard_per_issue_active.ratio","smsp__average_warps_issue_stalled_sleeping_per_issue_active.ratio","smsp__average_warps_issue_stalled_tex_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_wait_per_issue_active.ratio","smsp__branch_targets_threads_divergent","smsp__cycles_active.avg","smsp__cycles_active.max","smsp__cycles_active.min","smsp__cycles_active.sum","smsp__cycles_elapsed.avg","smsp__cycles_elapsed.avg.per_second","smsp__cycles_elapsed.max","smsp__cycles_elapsed.max.per_second","smsp__cycles_elapsed.min","smsp__cycles_elapsed.min.per_second","smsp__cycles_elapsed.sum","smsp__cycles_elapsed.sum.per_second","smsp__inst_executed.avg","smsp__inst_executed.max","smsp__inst_executed.min","smsp__inst_executed.sum","smsp__inst_executed_op_branch.avg","smsp__inst_executed_op_branch.max","smsp__inst_executed_op_branch.min","smsp__inst_executed_op_branch.sum","smsp__inst_executed_op_generic_atom_dot_alu.sum","smsp__inst_executed_op_generic_atom_dot_alu.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_generic_atom_dot_cas.sum","smsp__inst_executed_op_generic_atom_dot_cas.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_global_red.sum","smsp__inst_executed_op_global_red.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_ldgsts.sum","smsp__inst_executed_op_ldgsts.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_ldsm.sum","smsp__inst_executed_op_ldsm.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_shared_atom.sum","smsp__inst_executed_op_shared_atom.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_shared_stsm.sum","smsp__inst_executed_op_shared_stsm.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_atom_dot_alu.sum","smsp__inst_executed_op_surface_atom_dot_alu.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_atom_dot_cas.sum","smsp__inst_executed_op_surface_atom_dot_cas.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_ld.sum","smsp__inst_executed_op_surface_ld.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_red.sum","smsp__inst_executed_op_surface_red.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_st.sum","smsp__inst_executed_op_surface_st.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_texture.sum","smsp__inst_executed_op_texture.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_tma_ld.sum","smsp__inst_executed_op_tma_ld.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_tma_st.sum","smsp__inst_executed_op_tma_st.sum.pct_of_peak_sustained_elapsed","smsp__inst_issued.avg","smsp__inst_issued.max","smsp__inst_issued.min","smsp__inst_issued.sum","smsp__issue_active.avg.pct_of_peak_sustained_active","smsp__issue_active.avg.per_cycle_active","smsp__issue_active.max.pct_of_peak_sustained_active","smsp__issue_active.max.per_cycle_active","smsp__issue_active.min.pct_of_peak_sustained_active","smsp__issue_active.min.per_cycle_active","smsp__issue_active.sum.pct_of_peak_sustained_active","smsp__issue_active.sum.per_cycle_active","smsp__issue_inst0.avg.pct_of_peak_sustained_active","smsp__issue_inst0.max.pct_of_peak_sustained_active","smsp__issue_inst0.min.pct_of_peak_sustained_active","smsp__issue_inst0.sum.pct_of_peak_sustained_active","smsp__maximum_warps_avg_per_active_cycle","smsp__pcsamp_aggregated_passes","smsp__pcsamp_buffer_size_bytes","smsp__pcsamp_dropped_bytes","smsp__pcsamp_interval","smsp__pcsamp_interval_cycles","smsp__pcsamp_sample_count","smsp__pcsamp_warps_issue_stalled_barrier","smsp__pcsamp_warps_issue_stalled_barrier_not_issued","smsp__pcsamp_warps_issue_stalled_branch_resolving","smsp__pcsamp_warps_issue_stalled_branch_resolving_not_issued","smsp__pcsamp_warps_issue_stalled_dispatch_stall","smsp__pcsamp_warps_issue_stalled_dispatch_stall_not_issued","smsp__pcsamp_warps_issue_stalled_drain","smsp__pcsamp_warps_issue_stalled_drain_not_issued","smsp__pcsamp_warps_issue_stalled_lg_throttle","smsp__pcsamp_warps_issue_stalled_lg_throttle_not_issued","smsp__pcsamp_warps_issue_stalled_long_scoreboard","smsp__pcsamp_warps_issue_stalled_long_scoreboard_not_issued","smsp__pcsamp_warps_issue_stalled_math_pipe_throttle","smsp__pcsamp_warps_issue_stalled_math_pipe_throttle_not_issued","smsp__pcsamp_warps_issue_stalled_membar","smsp__pcsamp_warps_issue_stalled_membar_not_issued","smsp__pcsamp_warps_issue_stalled_mio_throttle","smsp__pcsamp_warps_issue_stalled_mio_throttle_not_issued","smsp__pcsamp_warps_issue_stalled_misc","smsp__pcsamp_warps_issue_stalled_misc_not_issued","smsp__pcsamp_warps_issue_stalled_no_instructions","smsp__pcsamp_warps_issue_stalled_no_instructions_not_issued","smsp__pcsamp_warps_issue_stalled_not_selected","smsp__pcsamp_warps_issue_stalled_not_selected_not_issued","smsp__pcsamp_warps_issue_stalled_selected","smsp__pcsamp_warps_issue_stalled_selected_not_issued","smsp__pcsamp_warps_issue_stalled_short_scoreboard","smsp__pcsamp_warps_issue_stalled_short_scoreboard_not_issued","smsp__pcsamp_warps_issue_stalled_sleeping","smsp__pcsamp_warps_issue_stalled_sleeping_not_issued","smsp__pcsamp_warps_issue_stalled_tex_throttle","smsp__pcsamp_warps_issue_stalled_tex_throttle_not_issued","smsp__pcsamp_warps_issue_stalled_wait","smsp__pcsamp_warps_issue_stalled_wait_not_issued","smsp__sass_average_branch_targets_threads_uniform.pct","smsp__sass_average_data_bytes_per_sector_mem_global_op_ld.max_rate","smsp__sass_average_data_bytes_per_sector_mem_global_op_ld.ratio","smsp__sass_average_data_bytes_per_sector_mem_global_op_st.max_rate","smsp__sass_average_data_bytes_per_sector_mem_global_op_st.ratio","smsp__sass_average_data_bytes_per_sector_mem_local_op_ld.max_rate","smsp__sass_average_data_bytes_per_sector_mem_local_op_ld.ratio","smsp__sass_average_data_bytes_per_sector_mem_local_op_st.max_rate","smsp__sass_average_data_bytes_per_sector_mem_local_op_st.ratio","smsp__sass_branch_targets_threads_divergent.avg","smsp__sass_branch_targets_threads_divergent.max","smsp__sass_branch_targets_threads_divergent.min","smsp__sass_branch_targets_threads_divergent.sum","smsp__sass_inst_executed_memdesc_explicit_evict_type","smsp__sass_inst_executed_memdesc_explicit_hitprop_evict_first","smsp__sass_inst_executed_memdesc_explicit_hitprop_evict_last","smsp__sass_inst_executed_memdesc_explicit_hitprop_evict_normal","smsp__sass_inst_executed_memdesc_explicit_hitprop_evict_normal_demote","smsp__sass_inst_executed_memdesc_explicit_missprop_evict_first","smsp__sass_inst_executed_memdesc_explicit_missprop_evict_normal","smsp__sass_inst_executed_op_dshared.sum","smsp__sass_inst_executed_op_dshared.sum.pct_of_peak_sustained_elapsed","smsp__sass_inst_executed_op_dshared_atom.sum","smsp__sass_inst_executed_op_dshared_ld.sum","smsp__sass_inst_executed_op_dshared_redas.sum","smsp__sass_inst_executed_op_dshared_st.sum","smsp__sass_inst_executed_op_dshared_stas.sum","smsp__sass_inst_executed_op_dshared_tma_red.sum","smsp__sass_inst_executed_op_dshared_tma_st.sum","smsp__sass_inst_executed_op_global_ld.sum","smsp__sass_inst_executed_op_global_st.sum","smsp__sass_inst_executed_op_local_ld.sum","smsp__sass_inst_executed_op_local_st.sum","smsp__sass_inst_executed_op_shared.sum","smsp__sass_inst_executed_op_shared.sum.pct_of_peak_sustained_elapsed","smsp__sass_inst_executed_op_shared_ld.sum","smsp__sass_inst_executed_op_shared_st.sum","smsp__sass_inst_executed_op_tma_ld.sum","smsp__sass_inst_executed_op_tma_red.sum","smsp__sass_inst_executed_op_tma_st.sum","smsp__sass_l1tex_data_pipe_lsu_wavefronts_mem_shared_op_ldgsts.sum","smsp__sass_l1tex_data_pipe_lsu_wavefronts_mem_shared_op_ldgsts.sum.pct_of_peak_sustained_elapsed","smsp__sass_l1tex_data_pipe_lsu_wavefronts_mem_shared_op_ldgsts_cache_access.sum","smsp__sass_l1tex_data_pipe_lsu_wavefronts_mem_shared_op_ldgsts_cache_access.sum.pct_of_peak_sustained_elapsed","smsp__sass_l1tex_m_xbar2l1tex_read_sectors_mem_global_op_ldgsts_cache_bypass.sum","smsp__sass_l1tex_m_xbar2l1tex_read_sectors_mem_global_op_ldgsts_cache_bypass.sum.pct_of_peak_sustained_elapsed","smsp__sass_thread_inst_executed_op_dadd_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dadd_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dadd_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dadd_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dfma_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dfma_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dfma_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dfma_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dmul_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dmul_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dmul_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dmul_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fadd_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fadd_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fadd_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fadd_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_ffma_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_ffma_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_ffma_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_ffma_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fmul_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fmul_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fmul_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fmul_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hadd_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hadd_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hadd_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hadd_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hfma_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hfma_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hfma_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hfma_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hmul_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hmul_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hmul_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hmul_pred_on.sum.per_cycle_elapsed","smsp__thread_inst_executed_per_inst_executed.ratio","smsp__thread_inst_executed_pred_on_per_inst_executed.ratio","smsp__warps_active.avg.peak_sustained","smsp__warps_active.avg.per_cycle_active","smsp__warps_active.max.peak_sustained","smsp__warps_active.max.per_cycle_active","smsp__warps_active.min.peak_sustained","smsp__warps_active.min.per_cycle_active","smsp__warps_active.sum.peak_sustained","smsp__warps_active.sum.per_cycle_active","smsp__warps_eligible.avg.per_cycle_active","smsp__warps_eligible.max.per_cycle_active","smsp__warps_eligible.min.per_cycle_active","smsp__warps_eligible.sum.per_cycle_active","thread_inst_executed","thread_inst_executed_true" +"","","","","","","","","","","","","","thread","thread","thread","thread","","","","%","Kbyte","Gbyte","","","byte","","%","%/register","%/Kbyte","thread","thread","thread","%","thread","thread","thread","thread","thread","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","request","%","request","%","","%","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","ms","ms","ms","ms","","","","","block","block","block","block","","","","","%","%","%","%","%","","%","inst","cycle","cycle","cycle","cycle","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","","","","","","%","%","%","%","%","%","%","%","%","%","%","%","","%","","%","","%","","%","%","%","%","%","%","%","%","%","%","%","%","%","cycle","%","","Mhz","%","%","%","%","%","%","%","%","Mbyte","%","Gbyte/s","byte","%","byte/s","byte","%","byte/s","Mbyte","%","Gbyte/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector","%","sector","%","sector","%","sector","%","sector/s","sector","%","sector/us","sector","%","sector","%","sector","%","sector","%","Gbyte","%","Gbyte/s","byte","%","byte/s","Gbyte","%","Gbyte/s","%","%","%","%","sector","%","sector/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector","%","sector","%","sector","%","sector/ns","sector","%","sector","%","sector","%","sector","%","byte","%","byte/s","","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","%","%","%","%","%","%","%","%","%","%","%","%","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","%","%","cycle","%","%","%","%","%","%","%","%","%","%","%","%","%","","block","block","block","","","","","cluster","block","","","","","","","","","","","","%","%","block","block","block","block","block","","","","","","Mbyte","","","","","register/thread","register/thread","Kbyte","Kbyte/block","Kbyte/block","Kbyte/block","Kbyte/block","byte/block","SM","","","thread","","","","","","","","","%","sector","%","","","","cycle","cycle","cycle","cycle","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","%","%","%","%","%","%","%","%","%","%","%","%","sector","%","sector/ns","sector","sector","sector","sector","%","","Mhz","%","","Mhz","%","","Mhz","%","","Ghz","request","request","request","request","request","request","request","request","%","%","%","sector","%","sector/cycle","sector/cycle","sector","%","sector","%","sector","%","sector/ns","sector","%","sector/us","%","sector/cycle","sector/cycle","sector","sector/s","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","sector/ns","%","sector/cycle","sector/cycle","sector","sector/us","sector","sector","%","sector/cycle","sector/cycle","%","%","sector","%","sector/ns","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","sector/s","%","sector/cycle","sector/cycle","sector","sector/s","sector","sector","%","sector/cycle","sector/cycle","sector","sector/s","sector","sector","sector","sector","sector","sector","sector","sector","%","sector/cycle","sector/cycle","sector","%","sector/ns","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","sector/cycle","sector/cycle","sector","%","sector/s","sector","sector","%","sector/cycle","sector/cycle","sector","%","sector/us","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","%","%","%","%","%","%","%","%","%","%","%","sector","%","sector/s","sector","%","sector/s","","","sectors","sectors","sectors","sectors","sectors","sectors","","","","","","","","","","","","","","","","","","","","","","","","block","","cycle","%","%","%","cycle","%","inst/cycle","%","%","%","warp","warp","","Mbyte","","","us","sample","","Gbyte","Gbyte","Gbyte","Gbyte","Gbyte","Gbyte","Gbyte","Gbyte","pass","pass","","","","","inst","inst","inst","inst","","","","","inst","inst","inst","inst","inst","inst","inst","","","","","cycle","cycle","cycle","cycle","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","%","cycle","%","cycle","%","cycle","%","%","inst/cycle","inst/cycle","%","inst/cycle","inst/cycle","%","inst/cycle","inst/cycle","%","inst/cycle","inst/cycle","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","inst/cycle","%","inst/cycle","%","inst/cycle","%","inst/cycle","%","%","%","%","%","%","%","%","%","%","%","%","warp","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","inst","inst","byte","%","byte/s","","%","","","sector","%","sector","%","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","%","%","%","%","%","warp","%","warp","%","warp","%","warp","cycle","cycle","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","branches","cycle","cycle","cycle","cycle","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","inst","inst","inst","inst","inst","inst","inst","inst","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","inst","inst","inst","%","","%","","%","","%","","%","%","%","%","warp","pass","Mbyte","byte","","cycle","","warp","warp","branches","branches","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","inst","inst","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","%","byte/sector","byte/sector","byte/sector","byte/sector","byte/sector","byte/sector","byte/sector","byte/sector","branches","branches","branches","branches","","inst","inst","inst","inst","inst","inst","inst","%","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","%","inst","inst","inst","inst","inst","","%","","%","sector","%","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","","","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","inst","inst" +"0","73","python3.12","127.0.0.1","void at::vectorized_elementwise_kernel<4, at::BinaryFunctor>, std::array>(int, T2, T3)","1","7","(128, 1, 1)","(1, 1, 1)","0","12.1","0","1","3802","2078","3874.000000","1694","6144","34334763.95","0","0.000000","1.024000","3.326180","0","0","0","692","5270.000000","10876.000000","1.959000","192.000000","12288.000000","12288.000000","0.075665","0","0.000000","0","0.128805","0","37","489","0","0","nan","0","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","4.000000","0.006289","30.000000","0.047170","672","0.528302","15900.000000","2.132511","15903.000000","2.132913","15897.000000","2.132108","63600.000000","8.530043","0.264263","0.861043","0.014001","0.264263","0.146132","0.448022","0.014001","0.146132","0.251575","0.918239","0.056604","0.251575","0","0","0","0","0.264263","0.918239","0.056604","0.264263","0.007456","0.007456","0.007456","0.007456","0","0","0","0","0.000000","0.000000","0.000000","0.000000","0","0","0","0","0.026468","1.270440","0.000000","0.026468","53.623188","69","0.018082","489","46.625000","2238.000000","0.000000","2238.000000","15900.000000","2.132511","15903.000000","2.132913","15897.000000","2.132108","763200.000000","102.360515","0","0","0","0","0","0.000008","0.000393","0.000000","0.000008","0.000037","0.001769","0.000000","0.000037","0.009565","0.459119","0.000000","0.009565","70","0.009172","0","0","0","0.000000","0","0.000000","0","0","0","0","0","0","0","0","0.000786","0.037736","0.000000","0.000786","2.000000","0.000262","48","0.268240","0.019130","0.918239","0.000000","0.019130","0.012972","0.031447","0.012579","0.012972","0.000032","0.000131","0.004292","0","0","0","0","0","0","0.000000","0.000000","0.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0.000000","0.000000","0.000000","1.000000","0.000131","0","0","0","0","0","0","0.000000","0.000262","0.008584","0","0","0","0.000000","0.000000","0.000000","0.000262","0.012579","0.000000","0.000262","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0.000000","0.000000","0.000000","2.000000","0.000262","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","2","0.000262","0","0","1","0.000131","0","0.000000","0","0.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","2","0.000262","0","0","1","0.000131","0","0.000000","0","0.000000","0","0","0","0","0","0","0","0","0","0","0.000000","0","0.000000","0","0","0.000000","0.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","2.000000","0.000000","2.000000","0","0","0","1.000000","0","1.000000","0.000000","0.000000","0","0.000000","0.000000","0.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0.056604","0.056604","0.056604","0.056604","19.302949","0.056604","313.136729","0.918239","19.302949","0.056604","19.302949","0.056604","0","128","1","1","128","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","1","1","1","1","","0","0","24.000000","24.000000","12.000000","32.000000","12.000000","336","2540","0","5264","956","4.718592","0","0","0","0","40","40.000000","32.768000","1.024000","1.024000","1.024000","0.000000","0","48","1024","7","128","24","all","0","0","0","0","0","0.00","100","0","0","0","0","0","1253.875000","3557.000000","149.000000","20062.000000","14285.000000","1.915907","14285.000000","1.915907","14285.000000","1.915907","228560.000000","30.654506","0","0","0","0","0","0","0","0","0.169758","0.434022","0.000000","0.169758","185.000000","0.080942","0.024812","0","0","0","0","0.169540","2","6.496446","0.700035","2","26.824034","0.007000","2","0.268240","0.169540","32","0.103943","301.000000","176.000000","3.000000","0","0","2.000000","0","1.000000","48.758278","73.684211","1.995012","75.500000","0.264263","2.000000","0.005285","246.000000","0.861043","2.000000","0.007000","1208.000000","0.264263","0.162017","401.000000","0.175446","53.782189","0","8.000000","0","0","0","57.000000","284.000000","0","0","0","0","549.000000","193.000000","589.000000","535.000000","0.000000","0.000000","0.000000","0.308015","1.000000","0.003080","704.000000","94.420601","533.000000","171.000000","0.000656","2.000000","0.000013","0.003500","0.000000","3.000000","0.000656","0.000402","0","0","0","0","0.187500","1.000000","0.000000","3.000000","0","0","0","0","0","0","0.000000","3.000000","0.000000","0.187500","1.000000","0.000000","3.000000","0","0","0","0","1.000000","0","0","0","0","0","0","1.000000","0","0","0","0","0","0","0","0","0","0","0","0.000438","2.000000","0.000009","2.000000","0.000438","0.000268","0","0","0","0","0","0","0.000000","2.000000","0.000000","2.000000","0","1.000000","0","0","0","0","0","0","0.000438","1.000000","0.000004","1.000000","0.000438","0.134120","0","0","0","0","0","0","0","1.000000","0","1.000000","0.139132","0.469023","0.000000","0.139132","0.264263","0.861043","0.014001","0.264263","0.251575","0.602030","0.014001","0.251575","0","0","0","0","0","0","0","0","3","0","0","3","3","0","0","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","0.000000","1.52","0.125000","","","","641.229167","0.000055","0.154119","0.000000","0.000117","0.000023","0.286469","13.750532","0","4.194304","1","1","3.000000","0","2","25.185856","25.185856","25.185856","957.062541","25.185856","25.185856","25.185856","957.062541","38.000000","0","1","1","1","1","2.000000","1.000000","0.000000","0.000000","489","489","489","489","0","no data","no data","0","0","0.000000","0.000000","8331","8331","8331","8331","46.625000","2238.000000","0.000000","2238.000000","15900.000000","2.132511","15903.000000","2.132913","15897.000000","2.132108","763200.000000","102.360515","0.000000","8.000000","0.001048","0.000000","73.333333","120.000000","0.031447","0.016018","0.218499","0.000641","0.768868","10.487936","0.030755","0.000000","0.000000","0.000000","0.016018","10.487936","0.030755","9.740840","0.028564","467.560322","1.371069","0.000000","0.000000","9.740840","0.028564","1.843164","0.005405","88.471850","0.259434","0.000000","0.000000","1.843164","0.005405","0.290438","0.000852","13.941019","0.040881","0.000000","0.000000","0.290438","0.000852","0.003407","0.163522","0.000000","0.003407","1.351653","0.003964","64.879357","0.190252","0.000000","0.000000","1.351653","0.003964","0.357462","0.001048","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","6.523682","0.019130","313.136729","0.918239","0.000000","0.000000","6.523682","0.019130","0.000000","0.000000","0","0","0","0","0","0","0","0","0","0","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0","0","0","0","0","0","0","0","0.016051","0.218945","0.770440","10.509383","0.000000","0.000000","0.016051","10.509383","0.016051","0.770440","0.000000","0.016051","0","0","0","0","0.016051","0.770440","0.000000","0.016051","48.000000","100.000000","0.028564","1.371069","0.000000","0.028564","0.001048","0.050314","0.000000","0.001048","0.002588","0.124214","0.000000","0.002588","0.015898","0.763103","0.000000","0.015898","0.000393","0.018868","0.000000","0.000393","0.000393","0.018868","0.000000","0.000393","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","49152","104817167381974.25","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","49152","104817167381974.25","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","49152","104817167381974.25","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","512","1091845493562.23","0","0","0","0","512","1091845493562.23","0","0","0","0","512","1091845493562.23","0","0","0","0","24576","52408583690987.12","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","49152","104817167381974.25","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","196608","419268669527897","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0.000000","4096","8734763948497.85","0","0","0","0.000000","4096","8734763948497.85","0","0","0","0.000000","4096","8734763948497.85","0","0","0","0.000000","196608","419268669527897","0","0","0","0","8192","17469527896995.71","0","0","0","0","8192","17469527896995.71","0","0","0","0","8192","17469527896995.71","0","0","0","0","393216","838537339055794","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","196608","419268669527897","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","196608","419268669527897","0","0","0","0","3.51","7478393791.52","0","0","0","0","3.51","7478393791.52","0","0","0","0","3.51","7478393791.52","0","0","0","0","168.33","358962901993.06","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","196608","419268669527897","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","2048","4367381974248.93","0","0","0","0","98304","209634334763948.50","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","196608","419268669527897","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","4096","8734763948497.85","0","0","0","0","196608","419268669527897","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","1024","2183690987124.46","0","0","0","0","49152","104817167381974.25","0","0","1.843164","0.005405","88.471850","0.259434","0.000000","0.000000","1.843164","0.005405","0.006420","0.308176","0.000000","0.006420","0.804290","0.002358","38.605898","0.113208","0.000000","0.000000","0.804290","0.002358","0.015789","0.757862","0.000000","0.015789","0","0","0","0","0","0","0","0","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0","0","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","2.000000","2.000000","2.000000","96.000000","128.000000","128.000000","128.000000","6144.000000","64.000000","64.000000","64.000000","3072.000000","0.028564","1.371069","0.000000","0.028564","32.530906","15.614835","1561.483467","749.512064","0.000000","0.000000","32.530906","749.512064","12.938776","12.965235","0.000000","0.381633","0.008163","0.016327","0","1.822449","0.022449","0","0.000000","0.012245","90.138776","0.000000","1.000000","2.073469","0.000000","0","2.055102","1","33.052083","2259.000000","0.000000","6346.000000","15900.000000","2.132511","15903.000000","2.132913","15897.000000","2.132108","3052800.000000","409.442060","2.546875","132.000000","0.000000","489.000000","0.192708","13.000000","0.000000","37.000000","0","0","0","0","0","0","0","0","0.000000","0.000000","0","0","0.000000","0.000000","0","0","0","0","0","0","0","0","0","0","0","0","0.000000","0.000000","0.000000","0.000000","2.552083","133.000000","0.000000","490.000000","7.721399","0.08","402.395210","4.02","0.000000","0","7.721399","14.83","92.278601","6432.272298","0.000000","92.278601","12.000000","2","33.554432","0","5","1024.000000","16","0","0","0","0","0","0","0","0","0","0","8","8","0","0","0","0","0","0","0","0","6","6","0","0","0","0","2","2","0","0","0","0","0","0","94.736842","32.000000","4.000000","32.000000","4.000000","32.000000","0.000000","32.000000","0.000000","0.005208","1.000000","0","1.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","2.000000","1.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0.000000","0","0.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0.000000","0.000000","0.000000","0.000000","0.000294","0.014088","0","0.056352","0","0","0","0","0.000168","0.008050","0","0.032201","0","0","0","0","31.15","17.04","12.000000","0.999055","12.000000","69.284589","12.000000","0.000000","2304.000000","191.818468","0.077214","4.023952","0.000000","14.825087","15233","8331" +"1","73","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu_stream_k","1","7","(384, 1, 1)","(42, 296, 1)","0","12.1","0","1","56167","52394","354166.000000","7598","6144","1620772988.98","696768","100.000000","1.024000","466.239645","294","44556288","0","625","316.000000","4225.000000","4.975000","192.000000","12288.000000","12288.000000","0.033180","0","3.443262","0","0.000000","0","42996614","1295861238","576","696768","1","1","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","2798544.000000","0.591359","9028.000000","0.001908","10712","0.001132","118309837.750000","2.148848","118334635.000000","2.149298","118285340.000000","2.148403","473239351.000000","8.595390","23.974686","23.993901","23.961532","23.974686","12.559269","12.628962","12.489907","12.559269","23.598833","23.616927","23.583289","23.598833","0","0","0","0","23.974686","23.993901","23.961532","23.974686","55.057344","55.057344","55.057344","55.057344","12384","12384","12384","12384","12384.000000","12384.000000","12384.000000","12384.000000","12432","12432","12432","12432","0.070323","0.070326","0.070310","0.070323","99.721630","1991232","0.070128","1259175041","118776167.583333","118805817.000000","118739425.000000","5701256044.000000","118309837.750000","2.148848","118334635.000000","2.149298","118285340.000000","2.148403","5678872212.000000","103.144682","44813290","0","44566514","0","250754","5.324639","5.324688","5.324632","5.324639","1.793456","1.793457","1.793455","1.793456","12.737439","12.738040","12.736932","12.737439","719808603","12.675203","0","0","713157733","12.558087","3440663","0.060587","0","0","0","0","0","0","0","0","11.787449","11.787449","11.787449","11.787449","697152.000000","0.012276","48","12.662289","7.999717","7.999717","7.999717","7.999717","3.560743","3.563408","3.559783","3.560743","406.757376","0.223833","7.387886","0","0","0","0","0","0","406.683648","0.223792","7.386547","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","12708864.000000","0.223792","230.829587","2304.000000","0.000041","0","0","0","0","0","0","25.664425","14.122756","466.139902","0","0","0","25.664422","14.122755","466.139847","14.122756","14.122756","14.122756","14.122756","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","802013184.000000","14.122755","14.566870","96.000000","0.000002","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","960","0.000017","0","0","0","0.000000","696192","0.012259","576","0.000010","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","960","0.000017","0","0","0","0.000000","696192","0.012259","576","0.000010","0","0","0","0","0","0","0","0","0","0","99.979413","0","90.000000","0","0","100.000000","97.916667","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","960.000000","864.000000","96.000000","0","0","0","0.000000","0","0.000000","696192.000000","696192.000000","0","2304.000000","2256.000000","48.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0.011638","0.011678","0.011589","0.011638","14.067309","14.122756","14.067309","14.122756","14.067309","14.122756","14.067309","14.122756","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","42","296","1","12432","","0","0","3.000000","24.000000","1.000000","1.000000","4.000000","300","157","0","2028","2388","4.718592","0","0","0","0","168","168.000000","102.400000","89.088000","89.088000","1.024000","88.064000","0","48","1024","7","4773888","24","all","0","0","0","0","0","259","100","0","0","0","1.00","0","107113733.250000","107135103.000000","107088888.000000","1713819732.000000","106226770.000000","1.929384","106226770.000000","1.929384","106226770.000000","1.929384","1699628320.000000","30.870147","0","0","0","0","0","0","0","0","17.584039","17.594977","17.573523","17.584039","367742418.000000","21.636637","6.679262","0","0","0","0","23.598833","2","910.624306","23.616927","2","911.322529","23.583289","2","910.024519","23.598833","32","14.569989","203788027.000000","2832.000000","203681184.000000","0","0","200503392.000000","0","3177792.000000","53.315402","54.157122","0.390314","50935069.437500","23.974686","2.000000","0.479494","50975893.000000","23.993901","50907123.000000","23.961532","814961111.000000","23.974686","14.802042","12769216.000000","0.751295","231.925754","0","8.000000","0","0","0","63635.000000","209527.000000","0","0","0","0","434448305.000000","380287608.000000","434499795.000000","380523451.000000","802013280.000000","5.898446","14.566872","0.000666","1.000000","0.000007","11328.000000","0.205749","8632.000000","2688.000000","23.967724","2.000000","0.479354","23.982646","23.954898","814724448.000000","23.967724","14.797743","0","0","0","0","23767776.125000","23784480.000000","23750968.000000","380284418.000000","0","0","0","0","0","0","434439826.000000","380285022.000000","434440114.000000","23767813.875000","23784508.000000","23750980.000000","380285022.000000","0","0","0","0","1.000000","0","0","0","0","0","0","1.000000","0","0","0","0","0","0","0","0","0","0","0","23.593784","2.000000","0.471876","802013280.000000","23.593784","14.566872","0","0","0","0","0","0","434439462.000000","367573666.000000","434439462.000000","367573666.000000","0","1.000000","0","0","0","0","0","0","0.747879","1.000000","0.007479","12711168.000000","0.747879","230.871435","0","0","0","0","0","0","0","12711168.000000","0","12711168.000000","11.990160","12.008335","11.982277","11.990160","23.974686","23.993901","23.961532","23.974686","11.920534","11.945496","11.914528","11.920534","0","0","0","0","0","0","0","0","960","716332272","671775984","960","960","698496","0","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","32.000000","15070513.85","13945712.000000","","","","116848615.062500","1.442512","1048.198735","24.916459","0.716718","0.027825","45312.875305","2175018.014626","0","28.770304","1","1","12.000000","0","2","25.185856","25.185856","25.185856","957.062541","25.185856","25.185856","25.185856","957.062541","38.000000","0","1","1","1","1","960.000000","0.000000","696192.000000","576.000000","1259175041","1259175041","1259175041","1259175041","696768","no data","no data","696192","576","89236896.000000","432.000000","31879588511","31879588511","31879588511","31879588511","118776167.583333","118805817.000000","118739425.000000","5701256044.000000","118309837.750000","2.148848","118334635.000000","2.149298","118285340.000000","2.148403","5678872212.000000","103.144682","99.622652","1476887.000000","0.026007","1471314.000000","99.945294","51933628.000000","1.829012","5.704747","0.227294","0.228190","5.705824","0.227337","0.228233","5.703584","0.227248","0.228143","5.704747","10.910112","10.953115","1.367979","1.373371","1.370839","1.376242","1.364895","1.370275","1.367979","1.373371","1.447909","1.453616","1.448026","1.453733","1.447743","1.453449","1.447909","1.453616","0.057268","0.057494","0.057434","0.057660","0.057106","0.057331","0.057268","0.057494","0.229003","0.229722","0.228182","0.229003","0.159191","0.159818","0.159191","0.159818","0.159191","0.159818","0.159191","0.159818","0.000000","0.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","7.968309","7.999717","7.968309","7.999717","7.968309","7.999717","7.968309","7.999717","25.008546","25.107119","0","0","0","0","0","0","0","0","0","0","0.157014","0.157633","0.157014","0.157633","0.157014","0.157633","0.157014","0.157633","1.046402","1.050527","1.046402","1.050527","1.046402","1.050527","1.046402","1.050527","0.000218","0.000219","0.000218","0.000219","0.000218","0.000219","0.000218","0.000219","0","0","0","0","0","0","0","0","5.748362","0.229032","5.749550","0.229079","5.747026","0.228978","5.748362","10.993524","25.107119","25.107119","25.107119","25.107119","0","0","0","0","5.748362","5.749550","5.747026","5.748362","12.000000","25.000000","7.999717","7.999717","7.999717","7.999717","0.026006","0.026006","0.026006","0.026006","2.955628","2.955628","2.955628","2.955628","3.159392","3.160349","3.158360","3.159392","1.329433","1.334115","1.324826","1.329433","1.329441","1.334122","1.324833","1.329441","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","49152","105620153872442.52","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","49152","105620153872442.52","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","49152","105620153872442.52","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","512","1100209936171.28","0","0","0","0","512","1100209936171.28","0","0","0","0","512","1100209936171.28","0","0","0","0","24576","52810076936221.26","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","49152","105620153872442.52","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","196608","422480615489770.06","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","121668370432","25.107119","4096","8801679489370.21","1028.39","2209848161800.18","121668370432","25.107119","4096","8801679489370.21","1028.39","2209848161800.18","121668370432","25.107119","4096","8801679489370.21","1028.39","2209848161800.18","5840081780736","25.107119","196608","422480615489770.06","49362.60","106072711766408.50","0","0","8192","17603358978740.42","0","0","0","0","8192","17603358978740.42","0","0","0","0","8192","17603358978740.42","0","0","0","0","393216","844961230979540.12","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","196608","422480615489770.06","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","196608","422480615489770.06","0","0","0","0","3.51","7535684494.32","0","0","0","0","3.51","7535684494.32","0","0","0","0","3.51","7535684494.32","0","0","0","0","168.33","361712855727.54","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","196608","422480615489770.06","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","2048","4400839744685.10","0","0","0","0","98304","211240307744885.03","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","196608","422480615489770.06","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","4096","8801679489370.21","0","0","0","0","196608","422480615489770.06","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","1024","2200419872342.55","0","0","0","0","49152","105620153872442.52","0","0","1.447899","1.453606","1.448020","1.453728","1.447714","1.453420","1.447899","1.453606","2.476614","2.476884","2.476247","2.476614","0.239030","0.239972","0.239497","0.240441","0.238574","0.239514","0.239030","0.239972","0.723077","0.723126","0.723030","0.723077","0","0","0","0","0","0","0","0","25.008546","25.107119","25.008546","25.107119","25.008546","25.107119","25.008546","25.107119","25.008546","25.107119","0","0","0.157014","0.157633","0.157014","0.157633","0.157014","0.157633","0.157014","0.157633","0","0","0","0","0","0","0","0","0","0","0","0","0","2.000000","2.000000","2.000000","96.000000","128.000000","128.000000","128.000000","6144.000000","64.000000","64.000000","64.000000","3072.000000","25.107119","25.107119","25.107119","25.107119","20.749477","9.959749","20.755541","9.962660","20.742386","9.956345","20.749477","478.067958","43.772834","44.107494","0.206873","0.162419","0.067445","0.000002","0","0.653241","3.781903","0","0.022919","0.018141","0.089313","0.278872","1.000002","0.105348","33.715651","0","3.744846","12480","118853673.520833","118890151.000000","118818410.000000","22819905316.000000","118309837.750000","2.148848","118334635.000000","2.149298","118285340.000000","2.148403","22715488848.000000","412.578726","6749277.281250","7300353.000000","6518321.000000","1295861238.000000","223940.697917","308958.000000","174699.000000","42996614.000000","0","0","0","0","0","0","0","0","133668864.000000","0.588448","0","0","1591296.000000","0.007005","0","0","0","0","0","0","0","0","0","0","0","0","2784768.000000","0.012259","99456.000000","0.000438","6800878.151042","7388883.000000","6547928.000000","1305768605.000000","5.722060","0.06","6.216790","0.06","5.509235","0.06","5.722060","10.99","94.277940","93.813901","94.461095","94.277940","3.000000","2","33.554432","0","5","1024.000000","3262109","32325","30631","11408","9613","5261","2048","1","1","0","0","59347","55050","294725","248195","0","0","2176","2043","1755","1465","7581","6623","21812","0","77232","0","7732","6612","2450234","2418252","0","0","290520","282914","99.937098","32.000000","4.000000","32.000000","0.000000","32.000000","4.000000","32.000000","1.000000","65.000000","307.000000","0","12480.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","960.000000","0.000000","696192.000000","576.000000","224497488.000000","0.988301","89236896.000000","432.000000","2784768.000000","0","99456.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0.008967","0.010594","0.007720","1.721631","0.000000","0.000000","0","0.000000","0","0","0","0","0.000000","0.000000","0","0.000000","0","0","0","0","31.18","25.44","12.000000","2.504708","12.000000","3.006371","12.000000","2.003303","2304.000000","480.903877","0.073179","0.078745","0.070831","14.050374","39306980848","31879588511" diff --git a/benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep b/benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep new file mode 100644 index 0000000..a28664a Binary files /dev/null and b/benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep differ diff --git a/benchmarks/gb10-fc2-nvfp4-baseline-profile-20260825.json b/benchmarks/gb10-fc2-nvfp4-baseline-profile-20260825.json new file mode 100644 index 0000000..8ed4f86 --- /dev/null +++ b/benchmarks/gb10-fc2-nvfp4-baseline-profile-20260825.json @@ -0,0 +1,1733 @@ +{ + "mode": "profile", + "environment": { + "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", + "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", + "torch": "2.9.1+cu130", + "torch_cuda": "13.0", + "device": "NVIDIA GB10", + "device_capability": [ + 12, + 1 + ], + "driver": null, + "git_commit": null, + "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", + "checkpoint_sha256": null, + "checkpoint_hash_note": "not calculated", + "environment_switches": { + "CUDA_DEVICE_MAX_CONNECTIONS": "1", + "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", + "CUDA_HOME": "/usr/local/cuda", + "CUDA_INC_PATH": "/usr/local/cuda/include", + "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", + "CUDA_MODULE_LOADING": "EAGER", + "CUDA_VERSION": "13.0.2", + "H3_FUSED_ELEMENTWISE": "1", + "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd", + "TORCH_COMPILE_DISABLE": "0", + "TORCH_CUDA_ARCH_LIST": "12.1a", + "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" + }, + "extension": { + "cuda_version": 13000, + "cublas_version": 130100, + "stream_k_public_control": false, + "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + }, + "comfy_kitchen": "0.2.31 package without __version__" + }, + "workload": { + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "steps": 12, + "sampler_step": 1, + "seed": 440420, + "text_tokens": 100, + "tokens": 37810, + "hidden_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ] + }, + "retained_blocks": [ + 0, + 24, + 49 + ], + "immutable_cloned_block_inputs": { + "0": [ + 37810, + 5376 + ], + "24": [ + 37810, + 5376 + ], + "49": [ + 37810, + 5376 + ] + }, + "fc2_boundary": { + "block": 24, + "gate_up_shape": [ + 37810, + 28672 + ], + "activation_qdata_shape": [ + 37824, + 7168 + ], + "weight_qdata_shape": [ + 5376, + 7168 + ], + "logical_mnk": [ + 37810, + 5376, + 14336 + ], + "descriptor_mnk_after_padding": [ + 37824, + 5376, + 14336 + ], + "producer": "vortex_native_quantize_swiglu_nvfp4", + "no_bias": true + }, + "baseline_kernel_metadata": { + "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", + "descriptors": { + "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", + "block_scale_mode": "VEC16_UE4M3", + "compute_and_scale": "FP32", + "scalar_pointer_mode": "device", + "bias": null, + "beta": 0.0, + "comfy_kitchen_version": "0.2.31" + }, + "profiler_cuda_events_available": false, + "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", + "top_cuda_events": [], + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + "heuristics": [ + { + "max_workspace_bytes": 0, + "requested_count": 32, + "returned_count": 5, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 4194304, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 8388608, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 16777216, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 33554432, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 67108864, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + } + ], + "explicit_split_k_checks": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 1, + "reduction_scheme": 0 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 4 + } + } + ], + "selected": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 + }, + "profile": { + "target": "baseline", + "selected_config": null, + "supplied_workspace_bytes": 33554432, + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + "errors_and_unsupported": [ + { + "feature": "Stream-K", + "supported_public_control": false, + "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + } + ] +} diff --git a/benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json b/benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json new file mode 100644 index 0000000..36ad8cf --- /dev/null +++ b/benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json @@ -0,0 +1,2014 @@ +{ + "mode": "block-gate", + "environment": { + "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", + "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", + "torch": "2.9.1+cu130", + "torch_cuda": "13.0", + "device": "NVIDIA GB10", + "device_capability": [ + 12, + 1 + ], + "driver": null, + "git_commit": null, + "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", + "checkpoint_sha256": null, + "checkpoint_hash_note": "not calculated", + "environment_switches": { + "CUDA_DEVICE_MAX_CONNECTIONS": "1", + "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", + "CUDA_HOME": "/usr/local/cuda", + "CUDA_INC_PATH": "/usr/local/cuda/include", + "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", + "CUDA_MODULE_LOADING": "EAGER", + "CUDA_VERSION": "13.0.2", + "H3_FUSED_ELEMENTWISE": "1", + "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd", + "TORCH_COMPILE_DISABLE": "0", + "TORCH_CUDA_ARCH_LIST": "12.1a", + "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" + }, + "extension": { + "cuda_version": 13000, + "cublas_version": 130100, + "stream_k_public_control": false, + "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + }, + "comfy_kitchen": "0.2.31 package without __version__" + }, + "workload": { + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "steps": 12, + "sampler_step": 1, + "seed": 440420, + "text_tokens": 100, + "tokens": 37810, + "hidden_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ] + }, + "retained_blocks": [ + 0, + 24, + 49 + ], + "immutable_cloned_block_inputs": { + "0": [ + 37810, + 5376 + ], + "24": [ + 37810, + 5376 + ], + "49": [ + 37810, + 5376 + ] + }, + "fc2_boundary": { + "block": 24, + "gate_up_shape": [ + 37810, + 28672 + ], + "activation_qdata_shape": [ + 37824, + 7168 + ], + "weight_qdata_shape": [ + 5376, + 7168 + ], + "logical_mnk": [ + 37810, + 5376, + 14336 + ], + "descriptor_mnk_after_padding": [ + 37824, + 5376, + 14336 + ], + "producer": "vortex_native_quantize_swiglu_nvfp4", + "no_bias": true + }, + "baseline_kernel_metadata": { + "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", + "descriptors": { + "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", + "block_scale_mode": "VEC16_UE4M3", + "compute_and_scale": "FP32", + "scalar_pointer_mode": "device", + "bias": null, + "beta": 0.0, + "comfy_kitchen_version": "0.2.31" + }, + "profiler_cuda_events_available": false, + "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", + "top_cuda_events": [], + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + "heuristics": [ + { + "max_workspace_bytes": 0, + "requested_count": 32, + "returned_count": 5, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 4194304, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 8388608, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 16777216, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 33554432, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 67108864, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + } + ], + "explicit_split_k_checks": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 1, + "reduction_scheme": 0 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 4 + } + } + ], + "selected": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 + }, + "block_gate": [ + { + "block": 0, + "only_monkeypatched_method": "block.mlp.fc2.forward_swiglu", + "accepted_gate_and_residual_path_preserved": true, + "candidate_vs_baseline": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", + "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" + }, + "baseline_vs_traversal": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", + "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" + }, + "candidate_vs_traversal": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", + "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" + }, + "timing": { + "order": "AB/BA alternates by round", + "baseline": { + "samples_ms": [ + 457.58404541015625, + 455.8675537109375, + 455.042724609375, + 456.5163879394531, + 456.7074279785156, + 459.7113037109375, + 456.3467102050781, + 459.0717468261719, + 457.6736755371094, + 458.6617736816406, + 458.61083984375, + 456.4378662109375, + 457.4807434082031, + 458.178466796875, + 457.4196472167969, + 458.90869140625, + 458.4776306152344, + 458.2261962890625, + 458.10919189453125, + 460.418701171875 + ], + "p50_ms": 457.8914337158203, + "p95_ms": 459.7466735839844, + "mean_ms": 457.7725662231445 + }, + "candidate": { + "samples_ms": [ + 428.5652160644531, + 425.1132507324219, + 420.88983154296875, + 420.71435546875, + 419.1617736816406, + 421.9552307128906, + 419.37896728515625, + 419.3441467285156, + 419.79351806640625, + 419.0124206542969, + 420.5487976074219, + 420.9779052734375, + 420.5229187011719, + 420.88104248046875, + 420.3728942871094, + 420.11834716796875, + 420.3209533691406, + 423.73919677734375, + 419.7261657714844, + 423.4720458984375 + ], + "p50_ms": 420.5358581542969, + "p95_ms": 425.2858489990234, + "mean_ms": 421.2304489135742 + }, + "parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", + "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" + } + }, + "supplied_workspace_bytes": 0 + }, + { + "block": 24, + "only_monkeypatched_method": "block.mlp.fc2.forward_swiglu", + "accepted_gate_and_residual_path_preserved": true, + "candidate_vs_baseline": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", + "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" + }, + "baseline_vs_traversal": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", + "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" + }, + "candidate_vs_traversal": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", + "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" + }, + "timing": { + "order": "AB/BA alternates by round", + "baseline": { + "samples_ms": [ + 458.4346008300781, + 459.98040771484375, + 457.2791748046875, + 456.69989013671875, + 458.1690368652344, + 459.9689025878906, + 459.0711975097656, + 461.275390625, + 458.7782897949219, + 460.40740966796875, + 459.8872985839844, + 461.1615905761719, + 459.39276123046875, + 459.6435546875, + 460.915283203125, + 461.53076171875, + 460.4534912109375, + 460.2518615722656, + 458.4383544921875, + 461.34747314453125 + ], + "p50_ms": 459.9281005859375, + "p95_ms": 461.3566375732422, + "mean_ms": 459.6543365478516 + }, + "candidate": { + "samples_ms": [ + 418.2376708984375, + 421.28668212890625, + 417.4606018066406, + 420.7315368652344, + 421.40478515625, + 419.4402770996094, + 417.5429382324219, + 419.78125, + 419.68603515625, + 421.91644287109375, + 422.23394775390625, + 420.081298828125, + 419.1646728515625, + 421.63360595703125, + 422.85125732421875, + 418.6084899902344, + 420.3404235839844, + 421.3265075683594, + 422.9977111816406, + 420.1449279785156 + ], + "p50_ms": 420.24267578125, + "p95_ms": 422.8585800170899, + "mean_ms": 420.3435531616211 + }, + "parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", + "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" + } + }, + "supplied_workspace_bytes": 0 + }, + { + "block": 49, + "only_monkeypatched_method": "block.mlp.fc2.forward_swiglu", + "accepted_gate_and_residual_path_preserved": true, + "candidate_vs_baseline": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", + "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" + }, + "baseline_vs_traversal": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", + "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" + }, + "candidate_vs_traversal": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", + "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" + }, + "timing": { + "order": "AB/BA alternates by round", + "baseline": { + "samples_ms": [ + 455.2886962890625, + 456.0704650878906, + 455.62127685546875, + 459.4961242675781, + 456.5694580078125, + 457.70477294921875, + 458.9316101074219, + 458.7506103515625, + 457.3757629394531, + 456.0811462402344, + 458.1807861328125, + 457.0354919433594, + 459.8126220703125, + 457.84747314453125, + 458.5899963378906, + 459.2362060546875, + 460.5073547363281, + 459.9343566894531, + 457.9271240234375, + 458.1864318847656 + ], + "p50_ms": 458.053955078125, + "p95_ms": 459.9630065917969, + "mean_ms": 457.95738830566404 + }, + "candidate": { + "samples_ms": [ + 414.79840087890625, + 421.3484191894531, + 416.3827209472656, + 416.5849914550781, + 419.1302490234375, + 414.7456970214844, + 419.312255859375, + 415.2906494140625, + 416.9346008300781, + 416.1675720214844, + 418.51458740234375, + 417.76995849609375, + 420.13006591796875, + 418.7623596191406, + 416.8761901855469, + 418.8280029296875, + 417.02777099609375, + 417.6993408203125, + 416.9612121582031, + 419.1832580566406 + ], + "p50_ms": 417.3635559082031, + "p95_ms": 420.190983581543, + "mean_ms": 417.62241516113284 + }, + "parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", + "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" + } + }, + "supplied_workspace_bytes": 0 + } + ], + "errors_and_unsupported": [ + { + "feature": "Stream-K", + "supported_public_control": false, + "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + } + ] +} diff --git a/benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json b/benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json new file mode 100644 index 0000000..163a75b --- /dev/null +++ b/benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json @@ -0,0 +1,2133 @@ +{ + "mode": "sweep", + "environment": { + "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", + "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", + "torch": "2.9.1+cu130", + "torch_cuda": "13.0", + "device": "NVIDIA GB10", + "device_capability": [ + 12, + 1 + ], + "driver": null, + "git_commit": null, + "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", + "checkpoint_sha256": null, + "checkpoint_hash_note": "not calculated", + "environment_switches": { + "CUDA_DEVICE_MAX_CONNECTIONS": "1", + "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", + "CUDA_HOME": "/usr/local/cuda", + "CUDA_INC_PATH": "/usr/local/cuda/include", + "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", + "CUDA_MODULE_LOADING": "EAGER", + "CUDA_VERSION": "13.0.2", + "H3_FUSED_ELEMENTWISE": "1", + "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd", + "TORCH_COMPILE_DISABLE": "0", + "TORCH_CUDA_ARCH_LIST": "12.1a", + "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" + }, + "extension": { + "cuda_version": 13000, + "cublas_version": 130100, + "stream_k_public_control": false, + "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + }, + "comfy_kitchen": "0.2.31 package without __version__" + }, + "workload": { + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "steps": 12, + "sampler_step": 1, + "seed": 440420, + "text_tokens": 100, + "tokens": 37810, + "hidden_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ] + }, + "retained_blocks": [ + 0, + 24, + 49 + ], + "immutable_cloned_block_inputs": { + "0": [ + 37810, + 5376 + ], + "24": [ + 37810, + 5376 + ], + "49": [ + 37810, + 5376 + ] + }, + "fc2_boundary": { + "block": 24, + "gate_up_shape": [ + 37810, + 28672 + ], + "activation_qdata_shape": [ + 37824, + 7168 + ], + "weight_qdata_shape": [ + 5376, + 7168 + ], + "logical_mnk": [ + 37810, + 5376, + 14336 + ], + "descriptor_mnk_after_padding": [ + 37824, + 5376, + 14336 + ], + "producer": "vortex_native_quantize_swiglu_nvfp4", + "no_bias": true + }, + "baseline_kernel_metadata": { + "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", + "descriptors": { + "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", + "block_scale_mode": "VEC16_UE4M3", + "compute_and_scale": "FP32", + "scalar_pointer_mode": "device", + "bias": null, + "beta": 0.0, + "comfy_kitchen_version": "0.2.31" + }, + "profiler_cuda_events_available": false, + "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", + "top_cuda_events": [], + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + "heuristics": [ + { + "max_workspace_bytes": 0, + "requested_count": 64, + "returned_count": 5, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 4194304, + "requested_count": 64, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 8388608, + "requested_count": 64, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 16777216, + "requested_count": 64, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 33554432, + "requested_count": 64, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 67108864, + "requested_count": 64, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + } + ], + "explicit_split_k_checks": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 1, + "reduction_scheme": 0 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 4 + } + } + ], + "selected": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 1, + "reduction_scheme": 0 + }, + "candidates": [ + { + "selected_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + "supplied_workspace_bytes": 67108864, + "required_workspace_bytes": 0, + "checked": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + "fc2_only": { + "order": "AB/BA alternates by round", + "baseline": { + "samples_ms": [ + 52.528385162353516, + 53.54975891113281, + 52.53424072265625, + 52.453086853027344, + 52.415489196777344, + 53.50998306274414, + 52.42780685424805, + 52.526817321777344, + 52.41756820678711, + 52.83504104614258, + 54.08259201049805, + 52.47983932495117, + 52.43289566040039, + 52.59030532836914, + 52.4370231628418, + 52.52783966064453, + 52.4318733215332, + 52.5145263671875, + 52.424705505371094, + 52.57904052734375 + ], + "p50_ms": 52.52067184448242, + "p95_ms": 53.57640056610108, + "mean_ms": 52.68494091033936, + "dense_tflop_s_p50": 110.9669507956279 + }, + "candidate": { + "samples_ms": [ + 52.557823181152344, + 53.029598236083984, + 52.549888610839844, + 52.55766296386719, + 52.51689529418945, + 53.11779022216797, + 54.02726364135742, + 52.74710464477539, + 52.530174255371094, + 52.540096282958984, + 52.56089782714844, + 52.54924774169922, + 52.527103424072266, + 52.53104019165039, + 53.064640045166016, + 53.72809600830078, + 53.41603088378906, + 52.76732635498047, + 52.54451370239258, + 53.056480407714844 + ], + "p50_ms": 52.55936050415039, + "p95_ms": 53.74305438995361, + "mean_ms": 52.84598369598389, + "dense_tflop_s_p50": 110.88526862612385 + }, + "parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", + "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + } + }, + "accepted_producer_plus_fc2": { + "order": "AB/BA alternates by round", + "baseline": { + "samples_ms": [ + 75.45340728759766, + 75.004638671875, + 76.25904083251953, + 75.5995864868164, + 75.02406311035156, + 75.02108764648438, + 74.97523498535156, + 76.13423919677734, + 76.12108612060547, + 76.02950286865234, + 76.11692810058594, + 75.5208969116211, + 75.1124496459961, + 75.58025360107422, + 74.99574279785156, + 76.13922882080078, + 76.13350677490234, + 76.16307067871094, + 76.70272064208984, + 75.0445785522461 + ], + "p50_ms": 75.58992004394531, + "p95_ms": 76.28122482299804, + "mean_ms": 75.65656318664551, + "dense_tflop_s_p50": 77.10100506696888 + }, + "candidate": { + "samples_ms": [ + 75.16223907470703, + 75.09487915039062, + 76.4610595703125, + 75.77107238769531, + 76.33715057373047, + 76.33900451660156, + 75.62342071533203, + 76.34333038330078, + 75.10733032226562, + 75.13775634765625, + 76.2429428100586, + 75.65090942382812, + 76.9865951538086, + 76.45164489746094, + 75.12268829345703, + 76.36873626708984, + 75.21382141113281, + 75.16233825683594, + 75.15340423583984, + 75.63337707519531 + ], + "p50_ms": 75.64214324951172, + "p95_ms": 76.48733634948731, + "mean_ms": 75.76818504333497, + "dense_tflop_s_p50": 77.04777466571349 + }, + "parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", + "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + } + }, + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + { + "selected_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + "supplied_workspace_bytes": 67108864, + "required_workspace_bytes": 0, + "checked": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + "fc2_only": { + "order": "AB/BA alternates by round", + "baseline": { + "samples_ms": [ + 53.98486328125, + 53.7977294921875, + 53.85007858276367, + 53.532447814941406, + 53.588897705078125, + 53.95724868774414, + 53.63395309448242, + 53.011390686035156, + 53.56748962402344, + 54.085472106933594, + 54.20851135253906, + 53.808990478515625, + 53.888832092285156, + 54.48992156982422, + 53.129215240478516, + 53.52640151977539, + 53.18348693847656, + 53.60192108154297, + 53.001953125, + 53.25020980834961 + ], + "p50_ms": 53.617937088012695, + "p95_ms": 54.22258186340332, + "mean_ms": 53.65495071411133, + "dense_tflop_s_p50": 108.69606562358724 + }, + "candidate": { + "samples_ms": [ + 15.689727783203125, + 15.966943740844727, + 15.744000434875488, + 15.626208305358887, + 15.685664176940918, + 15.68841552734375, + 15.331328392028809, + 15.71504020690918, + 14.792703628540039, + 15.66592025756836, + 14.791680335998535, + 15.645536422729492, + 14.812159538269043, + 14.813887596130371, + 14.793631553649902, + 15.76848030090332, + 15.557696342468262, + 15.463264465332031, + 14.7957763671875, + 15.723199844360352 + ], + "p50_ms": 15.63587236404419, + "p95_ms": 15.778403472900392, + "mean_ms": 15.403563261032104, + "dense_tflop_s_p50": 372.73640207770177 + }, + "parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", + "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + } + }, + "accepted_producer_plus_fc2": { + "order": "AB/BA alternates by round", + "baseline": { + "samples_ms": [ + 76.28396606445312, + 76.32662200927734, + 76.3351058959961, + 76.94000244140625, + 76.97782135009766, + 76.59494018554688, + 76.87884521484375, + 77.03215789794922, + 76.31446075439453, + 76.24060821533203, + 76.16102600097656, + 76.39116668701172, + 77.31603240966797, + 76.89494323730469, + 75.6058578491211, + 76.90121459960938, + 76.15897369384766, + 77.05766296386719, + 77.53421020507812, + 76.94409942626953 + ], + "p50_ms": 76.73689270019531, + "p95_ms": 77.32694129943847, + "mean_ms": 76.64448585510254, + "dense_tflop_s_p50": 75.94859008807853 + }, + "candidate": { + "samples_ms": [ + 38.138526916503906, + 39.03964614868164, + 39.00892639160156, + 38.78895950317383, + 38.25788879394531, + 39.111358642578125, + 37.374977111816406, + 39.55583953857422, + 38.676639556884766, + 39.315616607666016, + 39.011199951171875, + 38.85724639892578, + 38.27609634399414, + 39.127777099609375, + 38.3109130859375, + 39.1360969543457, + 38.23001480102539, + 39.10319900512695, + 38.683841705322266, + 38.4785270690918 + ], + "p50_ms": 38.823102951049805, + "p95_ms": 39.32762775421143, + "mean_ms": 38.72416458129883, + "dense_tflop_s_p50": 150.11831526368 + }, + "parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", + "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + } + }, + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + } + ], + "errors_and_unsupported": [ + { + "feature": "Stream-K", + "supported_public_control": false, + "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + } + ] +} diff --git a/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv b/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv new file mode 100644 index 0000000..dd317f9 --- /dev/null +++ b/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv @@ -0,0 +1,3 @@ +"ID","Process ID","Process Name","Host Name","Kernel Name","Context","Stream","Block Size","Grid Size","Device","CC","c2clink__enabled_mask","c2clink__present","derived__avg_thread_executed","derived__avg_thread_executed_true","derived__avg_thread_unexecuted_true","derived__derivative_avg_thread_executed_true","derived__l1tex__lsu_writeback_bytes_mem_lgds.sum.peak_sustained","derived__l1tex__lsu_writeback_bytes_mem_lgds.sum.per_second","derived__local_spilling_requests","derived__local_spilling_requests_pct","derived__lts__lts2xbar_bytes.sum.peak_sustained","derived__lts__lts2xbar_bytes.sum.per_second","derived__memory_l1_conflicts_shared_nway","derived__memory_l1_wavefronts_shared_excessive","derived__memory_l2_theoretical_sectors_global_excessive","derived__pct_occupancy_per_barrier_count","derived__pct_occupancy_per_block_size","derived__pct_occupancy_per_register_count","derived__pct_occupancy_per_shared_mem_size","derived__sm__sass_thread_inst_executed_op_dfma_pred_on_x2","derived__sm__sass_thread_inst_executed_op_ffma_pred_on_x2","derived__sm__sass_thread_inst_executed_op_hfma_pred_on_x4","derived__smsp__inst_executed_op_branch_pct","derived__smsp__sass_thread_inst_executed_op_dfma_pred_on_x2","derived__smsp__sass_thread_inst_executed_op_ffma_pred_on_x2","derived__smsp__sass_thread_inst_executed_op_hadd_pred_on_x2","derived__smsp__sass_thread_inst_executed_op_hfma_pred_on_x4","derived__smsp__sass_thread_inst_executed_op_hmul_pred_on_x2","derived_tempMetric0","derived_tempMetric1","derived_tempMetric2","derived_tempMetric3","derived_tempMetric4","derived_tempMetric5","device__attribute_architecture","device__attribute_async_engine_count","device__attribute_can_flush_remote_writes","device__attribute_can_map_host_memory","device__attribute_can_tex2d_gather","device__attribute_can_use_64_bit_stream_mem_ops","device__attribute_can_use_64_bit_stream_mem_ops_v1","device__attribute_can_use_host_pointer_for_registered_mem","device__attribute_can_use_stream_mem_ops_v1","device__attribute_can_use_stream_wait_value_nor","device__attribute_can_use_stream_wait_value_nor_v1","device__attribute_chip","device__attribute_clock_rate","device__attribute_cluster_launch","device__attribute_compute_capability_major","device__attribute_compute_capability_minor","device__attribute_compute_mode","device__attribute_compute_preemption_supported","device__attribute_concurrent_kernels","device__attribute_concurrent_managed_access","device__attribute_confidential_computing_mode","device__attribute_cooperative_launch","device__attribute_cooperative_multi_device_launch","device__attribute_deferred_mapping_cuda_array_supported","device__attribute_device_index","device__attribute_direct_managed_mem_access_from_host","device__attribute_display_name","device__attribute_dma_buf_supported","device__attribute_ecc_enabled","device__attribute_fb_bus_width","device__attribute_fbp_count","device__attribute_generic_compression_supported","device__attribute_global_l1_cache_supported","device__attribute_global_memory_bus_width","device__attribute_gpu_direct_rdma_flush_writes_options","device__attribute_gpu_direct_rdma_supported","device__attribute_gpu_direct_rdma_with_cuda_vmm_supported","device__attribute_gpu_direct_rdma_writes_ordering","device__attribute_gpu_overlap","device__attribute_gpu_pci_device_id","device__attribute_gpu_pci_ext_device_id","device__attribute_gpu_pci_ext_downstream_link_rate","device__attribute_gpu_pci_ext_downstream_link_width","device__attribute_gpu_pci_ext_gen","device__attribute_gpu_pci_ext_gpu_gen","device__attribute_gpu_pci_ext_gpu_link_rate","device__attribute_gpu_pci_ext_gpu_link_width","device__attribute_gpu_pci_revision_id","device__attribute_gpu_pci_sub_system_id","device__attribute_handle_type_fabric_supported","device__attribute_handle_type_posix_file_descriptor_supported","device__attribute_handle_type_win32_handle_supported","device__attribute_handle_type_win32_kmt_handle_supported","device__attribute_host_native_atomic_supported","device__attribute_host_numa_id","device__attribute_host_register_supported","device__attribute_implementation","device__attribute_integrated","device__attribute_ipc_event_supported","device__attribute_kernel_exec_timeout","device__attribute_l2_cache_size","device__attribute_l2s_count","device__attribute_limits_max_cta_per_sm","device__attribute_limits_num_tpcs","device__attribute_local_l1_cache_supported","device__attribute_managed_memory","device__attribute_max_access_policy_window_size","device__attribute_max_block_dim_x","device__attribute_max_block_dim_y","device__attribute_max_block_dim_z","device__attribute_max_blocks_per_multiprocessor","device__attribute_max_gpu_frequency_khz","device__attribute_max_grid_dim_x","device__attribute_max_grid_dim_y","device__attribute_max_grid_dim_z","device__attribute_max_ipc_per_multiprocessor","device__attribute_max_ipc_per_scheduler","device__attribute_max_mem_frequency_khz","device__attribute_max_persisting_l2_cache_size","device__attribute_max_pitch","device__attribute_max_registers_per_block","device__attribute_max_registers_per_multiprocessor","device__attribute_max_registers_per_thread","device__attribute_max_shared_memory_per_block","device__attribute_max_shared_memory_per_block_optin","device__attribute_max_shared_memory_per_multiprocessor","device__attribute_max_threads_per_block","device__attribute_max_threads_per_multiprocessor","device__attribute_max_warps_per_multiprocessor","device__attribute_max_warps_per_scheduler","device__attribute_maximum_surface1d_layered_layers","device__attribute_maximum_surface1d_layered_width","device__attribute_maximum_surface1d_width","device__attribute_maximum_surface2d_height","device__attribute_maximum_surface2d_layered_height","device__attribute_maximum_surface2d_layered_layers","device__attribute_maximum_surface2d_layered_width","device__attribute_maximum_surface2d_width","device__attribute_maximum_surface3d_depth","device__attribute_maximum_surface3d_height","device__attribute_maximum_surface3d_width","device__attribute_maximum_surfacecubemap_layered_layers","device__attribute_maximum_surfacecubemap_layered_width","device__attribute_maximum_surfacecubemap_width","device__attribute_maximum_texture1d_layered_layers","device__attribute_maximum_texture1d_layered_width","device__attribute_maximum_texture1d_linear_width","device__attribute_maximum_texture1d_mipmapped_width","device__attribute_maximum_texture1d_width","device__attribute_maximum_texture2d_gather_height","device__attribute_maximum_texture2d_gather_width","device__attribute_maximum_texture2d_height","device__attribute_maximum_texture2d_layered_height","device__attribute_maximum_texture2d_layered_layers","device__attribute_maximum_texture2d_layered_width","device__attribute_maximum_texture2d_linear_height","device__attribute_maximum_texture2d_linear_pitch","device__attribute_maximum_texture2d_linear_width","device__attribute_maximum_texture2d_mipmapped_height","device__attribute_maximum_texture2d_mipmapped_width","device__attribute_maximum_texture2d_width","device__attribute_maximum_texture3d_depth","device__attribute_maximum_texture3d_depth_alternate","device__attribute_maximum_texture3d_height","device__attribute_maximum_texture3d_height_alternate","device__attribute_maximum_texture3d_width","device__attribute_maximum_texture3d_width_alternate","device__attribute_maximum_texturecubemap_layered_layers","device__attribute_maximum_texturecubemap_layered_width","device__attribute_maximum_texturecubemap_width","device__attribute_mem_sync_domain_count","device__attribute_memory_clock_rate","device__attribute_memory_pools_supported","device__attribute_mempool_supported_handle_types","device__attribute_mps_enabled","device__attribute_multi_gpu_board","device__attribute_multi_gpu_board_group_id","device__attribute_multicast_supported","device__attribute_multiprocessor_count","device__attribute_num_l2s_per_fbp","device__attribute_num_schedulers_per_multiprocessor","device__attribute_num_tex_per_multiprocessor","device__attribute_numa_config","device__attribute_pageable_memory_access","device__attribute_pageable_memory_access_uses_host_page_tables","device__attribute_pci_bus_id","device__attribute_pci_device_id","device__attribute_pci_domain_id","device__attribute_ram_location","device__attribute_ram_type","device__attribute_reserved_shared_memory_per_block","device__attribute_sass_level","device__attribute_single_to_double_precision_perf_ratio","device__attribute_sparse_cuda_array_supported","device__attribute_stream_priorities_supported","device__attribute_surface_alignment","device__attribute_tcc_driver","device__attribute_tensor_map_access_supported","device__attribute_texture_alignment","device__attribute_texture_pitch_alignment","device__attribute_total_constant_memory","device__attribute_total_memory","device__attribute_unified_addressing","device__attribute_unified_function_pointers","device__attribute_virtual_address_management_supported","device__attribute_warp_size","gcc__cache_requests_type_constant.sum","gcc__cache_requests_type_constant.sum.pct_of_peak_sustained_elapsed","gcc__cache_requests_type_instruction.sum","gcc__cache_requests_type_instruction.sum.pct_of_peak_sustained_elapsed","gcc__xbar2gcc_sectors.sum","gcc__xbar2gcc_sectors.sum.pct_of_peak_sustained_elapsed","gpc__cycles_elapsed.avg","gpc__cycles_elapsed.avg.per_second","gpc__cycles_elapsed.max","gpc__cycles_elapsed.max.per_second","gpc__cycles_elapsed.min","gpc__cycles_elapsed.min.per_second","gpc__cycles_elapsed.sum","gpc__cycles_elapsed.sum.per_second","gpu__compute_memory_access_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_access_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_request_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.avg.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.max.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.min.pct_of_peak_sustained_elapsed","gpu__compute_memory_throughput.sum.pct_of_peak_sustained_elapsed","gpu__time_duration.avg","gpu__time_duration.max","gpu__time_duration.min","gpu__time_duration.sum","gr__workids_granted.avg","gr__workids_granted.max","gr__workids_granted.min","gr__workids_granted.sum","gr__workids_granted_as_ctas.avg","gr__workids_granted_as_ctas.max","gr__workids_granted_as_ctas.min","gr__workids_granted_as_ctas.sum","gr__workids_requested.avg","gr__workids_requested.max","gr__workids_requested.min","gr__workids_requested.sum","idc__request_cycles_active.avg.pct_of_peak_sustained_elapsed","idc__request_cycles_active.max.pct_of_peak_sustained_elapsed","idc__request_cycles_active.min.pct_of_peak_sustained_elapsed","idc__request_cycles_active.sum.pct_of_peak_sustained_elapsed","idc__request_hit_rate.pct","idc__requests.sum","idc__requests.sum.pct_of_peak_sustained_elapsed","inst_executed","l1tex__cycles_active.avg","l1tex__cycles_active.max","l1tex__cycles_active.min","l1tex__cycles_active.sum","l1tex__cycles_elapsed.avg","l1tex__cycles_elapsed.avg.per_second","l1tex__cycles_elapsed.max","l1tex__cycles_elapsed.max.per_second","l1tex__cycles_elapsed.min","l1tex__cycles_elapsed.min.per_second","l1tex__cycles_elapsed.sum","l1tex__cycles_elapsed.sum.per_second","l1tex__data_bank_conflicts_pipe_lsu_mem_shared.sum","l1tex__data_bank_conflicts_pipe_lsu_mem_shared_op_atom.sum","l1tex__data_bank_conflicts_pipe_lsu_mem_shared_op_ld.sum","l1tex__data_bank_conflicts_pipe_lsu_mem_shared_op_ldgsts.sum","l1tex__data_bank_conflicts_pipe_lsu_mem_shared_op_st.sum","l1tex__data_bank_reads.avg.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.max.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.min.pct_of_peak_sustained_elapsed","l1tex__data_bank_reads.sum.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.avg.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.max.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.min.pct_of_peak_sustained_elapsed","l1tex__data_bank_writes.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts_mem_shared.sum","l1tex__data_pipe_lsu_wavefronts_mem_shared.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_atom.sum","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_ld.sum","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_st.sum","l1tex__data_pipe_lsu_wavefronts_mem_shared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__data_pipe_tex_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.avg.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.max.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.min.pct_of_peak_sustained_elapsed","l1tex__f_wavefronts.sum.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.avg.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.max.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.min.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active.sum.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active_mem_lgds.sum","l1tex__lsu_writeback_active_mem_lgds.sum.pct_of_peak_sustained_elapsed","l1tex__lsu_writeback_active_mem_lgds.sum.peak_sustained","l1tex__lsu_writeback_active_mem_lgds.sum.per_second","l1tex__lsuin_requests.avg.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.max.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.min.pct_of_peak_sustained_elapsed","l1tex__lsuin_requests.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.avg.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.max.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.min.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_req_cycles_active.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes.sum","l1tex__m_l1tex2xbar_write_bytes.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes.sum.per_second","l1tex__m_l1tex2xbar_write_bytes_mem_dshared.sum","l1tex__m_l1tex2xbar_write_bytes_mem_dshared.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes_mem_dshared.sum.per_second","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_red.sum","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_red.sum.per_second","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_st.sum","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_bytes_mem_global_op_tma_st.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_atom.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_atom.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_ld.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_ld.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_redas.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_redas.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_redas.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_st.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_tma_red.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_tma_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_tma_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_dshared_op_tma_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_atom.sum","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_red.sum","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_red.sum","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_red.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_global_op_tma_st.sum.per_second","l1tex__m_l1tex2xbar_write_sectors_mem_lg_op_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_lg_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_atom.sum","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_red.sum","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_st.sum","l1tex__m_l1tex2xbar_write_sectors_mem_surface_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_bytes.sum","l1tex__m_xbar2l1tex_read_bytes.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_bytes.sum.per_second","l1tex__m_xbar2l1tex_read_bytes_mem_dshared.sum","l1tex__m_xbar2l1tex_read_bytes_mem_dshared.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_bytes_mem_dshared.sum.per_second","l1tex__m_xbar2l1tex_read_bytes_mem_global_op_tma_ld.sum","l1tex__m_xbar2l1tex_read_bytes_mem_global_op_tma_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_bytes_mem_global_op_tma_ld.sum.per_second","l1tex__m_xbar2l1tex_read_sectors.avg.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.max.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.min.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_atom.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_atom.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_ld.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_ld.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_redas.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_redas.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_redas.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_st.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_st.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_tma_red.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_tma_red.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_tma_st.sum","l1tex__m_xbar2l1tex_read_sectors_mem_dshared_op_tma_st.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_atom.sum","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_tma_ld.sum","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_tma_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_global_op_tma_ld.sum.per_second","l1tex__m_xbar2l1tex_read_sectors_mem_lg_op_ld.sum","l1tex__m_xbar2l1tex_read_sectors_mem_lg_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_surface_op_atom.sum","l1tex__m_xbar2l1tex_read_sectors_mem_surface_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_surface_op_ld.sum","l1tex__m_xbar2l1tex_read_sectors_mem_surface_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__m_xbar2l1tex_read_sectors_mem_texture.sum","l1tex__m_xbar2l1tex_read_sectors_mem_texture.sum.pct_of_peak_sustained_elapsed","l1tex__t_bytes_pipe_lsu_mem_global_op_ldgsts_cache_access.sum","l1tex__t_bytes_pipe_lsu_mem_global_op_ldgsts_cache_access.sum.pct_of_peak_sustained_elapsed","l1tex__t_bytes_pipe_lsu_mem_global_op_ldgsts_cache_access.sum.per_second","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_atom.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_ld.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_redas.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_redas.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_st.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_dshared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_atom.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_ld.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_red.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_st.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_global_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_local_op_ld.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_local_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_lsu_mem_local_op_st.sum","l1tex__t_output_wavefronts_pipe_lsu_mem_local_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_atom.sum","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_ld.sum","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_red.sum","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_st.sum","l1tex__t_output_wavefronts_pipe_tex_mem_surface_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_output_wavefronts_pipe_tex_mem_texture.sum","l1tex__t_output_wavefronts_pipe_tex_mem_texture.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_dshared_op_atom.sum","l1tex__t_requests_pipe_lsu_mem_dshared_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_dshared_op_ld.sum","l1tex__t_requests_pipe_lsu_mem_dshared_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_dshared_op_redas.sum","l1tex__t_requests_pipe_lsu_mem_dshared_op_redas.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_dshared_op_st.sum","l1tex__t_requests_pipe_lsu_mem_dshared_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_global_op_atom.sum","l1tex__t_requests_pipe_lsu_mem_global_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_global_op_ld.sum","l1tex__t_requests_pipe_lsu_mem_global_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_global_op_red.sum","l1tex__t_requests_pipe_lsu_mem_global_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_global_op_st.sum","l1tex__t_requests_pipe_lsu_mem_global_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_local_op_ld.sum","l1tex__t_requests_pipe_lsu_mem_local_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_lsu_mem_local_op_st.sum","l1tex__t_requests_pipe_lsu_mem_local_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_surface_op_atom.sum","l1tex__t_requests_pipe_tex_mem_surface_op_atom.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_surface_op_ld.sum","l1tex__t_requests_pipe_tex_mem_surface_op_ld.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_surface_op_red.sum","l1tex__t_requests_pipe_tex_mem_surface_op_red.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_surface_op_st.sum","l1tex__t_requests_pipe_tex_mem_surface_op_st.sum.pct_of_peak_sustained_elapsed","l1tex__t_requests_pipe_tex_mem_texture.sum","l1tex__t_requests_pipe_tex_mem_texture.sum.pct_of_peak_sustained_elapsed","l1tex__t_sector_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_global_op_atom_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_global_op_ld_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_global_op_red_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_global_op_st_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_local_op_ld_hit_rate.pct","l1tex__t_sector_pipe_lsu_mem_local_op_st_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_surface_op_atom_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_surface_op_ld_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_surface_op_red_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_surface_op_st_hit_rate.pct","l1tex__t_sector_pipe_tex_mem_texture_op_tex_hit_rate.pct","l1tex__t_sectors_pipe_lsu_mem_dshared_op_atom.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_atom_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_ld.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_ld_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_redas.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_redas_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_st.sum","l1tex__t_sectors_pipe_lsu_mem_dshared_op_st_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_atom.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_atom_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_atom_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_ld.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_ld_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_ld_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_red.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_red_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_red_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_st.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_st_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_global_op_st_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_ld.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_ld_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_ld_lookup_miss.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_st.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_st_lookup_hit.sum","l1tex__t_sectors_pipe_lsu_mem_local_op_st_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_atom.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_atom_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_atom_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_ld.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_ld_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_ld_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_red.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_red_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_red_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_st.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_st_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_surface_op_st_lookup_miss.sum","l1tex__t_sectors_pipe_tex_mem_texture.sum","l1tex__t_sectors_pipe_tex_mem_texture_lookup_hit.sum","l1tex__t_sectors_pipe_tex_mem_texture_lookup_miss.sum","l1tex__tex_writeback_active.avg.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.max.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.min.pct_of_peak_sustained_elapsed","l1tex__tex_writeback_active.sum","l1tex__tex_writeback_active.sum.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.avg.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.max.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.min.pct_of_peak_sustained_elapsed","l1tex__texin_sm2tex_req_cycles_active.sum.pct_of_peak_sustained_elapsed","l1tex__throughput.avg.pct_of_peak_sustained_active","l1tex__throughput.avg.pct_of_peak_sustained_elapsed","l1tex__throughput.max.pct_of_peak_sustained_active","l1tex__throughput.max.pct_of_peak_sustained_elapsed","l1tex__throughput.min.pct_of_peak_sustained_active","l1tex__throughput.min.pct_of_peak_sustained_elapsed","l1tex__throughput.sum.pct_of_peak_sustained_active","l1tex__throughput.sum.pct_of_peak_sustained_elapsed","launch__barrier_count","launch__block_dim_x","launch__block_dim_y","launch__block_dim_z","launch__block_size","launch__cluster_dim_x","launch__cluster_dim_y","launch__cluster_dim_z","launch__cluster_max_active","launch__cluster_max_potential_size","launch__cluster_scheduling_policy","launch__cluster_size","launch__context_id","launch__device_id","launch__func_cache_config","launch__function_pcs","launch__grid_dim_x","launch__grid_dim_y","launch__grid_dim_z","launch__grid_size","launch__kernel_name","launch__occupancy_cluster_gpu_pct","launch__occupancy_cluster_pct","launch__occupancy_limit_barriers","launch__occupancy_limit_blocks","launch__occupancy_limit_registers","launch__occupancy_limit_shared_mem","launch__occupancy_limit_warps","launch__occupancy_per_barrier_count","launch__occupancy_per_block_size","launch__occupancy_per_cluster_size","launch__occupancy_per_register_count","launch__occupancy_per_shared_mem_size","launch__persisting_l2_cache_size","launch__preferred_cluster_dim_x","launch__preferred_cluster_dim_y","launch__preferred_cluster_dim_z","launch__preferred_cluster_size","launch__registers_per_thread","launch__registers_per_thread_allocated","launch__shared_mem_config_size","launch__shared_mem_per_block","launch__shared_mem_per_block_allocated","launch__shared_mem_per_block_driver","launch__shared_mem_per_block_dynamic","launch__shared_mem_per_block_static","launch__sm_count","launch__stack_size","launch__stream_id","launch__thread_count","launch__tpc_count","launch__tpc_enabled","launch__uses_cdp","launch__uses_green_context","launch__uses_mps","launch__uses_nvlink_centric_scheduling","launch__uses_vgpu","launch__waves_per_multiprocessor","lrc__average_ilc_input_sector_success_rate.pct","lrc__ilc_input_sectors.sum","lts__average_gcomp_input_sector_success_rate.pct","lts__average_gcomp_output_sector_compression_achieved_rate.ratio","lts__average_t_sectors_requested_srcunit_tex_op_read_vs_returned_sectors_realtime.ratio","lts__average_xcomp_gxc_sector_input_compression_rate.ratio","lts__cycles_active.avg","lts__cycles_active.max","lts__cycles_active.min","lts__cycles_active.sum","lts__cycles_elapsed.avg","lts__cycles_elapsed.avg.per_second","lts__cycles_elapsed.max","lts__cycles_elapsed.max.per_second","lts__cycles_elapsed.min","lts__cycles_elapsed.min.per_second","lts__cycles_elapsed.sum","lts__cycles_elapsed.sum.per_second","lts__d_atomic_input_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.max.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.min.pct_of_peak_sustained_elapsed","lts__d_atomic_input_cycles_active.sum.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.avg.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.max.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.min.pct_of_peak_sustained_elapsed","lts__d_decomp_input_sectors.sum.pct_of_peak_sustained_elapsed","lts__d_sectors.avg.pct_of_peak_sustained_elapsed","lts__d_sectors.max.pct_of_peak_sustained_elapsed","lts__d_sectors.min.pct_of_peak_sustained_elapsed","lts__d_sectors.sum.pct_of_peak_sustained_elapsed","lts__d_sectors_fill_sysmem.sum","lts__d_sectors_fill_sysmem.sum.pct_of_peak_sustained_elapsed","lts__d_sectors_fill_sysmem.sum.per_second","lts__gcomp_input_sectors.avg","lts__gcomp_input_sectors.max","lts__gcomp_input_sectors.min","lts__gcomp_input_sectors.sum","lts__lts2xbar_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.avg.peak_sustained","lts__lts2xbar_cycles_active.avg.per_second","lts__lts2xbar_cycles_active.max.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.max.peak_sustained","lts__lts2xbar_cycles_active.max.per_second","lts__lts2xbar_cycles_active.min.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.min.peak_sustained","lts__lts2xbar_cycles_active.min.per_second","lts__lts2xbar_cycles_active.sum.pct_of_peak_sustained_elapsed","lts__lts2xbar_cycles_active.sum.peak_sustained","lts__lts2xbar_cycles_active.sum.per_second","lts__t_requests.sum","lts__t_requests_srcunit_gcc.sum","lts__t_requests_srcunit_tex.sum","lts__t_requests_srcunit_tex_op_atom_dot_alu.sum","lts__t_requests_srcunit_tex_op_atom_dot_cas.sum","lts__t_requests_srcunit_tex_op_read.sum","lts__t_requests_srcunit_tex_op_red.sum","lts__t_requests_srcunit_tex_op_write.sum","lts__t_sector_hit_rate.pct","lts__t_sector_op_read_hit_rate.pct","lts__t_sector_op_write_hit_rate.pct","lts__t_sectors.avg","lts__t_sectors.avg.pct_of_peak_sustained_elapsed","lts__t_sectors.avg.peak_sustained","lts__t_sectors.avg.per_cycle_elapsed","lts__t_sectors.max","lts__t_sectors.max.pct_of_peak_sustained_elapsed","lts__t_sectors.min","lts__t_sectors.min.pct_of_peak_sustained_elapsed","lts__t_sectors.sum","lts__t_sectors.sum.pct_of_peak_sustained_elapsed","lts__t_sectors.sum.per_second","lts__t_sectors_aperture_sysmem_op_write.sum","lts__t_sectors_aperture_sysmem_op_write.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_aperture_sysmem_op_write.sum.per_second","lts__t_sectors_data_ecc.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_data_ecc.avg.peak_sustained","lts__t_sectors_data_ecc.avg.per_cycle_elapsed","lts__t_sectors_data_ecc.sum","lts__t_sectors_data_ecc.sum.per_second","lts__t_sectors_evict_first_lookup_hit.sum","lts__t_sectors_evict_first_lookup_miss.sum","lts__t_sectors_evict_last_lookup_hit.sum","lts__t_sectors_evict_last_lookup_miss.sum","lts__t_sectors_evict_normal_demote_lookup_hit.sum","lts__t_sectors_evict_normal_demote_lookup_miss.sum","lts__t_sectors_evict_normal_lookup_hit.sum","lts__t_sectors_evict_normal_lookup_miss.sum","lts__t_sectors_lookup_hit.sum","lts__t_sectors_lookup_miss.sum","lts__t_sectors_requested_srcunit_tex_op_read_realtime.sum","lts__t_sectors_requested_srcunit_tex_op_read_realtime.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_requested_srcunit_tex_op_read_realtime.sum.per_second","lts__t_sectors_srcunit_gcc.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_gcc.avg.peak_sustained","lts__t_sectors_srcunit_gcc.avg.per_cycle_elapsed","lts__t_sectors_srcunit_gcc.sum","lts__t_sectors_srcunit_gcc.sum.per_second","lts__t_sectors_srcunit_gcc_lookup_hit.sum","lts__t_sectors_srcunit_gcc_lookup_miss.sum","lts__t_sectors_srcunit_tex.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex.avg.peak_sustained","lts__t_sectors_srcunit_tex.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex.max.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex.min.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex.sum","lts__t_sectors_srcunit_tex.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex.sum.per_second","lts__t_sectors_srcunit_tex_aperture_peer_lookup_miss.avg","lts__t_sectors_srcunit_tex_aperture_peer_lookup_miss.max","lts__t_sectors_srcunit_tex_aperture_peer_lookup_miss.min","lts__t_sectors_srcunit_tex_aperture_peer_lookup_miss.sum","lts__t_sectors_srcunit_tex_aperture_sysmem_lookup_miss.avg","lts__t_sectors_srcunit_tex_aperture_sysmem_lookup_miss.max","lts__t_sectors_srcunit_tex_aperture_sysmem_lookup_miss.min","lts__t_sectors_srcunit_tex_aperture_sysmem_lookup_miss.sum","lts__t_sectors_srcunit_tex_evict_first_lookup_hit.sum","lts__t_sectors_srcunit_tex_evict_first_lookup_miss.sum","lts__t_sectors_srcunit_tex_evict_last_lookup_hit.sum","lts__t_sectors_srcunit_tex_evict_last_lookup_miss.sum","lts__t_sectors_srcunit_tex_evict_normal_demote_lookup_hit.sum","lts__t_sectors_srcunit_tex_evict_normal_demote_lookup_miss.sum","lts__t_sectors_srcunit_tex_evict_normal_lookup_hit.sum","lts__t_sectors_srcunit_tex_evict_normal_lookup_miss.sum","lts__t_sectors_srcunit_tex_lookup_hit.sum","lts__t_sectors_srcunit_tex_lookup_miss.avg","lts__t_sectors_srcunit_tex_lookup_miss.max","lts__t_sectors_srcunit_tex_lookup_miss.min","lts__t_sectors_srcunit_tex_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom.sum","lts__t_sectors_srcunit_tex_op_atom.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_atom.sum.per_second","lts__t_sectors_srcunit_tex_op_atom_dot_alu.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_atom_dot_alu.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_atom_dot_alu.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_atom_dot_alu.sum","lts__t_sectors_srcunit_tex_op_atom_dot_alu.sum.per_second","lts__t_sectors_srcunit_tex_op_atom_dot_alu_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_dot_alu_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom_dot_cas.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_atom_dot_cas.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_atom_dot_cas.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_atom_dot_cas.sum","lts__t_sectors_srcunit_tex_op_atom_dot_cas.sum.per_second","lts__t_sectors_srcunit_tex_op_atom_dot_cas_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_dot_cas_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom_evict_first_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_evict_first_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom_evict_last_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_evict_last_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_atom_evict_normal_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_atom_evict_normal_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_read.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_read.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_read.sum","lts__t_sectors_srcunit_tex_op_read.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_read.sum.per_second","lts__t_sectors_srcunit_tex_op_read_evict_first_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_evict_first_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read_evict_last_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_evict_last_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read_evict_normal_demote_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_evict_normal_demote_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read_evict_normal_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_evict_normal_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_read_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_read_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_red.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_red.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_red.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_red.sum","lts__t_sectors_srcunit_tex_op_red.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_red.sum.per_second","lts__t_sectors_srcunit_tex_op_red_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_red_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write.avg.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_write.avg.peak_sustained","lts__t_sectors_srcunit_tex_op_write.avg.per_cycle_elapsed","lts__t_sectors_srcunit_tex_op_write.sum","lts__t_sectors_srcunit_tex_op_write.sum.pct_of_peak_sustained_elapsed","lts__t_sectors_srcunit_tex_op_write.sum.per_second","lts__t_sectors_srcunit_tex_op_write_evict_first_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_evict_first_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write_evict_last_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_evict_last_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write_evict_normal_demote_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_evict_normal_demote_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write_evict_normal_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_evict_normal_lookup_miss.sum","lts__t_sectors_srcunit_tex_op_write_lookup_hit.sum","lts__t_sectors_srcunit_tex_op_write_lookup_miss.sum","lts__t_tag_requests.avg.pct_of_peak_sustained_elapsed","lts__t_tag_requests.max.pct_of_peak_sustained_elapsed","lts__t_tag_requests.min.pct_of_peak_sustained_elapsed","lts__t_tag_requests.sum.pct_of_peak_sustained_elapsed","lts__throughput.avg.pct_of_peak_sustained_elapsed","lts__throughput.max.pct_of_peak_sustained_elapsed","lts__throughput.min.pct_of_peak_sustained_elapsed","lts__throughput.sum.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.avg.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.max.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.min.pct_of_peak_sustained_elapsed","lts__xbar2lts_cycles_active.sum.pct_of_peak_sustained_elapsed","lts__xcomp_gxc_sectors_input.sum","lts__xcomp_gxc_sectors_input.sum.pct_of_peak_sustained_elapsed","lts__xcomp_gxc_sectors_input.sum.per_second","lts__xcomp_gxc_sectors_output.sum","lts__xcomp_gxc_sectors_output.sum.pct_of_peak_sustained_elapsed","lts__xcomp_gxc_sectors_output.sum.per_second","memory_access_size_type","memory_access_type","memory_l1_tag_requests_global","memory_l1_wavefronts_shared","memory_l1_wavefronts_shared_ideal","memory_l2_theoretical_sectors_global","memory_l2_theoretical_sectors_global_ideal","memory_l2_theoretical_sectors_local","memory_type","numa__cpu_affinity","numa__dev_display_name_all","numa__id_cpu","numa__id_memory","nvlink__bandwidth","nvlink__count_logical","nvlink__count_physical","nvlink__destination_ports","nvlink__dev0Id","nvlink__dev0type","nvlink__dev1Id","nvlink__dev1type","nvlink__dev_display_name_all","nvlink__enabled_mask","nvlink__is_direct_link","nvlink__is_nvswitch_connected","nvlink__max_count","nvlink__peer_access","nvlink__peer_atomic","nvlink__source_ports","nvlink__system_access","nvlink__system_atomic","pmsampling:gr__ctas_launched_queue_sync_realtime.sum","pmsampling:l1tex__data_pipe_lsu_wavefronts.avg","pmsampling:l1tex__lsu_writeback_active.avg","pmsampling:l1tex__t_sector_hit_rate.pct","pmsampling:lts__average_t_sector_hit_rate_realtime.pct","pmsampling:sm__average_thread_inst_executed_pred_on_per_inst_executed_realtime.pct","pmsampling:sm__cycles_active.avg","pmsampling:sm__inst_executed_pipe_alu_realtime.avg.pct_of_peak_sustained_elapsed","pmsampling:sm__inst_executed_realtime.avg.per_cycle_active","pmsampling:sm__pipe_tensor_cycles_active_realtime.avg.pct_of_peak_sustained_elapsed","pmsampling:smsp__inst_executed_pipe_fmaheavy.avg.pct_of_peak_sustained_elapsed","pmsampling:smsp__inst_executed_pipe_fmalite.avg.pct_of_peak_sustained_elapsed","pmsampling:tpc__warps_active_realtime.avg.per_cycle_active","pmsampling:tpc__warps_active_realtime.sum.per_cycle_active","profiler__perfworks_session_reuse","profiler__pmsampler_buffer_size_bytes","profiler__pmsampler_ctxsw_0","profiler__pmsampler_ctxsw_1","profiler__pmsampler_interval_time","profiler__pmsampler_merged_samples","profiler__pmsampler_pass_groups","profiler__replayer_bytes_mem_accessible.avg","profiler__replayer_bytes_mem_accessible.max","profiler__replayer_bytes_mem_accessible.min","profiler__replayer_bytes_mem_accessible.sum","profiler__replayer_bytes_mem_backed_up.avg","profiler__replayer_bytes_mem_backed_up.max","profiler__replayer_bytes_mem_backed_up.min","profiler__replayer_bytes_mem_backed_up.sum","profiler__replayer_passes","profiler__replayer_passes_type_warmup","profiler__timestamp_workload_end_0","profiler__timestamp_workload_end_1","profiler__timestamp_workload_start_0","profiler__timestamp_workload_start_1","sass__inst_executed_global_loads","sass__inst_executed_global_stores","sass__inst_executed_local_loads","sass__inst_executed_local_stores","sass__inst_executed_per_opcode","sass__inst_executed_per_opcode_category","sass__inst_executed_per_opcode_with_modifier_all","sass__inst_executed_per_opcode_with_modifier_selective","sass__inst_executed_register_spilling","sass__inst_executed_register_spilling_mem_local","sass__inst_executed_register_spilling_mem_shared","sass__inst_executed_register_spilling_op_read","sass__inst_executed_register_spilling_op_write","sass__inst_executed_shared_loads","sass__inst_executed_shared_stores","sass__thread_inst_executed_per_opcode_category","sass__thread_inst_executed_true_per_opcode","sass__thread_inst_executed_true_per_opcode_with_modifier_all","sass__thread_inst_executed_true_per_opcode_with_modifier_selective","sm__cycles_active.avg","sm__cycles_active.max","sm__cycles_active.min","sm__cycles_active.sum","sm__cycles_elapsed.avg","sm__cycles_elapsed.avg.per_second","sm__cycles_elapsed.max","sm__cycles_elapsed.max.per_second","sm__cycles_elapsed.min","sm__cycles_elapsed.min.per_second","sm__cycles_elapsed.sum","sm__cycles_elapsed.sum.per_second","sm__dcc_request_hit_rate.pct","sm__dcc_requests.sum","sm__dcc_requests.sum.pct_of_peak_sustained_elapsed","sm__dcc_requests_lookup_hit.sum","sm__icc_request_hit_rate.pct","sm__icc_requests.sum","sm__icc_requests.sum.pct_of_peak_sustained_elapsed","sm__inst_executed.avg.pct_of_peak_sustained_elapsed","sm__inst_executed.avg.per_cycle_active","sm__inst_executed.avg.per_cycle_elapsed","sm__inst_executed.max.pct_of_peak_sustained_elapsed","sm__inst_executed.max.per_cycle_active","sm__inst_executed.max.per_cycle_elapsed","sm__inst_executed.min.pct_of_peak_sustained_elapsed","sm__inst_executed.min.per_cycle_active","sm__inst_executed.min.per_cycle_elapsed","sm__inst_executed.sum.pct_of_peak_sustained_elapsed","sm__inst_executed.sum.per_cycle_active","sm__inst_executed.sum.per_cycle_elapsed","sm__inst_executed_pipe_adu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_adu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_adu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_adu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_adu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_adu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_alu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_alu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_alu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_alu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_alu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_alu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_alu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_alu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_cbu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_cbu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_cbu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_cbu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_cbu_pred_on_any.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fma_type_fp16.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fma_type_fp16.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_dmma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_dmma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_dmma.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_dmma.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_dmma.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_dmma.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_dmma.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_dmma.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_fp64.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_fp64.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_fp64.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_fp64.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_fp64.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_fp64.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_fp64_op_fp64.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_fp64_op_fp64.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_ipa.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_lsu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_lsu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_lsu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_lsu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_lsu.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tensor_subpipe_hmma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_tensor_subpipe_hmma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tensor_subpipe_imma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_tensor_subpipe_imma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_tex.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_tex.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_tex.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tex.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_tex.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tma.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_tma.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tma.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_tma.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tma.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_tma.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_tma.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_tma.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_uniform.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_uniform.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_uniform.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_uniform.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_uniform.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_workid.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_workid.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_workid.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_workid.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_workid.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_workid.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_workid.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_workid.sum.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.avg.pct_of_peak_sustained_active","sm__inst_executed_pipe_xu.avg.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.max.pct_of_peak_sustained_active","sm__inst_executed_pipe_xu.max.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.min.pct_of_peak_sustained_active","sm__inst_executed_pipe_xu.min.pct_of_peak_sustained_elapsed","sm__inst_executed_pipe_xu.sum.pct_of_peak_sustained_active","sm__inst_executed_pipe_xu.sum.pct_of_peak_sustained_elapsed","sm__inst_issued.avg.pct_of_peak_sustained_elapsed","sm__inst_issued.avg.per_cycle_active","sm__inst_issued.max.pct_of_peak_sustained_elapsed","sm__inst_issued.max.per_cycle_active","sm__inst_issued.min.pct_of_peak_sustained_elapsed","sm__inst_issued.min.per_cycle_active","sm__inst_issued.sum.pct_of_peak_sustained_elapsed","sm__inst_issued.sum.per_cycle_active","sm__instruction_throughput.avg.pct_of_peak_sustained_elapsed","sm__instruction_throughput.max.pct_of_peak_sustained_elapsed","sm__instruction_throughput.min.pct_of_peak_sustained_elapsed","sm__instruction_throughput.sum.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","sm__instruction_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","sm__issue_active.avg.pct_of_peak_sustained_elapsed","sm__issue_active.max.pct_of_peak_sustained_elapsed","sm__issue_active.min.pct_of_peak_sustained_elapsed","sm__issue_active.sum.pct_of_peak_sustained_elapsed","sm__maximum_warps_avg_per_active_cycle","sm__maximum_warps_per_active_cycle_pct","sm__memory_throughput.avg.pct_of_peak_sustained_elapsed","sm__memory_throughput.max.pct_of_peak_sustained_elapsed","sm__memory_throughput.min.pct_of_peak_sustained_elapsed","sm__memory_throughput.sum.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.avg.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.max.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.min.pct_of_peak_sustained_elapsed","sm__memory_throughput_internal_activity.sum.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.avg.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.max.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.min.pct_of_peak_sustained_elapsed","sm__mio2rf_writeback_active.sum.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.avg.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.max.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.min.pct_of_peak_sustained_elapsed","sm__mio_inst_issued.sum.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.max.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.min.pct_of_peak_sustained_elapsed","sm__mio_pq_read_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.max.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.min.pct_of_peak_sustained_elapsed","sm__mio_pq_write_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_bf16_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.avg.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.max.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.min.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_off.sum.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.avg.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.max.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.min.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp16_sparsity_on.sum.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_fp16_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_hmma_src_tf32_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.avg.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.max.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.min.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_off.sum.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.avg.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.max.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.min.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_op_imma_src_int8_sparsity_on.sum.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.avg","sm__ops_path_tensor_src_bf16_dst_fp32.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.avg.peak_sustained","sm__ops_path_tensor_src_bf16_dst_fp32.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.avg.per_cycle_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.avg.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.max","sm__ops_path_tensor_src_bf16_dst_fp32.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.max.peak_sustained","sm__ops_path_tensor_src_bf16_dst_fp32.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.max.per_cycle_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.max.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.min","sm__ops_path_tensor_src_bf16_dst_fp32.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.min.peak_sustained","sm__ops_path_tensor_src_bf16_dst_fp32.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.min.per_cycle_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.min.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.sum","sm__ops_path_tensor_src_bf16_dst_fp32.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.sum.peak_sustained","sm__ops_path_tensor_src_bf16_dst_fp32.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_bf16_dst_fp32.sum.per_cycle_elapsed","sm__ops_path_tensor_src_bf16_dst_fp32.sum.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.avg","sm__ops_path_tensor_src_fp16_dst_fp16.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.avg.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp16.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.avg.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.max","sm__ops_path_tensor_src_fp16_dst_fp16.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.max.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp16.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.max.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.min","sm__ops_path_tensor_src_fp16_dst_fp16.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.min.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp16.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.min.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.sum","sm__ops_path_tensor_src_fp16_dst_fp16.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.sum.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp16.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp16.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp16.sum.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.avg","sm__ops_path_tensor_src_fp16_dst_fp32.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.avg.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp32.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.avg.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.max","sm__ops_path_tensor_src_fp16_dst_fp32.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.max.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp32.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.max.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.min","sm__ops_path_tensor_src_fp16_dst_fp32.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.min.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp32.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.min.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.sum","sm__ops_path_tensor_src_fp16_dst_fp32.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.sum.peak_sustained","sm__ops_path_tensor_src_fp16_dst_fp32.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp16_dst_fp32.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp16_dst_fp32.sum.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp16_sparsity_on.sum.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp4_fp6_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_src_fp64.avg","sm__ops_path_tensor_src_fp64.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp64.avg.peak_sustained","sm__ops_path_tensor_src_fp64.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp64.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp64.avg.per_second","sm__ops_path_tensor_src_fp64.max","sm__ops_path_tensor_src_fp64.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp64.max.peak_sustained","sm__ops_path_tensor_src_fp64.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp64.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp64.max.per_second","sm__ops_path_tensor_src_fp64.min","sm__ops_path_tensor_src_fp64.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp64.min.peak_sustained","sm__ops_path_tensor_src_fp64.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp64.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp64.min.per_second","sm__ops_path_tensor_src_fp64.sum","sm__ops_path_tensor_src_fp64.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp64.sum.peak_sustained","sm__ops_path_tensor_src_fp64.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp64.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp64.sum.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp16_sparsity_on.sum.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.avg.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.max.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.min.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_off.sum.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.avg.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.max.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.min.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.peak_sustained","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.per_cycle_elapsed","sm__ops_path_tensor_src_fp8_dst_fp32_sparsity_on.sum.per_second","sm__ops_path_tensor_src_int8.avg","sm__ops_path_tensor_src_int8.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_int8.avg.peak_sustained","sm__ops_path_tensor_src_int8.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_int8.avg.per_cycle_elapsed","sm__ops_path_tensor_src_int8.avg.per_second","sm__ops_path_tensor_src_int8.max","sm__ops_path_tensor_src_int8.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_int8.max.peak_sustained","sm__ops_path_tensor_src_int8.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_int8.max.per_cycle_elapsed","sm__ops_path_tensor_src_int8.max.per_second","sm__ops_path_tensor_src_int8.min","sm__ops_path_tensor_src_int8.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_int8.min.peak_sustained","sm__ops_path_tensor_src_int8.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_int8.min.per_cycle_elapsed","sm__ops_path_tensor_src_int8.min.per_second","sm__ops_path_tensor_src_int8.sum","sm__ops_path_tensor_src_int8.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_int8.sum.peak_sustained","sm__ops_path_tensor_src_int8.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_int8.sum.per_cycle_elapsed","sm__ops_path_tensor_src_int8.sum.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.avg","sm__ops_path_tensor_src_tf32_dst_fp32.avg.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.avg.peak_sustained","sm__ops_path_tensor_src_tf32_dst_fp32.avg.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.avg.per_cycle_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.avg.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.max","sm__ops_path_tensor_src_tf32_dst_fp32.max.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.max.peak_sustained","sm__ops_path_tensor_src_tf32_dst_fp32.max.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.max.per_cycle_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.max.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.min","sm__ops_path_tensor_src_tf32_dst_fp32.min.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.min.peak_sustained","sm__ops_path_tensor_src_tf32_dst_fp32.min.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.min.per_cycle_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.min.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.sum","sm__ops_path_tensor_src_tf32_dst_fp32.sum.pct_of_peak_sustained_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.sum.peak_sustained","sm__ops_path_tensor_src_tf32_dst_fp32.sum.peak_sustained_elapsed.per_second","sm__ops_path_tensor_src_tf32_dst_fp32.sum.per_cycle_elapsed","sm__ops_path_tensor_src_tf32_dst_fp32.sum.per_second","sm__pipe_alu_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_alu_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_alu_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_alu_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_alu_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_alu_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_alu_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_alu_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_aluheavy_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_fma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_fma_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_fma_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fma_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_fma_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fmaheavy_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_fp64_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_fp64_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_fp64_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_fp64_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_fp64_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_tensor_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_tensor_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_tensor_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_tensor_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_tensor_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__pipe_tensor_subpipe_hmma_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_tensor_subpipe_hmma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tensor_subpipe_imma_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_tensor_subpipe_imma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.avg.pct_of_peak_sustained_active","sm__pipe_tma_cycles_active.avg.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.max.pct_of_peak_sustained_active","sm__pipe_tma_cycles_active.max.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.min.pct_of_peak_sustained_active","sm__pipe_tma_cycles_active.min.pct_of_peak_sustained_elapsed","sm__pipe_tma_cycles_active.sum.pct_of_peak_sustained_active","sm__pipe_tma_cycles_active.sum.pct_of_peak_sustained_elapsed","sm__sass_inst_executed_op_ldgsts_cache_access.sum","sm__sass_inst_executed_op_ldgsts_cache_bypass.sum","sm__sass_l1tex_m_xbar2l1tex_read_bytes_mem_global_op_ldgsts_cache_bypass.sum","sm__sass_l1tex_m_xbar2l1tex_read_bytes_mem_global_op_ldgsts_cache_bypass.sum.pct_of_peak_sustained_elapsed","sm__sass_l1tex_m_xbar2l1tex_read_bytes_mem_global_op_ldgsts_cache_bypass.sum.per_second","sm__sass_l1tex_t_requests_pipe_lsu_mem_global_op_ldgsts.sum","sm__sass_l1tex_t_requests_pipe_lsu_mem_global_op_ldgsts.sum.pct_of_peak_sustained_elapsed","sm__sass_l1tex_t_requests_pipe_lsu_mem_global_op_ldgsts_cache_access.sum","sm__sass_l1tex_t_requests_pipe_lsu_mem_global_op_ldgsts_cache_bypass.sum","sm__sass_l1tex_t_sectors_pipe_lsu_mem_global_op_ldgsts_cache_access.sum","sm__sass_l1tex_t_sectors_pipe_lsu_mem_global_op_ldgsts_cache_access.sum.pct_of_peak_sustained_elapsed","sm__sass_l1tex_t_sectors_pipe_lsu_mem_global_op_ldgsts_cache_bypass.sum","sm__sass_l1tex_t_sectors_pipe_lsu_mem_global_op_ldgsts_cache_bypass.sum.pct_of_peak_sustained_elapsed","sm__sass_thread_inst_executed_op_dfma_pred_on.avg.peak_sustained","sm__sass_thread_inst_executed_op_dfma_pred_on.max.peak_sustained","sm__sass_thread_inst_executed_op_dfma_pred_on.min.peak_sustained","sm__sass_thread_inst_executed_op_dfma_pred_on.sum.peak_sustained","sm__sass_thread_inst_executed_op_ffma_pred_on.avg.peak_sustained","sm__sass_thread_inst_executed_op_ffma_pred_on.max.peak_sustained","sm__sass_thread_inst_executed_op_ffma_pred_on.min.peak_sustained","sm__sass_thread_inst_executed_op_ffma_pred_on.sum.peak_sustained","sm__sass_thread_inst_executed_op_hfma_pred_on.avg.peak_sustained","sm__sass_thread_inst_executed_op_hfma_pred_on.max.peak_sustained","sm__sass_thread_inst_executed_op_hfma_pred_on.min.peak_sustained","sm__sass_thread_inst_executed_op_hfma_pred_on.sum.peak_sustained","sm__throughput.avg.pct_of_peak_sustained_elapsed","sm__throughput.max.pct_of_peak_sustained_elapsed","sm__throughput.min.pct_of_peak_sustained_elapsed","sm__throughput.sum.pct_of_peak_sustained_elapsed","sm__warps_active.avg.pct_of_peak_sustained_active","sm__warps_active.avg.per_cycle_active","sm__warps_active.max.pct_of_peak_sustained_active","sm__warps_active.max.per_cycle_active","sm__warps_active.min.pct_of_peak_sustained_active","sm__warps_active.min.per_cycle_active","sm__warps_active.sum.pct_of_peak_sustained_active","sm__warps_active.sum.per_cycle_active","smsp__average_warp_latency_per_inst_issued.ratio","smsp__average_warps_active_per_inst_executed.ratio","smsp__average_warps_issue_stalled_barrier_per_issue_active.ratio","smsp__average_warps_issue_stalled_branch_resolving_per_issue_active.ratio","smsp__average_warps_issue_stalled_dispatch_stall_per_issue_active.ratio","smsp__average_warps_issue_stalled_drain_per_issue_active.ratio","smsp__average_warps_issue_stalled_lg_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_long_scoreboard_per_issue_active.ratio","smsp__average_warps_issue_stalled_math_pipe_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_membar_per_issue_active.ratio","smsp__average_warps_issue_stalled_mio_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_misc_per_issue_active.ratio","smsp__average_warps_issue_stalled_no_instruction_per_issue_active.ratio","smsp__average_warps_issue_stalled_not_selected_per_issue_active.ratio","smsp__average_warps_issue_stalled_selected_per_issue_active.ratio","smsp__average_warps_issue_stalled_short_scoreboard_per_issue_active.ratio","smsp__average_warps_issue_stalled_sleeping_per_issue_active.ratio","smsp__average_warps_issue_stalled_tex_throttle_per_issue_active.ratio","smsp__average_warps_issue_stalled_wait_per_issue_active.ratio","smsp__branch_targets_threads_divergent","smsp__cycles_active.avg","smsp__cycles_active.max","smsp__cycles_active.min","smsp__cycles_active.sum","smsp__cycles_elapsed.avg","smsp__cycles_elapsed.avg.per_second","smsp__cycles_elapsed.max","smsp__cycles_elapsed.max.per_second","smsp__cycles_elapsed.min","smsp__cycles_elapsed.min.per_second","smsp__cycles_elapsed.sum","smsp__cycles_elapsed.sum.per_second","smsp__inst_executed.avg","smsp__inst_executed.max","smsp__inst_executed.min","smsp__inst_executed.sum","smsp__inst_executed_op_branch.avg","smsp__inst_executed_op_branch.max","smsp__inst_executed_op_branch.min","smsp__inst_executed_op_branch.sum","smsp__inst_executed_op_generic_atom_dot_alu.sum","smsp__inst_executed_op_generic_atom_dot_alu.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_generic_atom_dot_cas.sum","smsp__inst_executed_op_generic_atom_dot_cas.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_global_red.sum","smsp__inst_executed_op_global_red.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_ldgsts.sum","smsp__inst_executed_op_ldgsts.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_ldsm.sum","smsp__inst_executed_op_ldsm.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_shared_atom.sum","smsp__inst_executed_op_shared_atom.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_shared_stsm.sum","smsp__inst_executed_op_shared_stsm.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_atom_dot_alu.sum","smsp__inst_executed_op_surface_atom_dot_alu.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_atom_dot_cas.sum","smsp__inst_executed_op_surface_atom_dot_cas.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_ld.sum","smsp__inst_executed_op_surface_ld.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_red.sum","smsp__inst_executed_op_surface_red.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_surface_st.sum","smsp__inst_executed_op_surface_st.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_texture.sum","smsp__inst_executed_op_texture.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_tma_ld.sum","smsp__inst_executed_op_tma_ld.sum.pct_of_peak_sustained_elapsed","smsp__inst_executed_op_tma_st.sum","smsp__inst_executed_op_tma_st.sum.pct_of_peak_sustained_elapsed","smsp__inst_issued.avg","smsp__inst_issued.max","smsp__inst_issued.min","smsp__inst_issued.sum","smsp__issue_active.avg.pct_of_peak_sustained_active","smsp__issue_active.avg.per_cycle_active","smsp__issue_active.max.pct_of_peak_sustained_active","smsp__issue_active.max.per_cycle_active","smsp__issue_active.min.pct_of_peak_sustained_active","smsp__issue_active.min.per_cycle_active","smsp__issue_active.sum.pct_of_peak_sustained_active","smsp__issue_active.sum.per_cycle_active","smsp__issue_inst0.avg.pct_of_peak_sustained_active","smsp__issue_inst0.max.pct_of_peak_sustained_active","smsp__issue_inst0.min.pct_of_peak_sustained_active","smsp__issue_inst0.sum.pct_of_peak_sustained_active","smsp__maximum_warps_avg_per_active_cycle","smsp__pcsamp_aggregated_passes","smsp__pcsamp_buffer_size_bytes","smsp__pcsamp_dropped_bytes","smsp__pcsamp_interval","smsp__pcsamp_interval_cycles","smsp__pcsamp_sample_count","smsp__pcsamp_warps_issue_stalled_barrier","smsp__pcsamp_warps_issue_stalled_barrier_not_issued","smsp__pcsamp_warps_issue_stalled_branch_resolving","smsp__pcsamp_warps_issue_stalled_branch_resolving_not_issued","smsp__pcsamp_warps_issue_stalled_dispatch_stall","smsp__pcsamp_warps_issue_stalled_dispatch_stall_not_issued","smsp__pcsamp_warps_issue_stalled_drain","smsp__pcsamp_warps_issue_stalled_drain_not_issued","smsp__pcsamp_warps_issue_stalled_lg_throttle","smsp__pcsamp_warps_issue_stalled_lg_throttle_not_issued","smsp__pcsamp_warps_issue_stalled_long_scoreboard","smsp__pcsamp_warps_issue_stalled_long_scoreboard_not_issued","smsp__pcsamp_warps_issue_stalled_math_pipe_throttle","smsp__pcsamp_warps_issue_stalled_math_pipe_throttle_not_issued","smsp__pcsamp_warps_issue_stalled_membar","smsp__pcsamp_warps_issue_stalled_membar_not_issued","smsp__pcsamp_warps_issue_stalled_mio_throttle","smsp__pcsamp_warps_issue_stalled_mio_throttle_not_issued","smsp__pcsamp_warps_issue_stalled_misc","smsp__pcsamp_warps_issue_stalled_misc_not_issued","smsp__pcsamp_warps_issue_stalled_no_instructions","smsp__pcsamp_warps_issue_stalled_no_instructions_not_issued","smsp__pcsamp_warps_issue_stalled_not_selected","smsp__pcsamp_warps_issue_stalled_not_selected_not_issued","smsp__pcsamp_warps_issue_stalled_selected","smsp__pcsamp_warps_issue_stalled_selected_not_issued","smsp__pcsamp_warps_issue_stalled_short_scoreboard","smsp__pcsamp_warps_issue_stalled_short_scoreboard_not_issued","smsp__pcsamp_warps_issue_stalled_sleeping","smsp__pcsamp_warps_issue_stalled_sleeping_not_issued","smsp__pcsamp_warps_issue_stalled_tex_throttle","smsp__pcsamp_warps_issue_stalled_tex_throttle_not_issued","smsp__pcsamp_warps_issue_stalled_wait","smsp__pcsamp_warps_issue_stalled_wait_not_issued","smsp__sass_average_branch_targets_threads_uniform.pct","smsp__sass_average_data_bytes_per_sector_mem_global_op_ld.max_rate","smsp__sass_average_data_bytes_per_sector_mem_global_op_ld.ratio","smsp__sass_average_data_bytes_per_sector_mem_global_op_st.max_rate","smsp__sass_average_data_bytes_per_sector_mem_global_op_st.ratio","smsp__sass_average_data_bytes_per_sector_mem_local_op_ld.max_rate","smsp__sass_average_data_bytes_per_sector_mem_local_op_ld.ratio","smsp__sass_average_data_bytes_per_sector_mem_local_op_st.max_rate","smsp__sass_average_data_bytes_per_sector_mem_local_op_st.ratio","smsp__sass_branch_targets_threads_divergent.avg","smsp__sass_branch_targets_threads_divergent.max","smsp__sass_branch_targets_threads_divergent.min","smsp__sass_branch_targets_threads_divergent.sum","smsp__sass_inst_executed_memdesc_explicit_evict_type","smsp__sass_inst_executed_memdesc_explicit_hitprop_evict_first","smsp__sass_inst_executed_memdesc_explicit_hitprop_evict_last","smsp__sass_inst_executed_memdesc_explicit_hitprop_evict_normal","smsp__sass_inst_executed_memdesc_explicit_hitprop_evict_normal_demote","smsp__sass_inst_executed_memdesc_explicit_missprop_evict_first","smsp__sass_inst_executed_memdesc_explicit_missprop_evict_normal","smsp__sass_inst_executed_op_dshared.sum","smsp__sass_inst_executed_op_dshared.sum.pct_of_peak_sustained_elapsed","smsp__sass_inst_executed_op_dshared_atom.sum","smsp__sass_inst_executed_op_dshared_ld.sum","smsp__sass_inst_executed_op_dshared_redas.sum","smsp__sass_inst_executed_op_dshared_st.sum","smsp__sass_inst_executed_op_dshared_stas.sum","smsp__sass_inst_executed_op_dshared_tma_red.sum","smsp__sass_inst_executed_op_dshared_tma_st.sum","smsp__sass_inst_executed_op_global_ld.sum","smsp__sass_inst_executed_op_global_st.sum","smsp__sass_inst_executed_op_local_ld.sum","smsp__sass_inst_executed_op_local_st.sum","smsp__sass_inst_executed_op_shared.sum","smsp__sass_inst_executed_op_shared.sum.pct_of_peak_sustained_elapsed","smsp__sass_inst_executed_op_shared_ld.sum","smsp__sass_inst_executed_op_shared_st.sum","smsp__sass_inst_executed_op_tma_ld.sum","smsp__sass_inst_executed_op_tma_red.sum","smsp__sass_inst_executed_op_tma_st.sum","smsp__sass_l1tex_data_pipe_lsu_wavefronts_mem_shared_op_ldgsts.sum","smsp__sass_l1tex_data_pipe_lsu_wavefronts_mem_shared_op_ldgsts.sum.pct_of_peak_sustained_elapsed","smsp__sass_l1tex_data_pipe_lsu_wavefronts_mem_shared_op_ldgsts_cache_access.sum","smsp__sass_l1tex_data_pipe_lsu_wavefronts_mem_shared_op_ldgsts_cache_access.sum.pct_of_peak_sustained_elapsed","smsp__sass_l1tex_m_xbar2l1tex_read_sectors_mem_global_op_ldgsts_cache_bypass.sum","smsp__sass_l1tex_m_xbar2l1tex_read_sectors_mem_global_op_ldgsts_cache_bypass.sum.pct_of_peak_sustained_elapsed","smsp__sass_thread_inst_executed_op_dadd_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dadd_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dadd_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dadd_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dfma_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dfma_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dfma_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dfma_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dmul_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dmul_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dmul_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_dmul_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fadd_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fadd_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fadd_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fadd_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_ffma_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_ffma_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_ffma_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_ffma_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fmul_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fmul_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fmul_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_fmul_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hadd_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hadd_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hadd_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hadd_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hfma_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hfma_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hfma_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hfma_pred_on.sum.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hmul_pred_on.avg.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hmul_pred_on.max.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hmul_pred_on.min.per_cycle_elapsed","smsp__sass_thread_inst_executed_op_hmul_pred_on.sum.per_cycle_elapsed","smsp__thread_inst_executed_per_inst_executed.ratio","smsp__thread_inst_executed_pred_on_per_inst_executed.ratio","smsp__warps_active.avg.peak_sustained","smsp__warps_active.avg.per_cycle_active","smsp__warps_active.max.peak_sustained","smsp__warps_active.max.per_cycle_active","smsp__warps_active.min.peak_sustained","smsp__warps_active.min.per_cycle_active","smsp__warps_active.sum.peak_sustained","smsp__warps_active.sum.per_cycle_active","smsp__warps_eligible.avg.per_cycle_active","smsp__warps_eligible.max.per_cycle_active","smsp__warps_eligible.min.per_cycle_active","smsp__warps_eligible.sum.per_cycle_active","thread_inst_executed","thread_inst_executed_true" +"","","","","","","","","","","","","","thread","thread","thread","thread","","","","%","Kbyte","Tbyte","","","byte","","%","%/register","%/Kbyte","thread","thread","thread","%","thread","thread","thread","thread","thread","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","","request","%","request","%","","%","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","ms","ms","ms","ms","","","","","block","block","block","block","","","","","%","%","%","%","%","","%","inst","cycle","cycle","cycle","cycle","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","","","","","","%","%","%","%","%","%","%","%","%","%","%","%","","%","","%","","%","","%","%","%","%","%","%","%","%","%","%","%","%","%","cycle","%","","Khz","%","%","%","%","%","%","%","%","Mbyte","%","Gbyte/s","byte","%","byte/s","byte","%","byte/s","Mbyte","%","Gbyte/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector","%","sector","%","sector","%","sector","%","sector/s","sector","%","sector/us","sector","%","sector","%","sector","%","sector","%","Gbyte","%","Tbyte/s","byte","%","byte/s","Gbyte","%","Tbyte/s","%","%","%","%","sector","%","sector/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector/s","sector","%","sector","%","sector","%","sector","%","sector/ns","sector","%","sector","%","sector","%","sector","%","byte","%","byte/s","","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","","%","%","%","%","%","%","%","%","%","%","%","%","%","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","%","%","cycle","%","%","%","%","%","%","%","%","%","%","%","%","%","","block","block","block","","","","","cluster","block","","","","","","","","","","","","%","%","block","block","block","block","block","","","","","","Mbyte","","","","","register/thread","register/thread","Kbyte","Kbyte/block","Kbyte/block","Kbyte/block","Kbyte/block","byte/block","SM","","","thread","","","","","","","","","%","sector","%","","","","cycle","cycle","cycle","cycle","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","%","%","%","%","%","%","%","%","%","%","%","%","sector","%","sector/ns","sector","sector","sector","sector","%","","Ghz","%","","Ghz","%","","Ghz","%","","Ghz","request","request","request","request","request","request","request","request","%","%","%","sector","%","sector/cycle","sector/cycle","sector","%","sector","%","sector","%","sector/ns","sector","%","sector/us","%","sector/cycle","sector/cycle","sector","sector/s","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","sector/ns","%","sector/cycle","sector/cycle","sector","sector/ms","sector","sector","%","sector/cycle","sector/cycle","%","%","sector","%","sector/ns","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","sector/s","%","sector/cycle","sector/cycle","sector","sector/s","sector","sector","%","sector/cycle","sector/cycle","sector","sector/s","sector","sector","sector","sector","sector","sector","sector","sector","%","sector/cycle","sector/cycle","sector","%","sector/ns","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","sector/cycle","sector/cycle","sector","%","sector/s","sector","sector","%","sector/cycle","sector/cycle","sector","%","sector/us","sector","sector","sector","sector","sector","sector","sector","sector","sector","sector","%","%","%","%","%","%","%","%","%","%","%","%","sector","%","sector/s","sector","%","sector/s","","","sectors","sectors","sectors","sectors","sectors","sectors","","","","","","","","","","","","","","","","","","","","","","","","block","","cycle","%","%","%","cycle","%","inst/cycle","%","%","%","warp","warp","","Mbyte","","","us","sample","","Gbyte","Gbyte","Gbyte","Gbyte","Gbyte","Gbyte","Gbyte","Gbyte","pass","pass","","","","","inst","inst","inst","inst","","","","","inst","inst","inst","inst","inst","inst","inst","","","","","cycle","cycle","cycle","cycle","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","%","cycle","%","cycle","%","cycle","%","%","inst/cycle","inst/cycle","%","inst/cycle","inst/cycle","%","inst/cycle","inst/cycle","%","inst/cycle","inst/cycle","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","inst/cycle","%","inst/cycle","%","inst/cycle","%","inst/cycle","%","%","%","%","%","%","%","%","%","%","%","%","warp","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","","%","","","","","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","%","inst","inst","byte","%","byte/s","","%","","","sector","%","sector","%","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","%","%","%","%","%","warp","%","warp","%","warp","%","warp","cycle","cycle","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","branches","cycle","cycle","cycle","cycle","cycle","Ghz","cycle","Ghz","cycle","Ghz","cycle","Ghz","inst","inst","inst","inst","inst","inst","inst","inst","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","%","inst","inst","inst","inst","%","","%","","%","","%","","%","%","%","%","warp","pass","Mbyte","byte","","cycle","","warp","warp","branches","branches","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","inst","inst","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","%","byte/sector","byte/sector","byte/sector","byte/sector","byte/sector","byte/sector","byte/sector","byte/sector","branches","branches","branches","branches","","inst","inst","inst","inst","inst","inst","inst","%","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","inst","%","inst","inst","inst","inst","inst","","%","","%","sector","%","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","inst/cycle","","","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","warp","inst","inst" +"0","73","python3.12","127.0.0.1","cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu","1","7","(384, 1, 1)","(296, 42, 1)","0","12.1","0","1","49636","46015","95873.000000","6036","6144","7215061.44","0","0","1.024000","1.507107","293","44556288","0","625","316.000000","4225.000000","4.975000","192.000000","12288.000000","12288.000000","0.030323","0","11.216880","0","0","0","36608229","1207261415","0","0","nan","0","432","1","0","1","1","1","0","1","0","1","0","443","2418000","1","12","1","0","1","1","1","No-CC","1","1","1","0","0","NVIDIA GB10","0","0","256","4","1","1","256","1","0","0","100","1","772935902","11794","32000","16","0","4","2500","16","161","4318","0","1","0","0","1","0","1","443","1","1","1","25165824","24","24","24","1","1","134217728","1024","1024","64","24","2418000","2147483647","65535","65535","4","1","8533000","18874368","2147483647","65536","65536","255","49152","101376","102400","1024","1536","48","12","2048","32768","32768","65536","32768","2048","32768","131072","16384","16384","16384","2046","32768","32768","2048","32768","268435456","32768","131072","32768","32768","65536","32768","2048","32768","65000","2097120","131072","32768","32768","131072","16384","32768","16384","8192","16384","8192","2046","32768","32768","4","8533000","1","1","0","0","0","0","48","6","4","1","0","1","1","1","0","15","1","0","1024","12","64","1","1","512","0","1","512","32","65536","130661769216","1","1","1","32","2798220.000000","1.926207","5640.000000","0.003882","8224","0.002831","36317742.500000","2.132444","36338193.000000","2.133645","36297325.000000","2.131245","145270970.000000","8.529777","77.673393","77.716884","77.629395","77.673393","38.962587","39.005204","38.932311","38.962587","76.458743","76.606695","76.298073","76.458743","0","0","0","0","77.673393","77.716884","77.629395","77.673393","17.031040","17.031040","17.031040","17.031040","12384","12384","12384","12384","12384.000000","12384.000000","12384.000000","12384.000000","12432","12432","12432","12432","0.208009","0.210415","0.206373","0.208009","99.819427","1809792","0.207634","1180803976","36039245.812500","36113066.000000","35944597.000000","1729883799.000000","36317742.500000","2.132444","36338193.000000","2.133645","36297325.000000","2.131245","1743251640.000000","102.357322","47563205","0","46508786","0","1034202","17.344946","17.479037","17.210811","17.344946","5.842399","5.887514","5.797284","5.842399","41.880489","42.205724","41.556958","41.880489","727700427","41.743854","0","0","715093417","41.020665","4223650","0.242286","0","0","0","0","0","0","0","0","38.684431","39.132509","38.236352","38.684431","960.000000","0.000055","48","56.367668","26.632013","26.940474","26.426373","26.632013","11.606359","11.735944","11.477431","11.606359","406.683648","0.729032","23.878967","0","0","0","0","0","0","406.683648","0.729032","23.878967","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","12708864.000000","0.729032","746.217730","0","0","0","0","0","0","0","0","25.664425","46.006742","1.506921","0","0","0","25.664422","46.006736","1.506920","46.006742","46.539638","45.651477","46.006742","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","802013184.000000","46.006736","47.091263","96.000000","0.000006","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","960","0.000055","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","960","0.000055","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","90.000000","0","90.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","960.000000","864.000000","96.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0.042018","0.042398","0.041635","0.042018","46.362263","46.006742","46.899278","46.539638","46.004253","45.651477","46.362263","46.006742","8","384","1","1","384","0","0","0","0","8","PolicySpread","0","1","0","CachePreferNone","1","296","42","1","12432","","0","0","3.000000","24.000000","1.000000","1.000000","4.000000","300","157","0","2028","2388","4.718592","0","0","0","0","168","168.000000","102.400000","89.088000","89.088000","1.024000","88.064000","0","48","1024","7","4773888","24","all","0","0","0","0","0","259","100","0","0","0","1.00","0","32545999.437500","32657096.000000","32492211.000000","520735991.000000","32783718.000000","1.924939","32783718.000000","1.924939","32783718.000000","1.924939","524539488.000000","30.799029","0","0","0","0","0","0","0","0","42.276093","42.299693","42.254622","42.276093","59982309.000000","11.435232","3.521940","0","0","0","0","76.458743","2","2.943569","76.606695","2","2.949265","76.298073","2","2.937383","76.458743","32","47.097100","203738812.000000","2176.000000","203680608.000000","0","0","200503392.000000","0","3177216.000000","91.101458","92.547312","0.149636","50928452.437500","77.673393","2.000000","1.553468","50956968.000000","77.716884","50899604.000000","77.629395","814855239.000000","77.673393","47.845301","12745592.000000","2.429863","748.374263","0","8.000000","0","0","0","19228.000000","76601.000000","0","0","0","0","742371185.000000","72417342.000000","742345007.000000","72310114.000000","802013280.000000","19.112319","47.091269","0.001659","1.000000","0.000017","8704.000000","511.066852","6648.000000","2056.000000","77.660706","2.000000","1.553214","77.704963","77.619970","814722144.000000","77.660706","47.837486","0","0","0","0","4538341.875000","4540968.000000","4534792.000000","72613470.000000","0","0","0","0","0","0","742137138.000000","72647642.000000","742437594.000000","4540477.625000","4542744.000000","4537720.000000","72647642.000000","0","0","0","0","1.000000","0","0","0","0","0","0","1.000000","0","0","0","0","0","0","0","0","0","0","0","76.449276","2.000000","1.528986","802013280.000000","76.449276","47.091269","0","0","0","0","0","0","742415110.000000","59664462.000000","742415110.000000","59664462.000000","0","1.000000","0","0","0","0","0","0","2.422861","1.000000","0.024229","12708864.000000","2.422861","746.217730","0","0","0","0","0","0","0","12708864.000000","0","12708864.000000","38.843584","38.867529","38.819355","38.843584","77.673393","77.716884","77.629395","77.673393","38.658720","38.685603","38.634477","38.658720","0","0","0","0","0","0","0","0","960","716319840","671763552","960","960","0","0","1","1","1","1","0","0","0","0","0","0","0","0","1","0","0","0","0","0","0","0","0","0","32.000000","15209492.98","14049312.000000","","","","34583650.437500","3.723238","1969.697337","84.262067","2.030255","0.099917","27009.964736","1296478.307310","0","18.022400","1","1","6.000000","0","2","25.185856","25.185856","25.185856","957.062541","25.185856","25.185856","25.185856","957.062541","38.000000","0","1","1","1","1","1000.000000","0","0","0","1180803976","1180803976","1180803976","1180803976","0","no data","no data","0","0","89236896.000000","450.000000","29478400372","29478400372","29478400372","29478400372","36039245.812500","36113066.000000","35944597.000000","1729883799.000000","36317742.500000","2.132444","36338193.000000","2.133645","36297325.000000","2.131245","1743251640.000000","102.357322","99.463631","816788.000000","0.046854","812407.000000","99.970991","36381373.000000","4.173967","17.313356","0.697886","0.692534","17.505091","0.705615","0.700204","17.184602","0.692696","0.687384","17.313356","33.498521","33.241643","4.263262","4.230570","4.283103","4.250259","4.246226","4.213665","4.263262","4.230570","3.641974","3.614047","3.684159","3.655907","3.613817","3.586105","3.641974","3.614047","0.146173","0.145052","0.146973","0.145846","0.145061","0.143948","0.146173","0.145052","0.520434","0.527147","0.511312","0.520434","0.531922","0.527843","0.536057","0.531947","0.527794","0.523747","0.531922","0.527843","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","26.837815","26.632013","27.045044","26.837654","26.630585","26.426373","26.837815","26.632013","82.421791","81.789753","0","0","0","0","0","0","0","0","0","0","0.517478","0.513509","0.523471","0.519457","0.513482","0.509544","0.517478","0.513509","3.397567","3.371514","3.423808","3.397554","3.371328","3.345476","3.397567","3.371514","0.000719","0.000713","0.000727","0.000721","0.000713","0.000708","0.000719","0.000713","0","0","0","0","0","0","0","0","17.420154","0.702191","17.611351","0.709898","17.289200","0.696912","17.420154","33.705158","81.789753","82.421334","80.842382","81.789753","0","0","0","0","17.420154","17.611351","17.289200","17.420154","12.000000","25.000000","26.632013","26.837654","26.426373","26.632013","0.046850","0.047387","0.046492","0.046850","9.697062","9.809381","9.584743","9.697062","10.401641","10.478072","10.326578","10.401641","3.720293","3.751585","3.677941","3.720293","3.720318","3.751610","3.677965","3.720318","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","49152","104813897410845.14","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","49152","104813897410845.14","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","49152","104813897410845.14","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","512","1091811431362.97","0","0","0","0","512","1091811431362.97","0","0","0","0","512","1091811431362.97","0","0","0","0","24576","52406948705422.57","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","49152","104813897410845.14","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","196608","419255589643380.56","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","121668370432","81.789753","4096","8734491450903.76","3350.11","7143919010935.33","122607894528","82.421334","4096","8734491450903.76","3375.98","7199084408703.17","120728846336","81.158172","4096","8734491450903.76","3324.24","7088753613167.49","5840081780736","81.789753","196608","419255589643380.56","160805.20","342908112524895.69","0","0","8192","17468982901807.52","0","0","0","0","8192","17468982901807.52","0","0","0","0","8192","17468982901807.52","0","0","0","0","393216","838511179286761.12","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","196608","419255589643380.56","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","196608","419255589643380.56","0","0","0","0","3.51","7478160488.79","0","0","0","0","3.51","7478160488.79","0","0","0","0","3.51","7478160488.79","0","0","0","0","168.33","358951703461.80","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","196608","419255589643380.56","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","2048","4367245725451.88","0","0","0","0","98304","209627794821690.28","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","196608","419255589643380.56","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","4096","8734491450903.76","0","0","0","0","196608","419255589643380.56","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","1024","2183622862725.94","0","0","0","0","49152","104813897410845.14","0","0","3.641990","3.614062","3.670124","3.641980","3.601820","3.574200","3.641990","3.614062","6.216056","6.264179","6.143976","6.216056","0.676900","0.671709","0.681537","0.676310","0.672651","0.667493","0.676900","0.671709","1.999371","2.014814","1.976159","1.999371","0","0","0","0","0","0","0","0","82.421791","81.789753","83.058253","82.421334","81.467099","80.842382","82.421791","81.789753","82.421791","81.789753","0","0","0.517478","0.513509","0.523471","0.519457","0.513482","0.509544","0.517478","0.513509","0","0","0","0","0","0","0","0","0","0","0","0","0","2.000000","2.000000","2.000000","96.000000","128.000000","128.000000","128.000000","6144.000000","64.000000","64.000000","64.000000","3072.000000","81.789753","82.421334","80.842382","81.789753","20.956326","10.059036","21.002397","10.081151","20.916088","10.039722","20.956326","482.833740","13.603819","13.687736","0.172007","0.158464","0.043709","0.000001","0","0.601708","4.064508","0","0.049404","0.014477","0.071353","0.306477","1.000138","0.047087","3.751652","0","3.981793","12480","37642027.255208","37727010.000000","37561660.000000","7227269233.000000","36317742.500000","2.132444","36338193.000000","2.133645","36297325.000000","2.131245","6973006560.000000","409.429287","6287819.869792","6850658.000000","6041254.000000","1207261415.000000","190667.859375","252741.000000","152368.000000","36608229.000000","0","0","0","0","0","0","0","0","133668864.000000","1.916947","0","0","1591296.000000","0.022821","0","0","0","0","0","0","0","0","0","0","0","0","2784768.000000","0.039936","99456.000000","0.001426","6326606.692708","6912027.000000","6063398.000000","1214708485.000000","16.807295","0.17","18.362526","0.18","16.108054","0.16","16.807295","32.27","83.192705","81.863240","83.678442","83.192705","3.000000","2","33.554432","0","5","1024.000000","1619549","24610","21842","18275","15459","5250","3226","1","0","0","0","69715","61673","503007","433438","0","0","5470","4938","1697","1188","8151","6640","37460","0","122716","0","6196","5230","326651","280297","0","0","490350","479775","99.927449","32.000000","4.000000","32.000000","0","32.000000","0","32.000000","0","65.010417","336.000000","0","12482.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","1000.000000","0","0","0","224497506.000000","3.219522","89236896.000000","450.000000","2784768.000000","0","99456.000000","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0.029211","0.037782","0.022331","5.608440","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","0","31.21","25.05","12.000000","2.286434","12.000000","2.749732","12.000000","1.825086","2304.000000","438.995350","0.219602","0.242512","0.209829","42.163582","36861708726","29478400372" diff --git a/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep b/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep new file mode 100644 index 0000000..1ed55b6 Binary files /dev/null and b/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep differ diff --git a/benchmarks/gb10-fc2-nvfp4-splitk1-profile-20260825.json b/benchmarks/gb10-fc2-nvfp4-splitk1-profile-20260825.json new file mode 100644 index 0000000..0264421 --- /dev/null +++ b/benchmarks/gb10-fc2-nvfp4-splitk1-profile-20260825.json @@ -0,0 +1,1741 @@ +{ + "mode": "profile", + "environment": { + "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", + "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", + "torch": "2.9.1+cu130", + "torch_cuda": "13.0", + "device": "NVIDIA GB10", + "device_capability": [ + 12, + 1 + ], + "driver": null, + "git_commit": null, + "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", + "checkpoint_sha256": null, + "checkpoint_hash_note": "not calculated", + "environment_switches": { + "CUDA_DEVICE_MAX_CONNECTIONS": "1", + "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", + "CUDA_HOME": "/usr/local/cuda", + "CUDA_INC_PATH": "/usr/local/cuda/include", + "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", + "CUDA_MODULE_LOADING": "EAGER", + "CUDA_VERSION": "13.0.2", + "H3_FUSED_ELEMENTWISE": "1", + "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd", + "TORCH_COMPILE_DISABLE": "0", + "TORCH_CUDA_ARCH_LIST": "12.1a", + "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" + }, + "extension": { + "cuda_version": 13000, + "cublas_version": 130100, + "stream_k_public_control": false, + "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + }, + "comfy_kitchen": "0.2.31 package without __version__" + }, + "workload": { + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "steps": 12, + "sampler_step": 1, + "seed": 440420, + "text_tokens": 100, + "tokens": 37810, + "hidden_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ] + }, + "retained_blocks": [ + 0, + 24, + 49 + ], + "immutable_cloned_block_inputs": { + "0": [ + 37810, + 5376 + ], + "24": [ + 37810, + 5376 + ], + "49": [ + 37810, + 5376 + ] + }, + "fc2_boundary": { + "block": 24, + "gate_up_shape": [ + 37810, + 28672 + ], + "activation_qdata_shape": [ + 37824, + 7168 + ], + "weight_qdata_shape": [ + 5376, + 7168 + ], + "logical_mnk": [ + 37810, + 5376, + 14336 + ], + "descriptor_mnk_after_padding": [ + 37824, + 5376, + 14336 + ], + "producer": "vortex_native_quantize_swiglu_nvfp4", + "no_bias": true + }, + "baseline_kernel_metadata": { + "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", + "descriptors": { + "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", + "block_scale_mode": "VEC16_UE4M3", + "compute_and_scale": "FP32", + "scalar_pointer_mode": "device", + "bias": null, + "beta": 0.0, + "comfy_kitchen_version": "0.2.31" + }, + "profiler_cuda_events_available": false, + "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", + "top_cuda_events": [], + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + "heuristics": [ + { + "max_workspace_bytes": 0, + "requested_count": 32, + "returned_count": 5, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 4194304, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 8388608, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 16777216, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 33554432, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 67108864, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + } + ], + "explicit_split_k_checks": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 1, + "reduction_scheme": 0 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 4 + } + } + ], + "selected": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 + }, + "profile": { + "target": "candidate", + "selected_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 + }, + "supplied_workspace_bytes": 0, + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + "errors_and_unsupported": [ + { + "feature": "Stream-K", + "supported_public_control": false, + "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + } + ] +} diff --git a/benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json b/benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json new file mode 100644 index 0000000..e5499d6 --- /dev/null +++ b/benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json @@ -0,0 +1,1769 @@ +{ + "mode": "trajectory", + "environment": { + "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", + "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", + "torch": "2.9.1+cu130", + "torch_cuda": "13.0", + "device": "NVIDIA GB10", + "device_capability": [ + 12, + 1 + ], + "driver": null, + "git_commit": null, + "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", + "checkpoint_sha256": null, + "checkpoint_hash_note": "not calculated", + "environment_switches": { + "CUDA_DEVICE_MAX_CONNECTIONS": "1", + "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", + "CUDA_HOME": "/usr/local/cuda", + "CUDA_INC_PATH": "/usr/local/cuda/include", + "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", + "CUDA_MODULE_LOADING": "EAGER", + "CUDA_VERSION": "13.0.2", + "H3_FUSED_ELEMENTWISE": "1", + "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd", + "TORCH_COMPILE_DISABLE": "0", + "TORCH_CUDA_ARCH_LIST": "12.1a", + "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" + }, + "extension": { + "cuda_version": 13000, + "cublas_version": 130100, + "stream_k_public_control": false, + "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + }, + "comfy_kitchen": "0.2.31 package without __version__" + }, + "workload": { + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "steps": 12, + "sampler_step": 1, + "seed": 440420, + "text_tokens": 100, + "tokens": 37810, + "hidden_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ] + }, + "retained_blocks": [ + 0, + 24, + 49 + ], + "immutable_cloned_block_inputs": { + "0": [ + 37810, + 5376 + ], + "24": [ + 37810, + 5376 + ], + "49": [ + 37810, + 5376 + ] + }, + "fc2_boundary": { + "block": 24, + "gate_up_shape": [ + 37810, + 28672 + ], + "activation_qdata_shape": [ + 37824, + 7168 + ], + "weight_qdata_shape": [ + 5376, + 7168 + ], + "logical_mnk": [ + 37810, + 5376, + 14336 + ], + "descriptor_mnk_after_padding": [ + 37824, + 5376, + 14336 + ], + "producer": "vortex_native_quantize_swiglu_nvfp4", + "no_bias": true + }, + "baseline_kernel_metadata": { + "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", + "descriptors": { + "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", + "block_scale_mode": "VEC16_UE4M3", + "compute_and_scale": "FP32", + "scalar_pointer_mode": "device", + "bias": null, + "beta": 0.0, + "comfy_kitchen_version": "0.2.31" + }, + "profiler_cuda_events_available": false, + "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", + "top_cuda_events": [], + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + "heuristics": [ + { + "max_workspace_bytes": 0, + "requested_count": 32, + "returned_count": 5, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 4194304, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 8388608, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 16777216, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 33554432, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 67108864, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + } + ], + "explicit_split_k_checks": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 1, + "reduction_scheme": 0 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 4 + } + } + ], + "selected": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 + }, + "trajectory": { + "steps": 12, + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "seed": 440420, + "baseline_seconds": 278.20053176092915, + "candidate_seconds": 255.37142011406831, + "improvement_percent": 8.205991376924814, + "video_parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578", + "expected_sha256": "c62d23a42972eab907ba42f93c50247ff17a9c454b4a53fe93d2e34f9fefe578" + }, + "audio_parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2", + "expected_sha256": "852005383770480a6503504e1ffec86dd1fb63a69c6400f92da18e39e0986de2" + }, + "bf16_exact": true, + "selected_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 + }, + "supplied_workspace_bytes": 0, + "all_50_fc2_calls_replaced": true, + "accepted_swiglu_producer_preserved": true, + "accepted_gate_and_residual_path_preserved": true + }, + "errors_and_unsupported": [ + { + "feature": "Stream-K", + "supported_public_control": false, + "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + } + ] +} diff --git a/benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json b/benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json new file mode 100644 index 0000000..364fd0a --- /dev/null +++ b/benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json @@ -0,0 +1,1769 @@ +{ + "mode": "trajectory", + "environment": { + "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", + "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", + "torch": "2.9.1+cu130", + "torch_cuda": "13.0", + "device": "NVIDIA GB10", + "device_capability": [ + 12, + 1 + ], + "driver": null, + "git_commit": null, + "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", + "checkpoint_sha256": null, + "checkpoint_hash_note": "not calculated", + "environment_switches": { + "CUDA_DEVICE_MAX_CONNECTIONS": "1", + "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", + "CUDA_HOME": "/usr/local/cuda", + "CUDA_INC_PATH": "/usr/local/cuda/include", + "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", + "CUDA_MODULE_LOADING": "EAGER", + "CUDA_VERSION": "13.0.2", + "H3_FUSED_ELEMENTWISE": "1", + "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", + "H3_NVFP4_MODULATE_FUSION": "1", + "H3_NVFP4_SCALE_BACKEND": "vortex", + "H3_NVFP4_SCALE_VERSION": "1", + "H3_NVFP4_SWIGLU_FUSION": "1", + "H3_SAGE_QKV_LAYOUT": "strided_nhd", + "TORCH_COMPILE_DISABLE": "0", + "TORCH_CUDA_ARCH_LIST": "12.1a", + "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" + }, + "extension": { + "cuda_version": 13000, + "cublas_version": 130100, + "stream_k_public_control": false, + "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + }, + "comfy_kitchen": "0.2.31 package without __version__" + }, + "workload": { + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "steps": 2, + "sampler_step": 1, + "seed": 440420, + "text_tokens": 100, + "tokens": 37810, + "hidden_shape": [ + 37810, + 5376 + ], + "segments": [ + [ + 0, + 100, + 1 + ], + [ + 100, + 514, + 2 + ], + [ + 514, + 37810, + 0 + ] + ] + }, + "retained_blocks": [ + 0, + 24, + 49 + ], + "immutable_cloned_block_inputs": { + "0": [ + 37810, + 5376 + ], + "24": [ + 37810, + 5376 + ], + "49": [ + 37810, + 5376 + ] + }, + "fc2_boundary": { + "block": 24, + "gate_up_shape": [ + 37810, + 28672 + ], + "activation_qdata_shape": [ + 37824, + 7168 + ], + "weight_qdata_shape": [ + 5376, + 7168 + ], + "logical_mnk": [ + 37810, + 5376, + 14336 + ], + "descriptor_mnk_after_padding": [ + 37824, + 5376, + 14336 + ], + "producer": "vortex_native_quantize_swiglu_nvfp4", + "no_bias": true + }, + "baseline_kernel_metadata": { + "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", + "descriptors": { + "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", + "block_scale_mode": "VEC16_UE4M3", + "compute_and_scale": "FP32", + "scalar_pointer_mode": "device", + "bias": null, + "beta": 0.0, + "comfy_kitchen_version": "0.2.31" + }, + "profiler_cuda_events_available": false, + "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", + "top_cuda_events": [], + "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" + }, + "heuristics": [ + { + "max_workspace_bytes": 0, + "requested_count": 32, + "returned_count": 5, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 4194304, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 8388608, + "requested_count": 32, + "returned_count": 7, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 16777216, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 33554432, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + }, + { + "max_workspace_bytes": 67108864, + "requested_count": 32, + "returned_count": 6, + "algorithms": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": -2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + } + } + ] + } + ], + "explicit_split_k_checks": [ + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 1.0, + "state": 0, + "api_status": 0, + "valid": true, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 1, + "reduction_scheme": 0 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 2, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 2, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 4, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 4, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 8, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 8, + "reduction_scheme": 4 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 2, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 2 + } + }, + { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 16, + "reduction_scheme": 4, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "required_workspace_bytes": 0, + "waves": 0.0, + "state": 15, + "api_status": 15, + "valid": false, + "capabilities": { + "split_k_support": 1, + "reduction_scheme_mask": 6, + "cta_swizzle_support": 0, + "custom_option_max": 0, + "strided_batch_support": 1, + "out_of_place_result_support": 1, + "tile_ids": [ + 20 + ], + "stages_ids": [ + 37 + ], + "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." + }, + "requested_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "custom_option": 0, + "cta_swizzle": 0, + "inner_shape": null, + "cluster_shape": null, + "split_k": 16, + "reduction_scheme": 4 + } + } + ], + "selected": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 + }, + "trajectory": { + "steps": 2, + "resolution": [ + 1344, + 768 + ], + "frames": 124, + "seed": 440420, + "baseline_seconds": 45.9624265920138, + "candidate_seconds": 42.51584641600493, + "improvement_percent": 7.49869062963644, + "video_parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "7d5e50d8f0ee9a639feede8a0e6aea1bd160087634ecc7ae80d0ef5b96fd8e78", + "expected_sha256": "7d5e50d8f0ee9a639feede8a0e6aea1bd160087634ecc7ae80d0ef5b96fd8e78" + }, + "audio_parity": { + "bf16_exact": true, + "different_elements": 0, + "max_abs": 0.0, + "mean_abs": 0.0, + "actual_sha256": "df518aa8c645ec45e7cc46be58a941790f9766b420adf2cdcbaf14ef8b0e5605", + "expected_sha256": "df518aa8c645ec45e7cc46be58a941790f9766b420adf2cdcbaf14ef8b0e5605" + }, + "bf16_exact": true, + "selected_config": { + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 + }, + "supplied_workspace_bytes": 0, + "all_50_fc2_calls_replaced": true, + "accepted_swiglu_producer_preserved": true, + "accepted_gate_and_residual_path_preserved": true + }, + "errors_and_unsupported": [ + { + "feature": "Stream-K", + "supported_public_control": false, + "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." + } + ] +} diff --git a/research/artifact_manifest.json b/research/artifact_manifest.json index b916e25..2dfd7f2 100644 --- a/research/artifact_manifest.json +++ b/research/artifact_manifest.json @@ -2,39 +2,199 @@ "metadata": { "algorithm": "SHA-256", "generated_date": "2026-08-25", - "scope_note": "Local selected scope is the established 167-file set. Spark records are the complete reproducible current top-level benchmark output set; the audit-retained 279-file/665950155-byte aggregate cannot be reconstructed because its path list was not retained.", + "scope_note": "Local selected scope is the established 167-file set plus 10 retained FC2 scheduling artifacts. Spark records are the complete reproducible current top-level benchmark output set; the audit-retained 279-file/665950155-byte aggregate cannot be reconstructed because its path list was not retained.", "summary": { "local": { - "record_count": 167, - "size_bytes": 210388688, - "expected_record_count": 167, - "expected_size_bytes": 210388688, + "record_count": 177, + "size_bytes": 230157548, + "expected_record_count": 177, + "expected_size_bytes": 230157548, "reconciled": true }, "spark": { - "record_count": 280, - "size_bytes": 547708583, + "record_count": 290, + "size_bytes": 567477443, "expected_record_count": 279, "expected_size_bytes": 665950155, "reconciled": false, - "record_count_delta": 1, - "size_bytes_delta": -118241572 + "record_count_delta": 11, + "size_bytes_delta": -98472712 }, "total": { - "record_count": 447, - "size_bytes": 758097271 + "record_count": 467, + "size_bytes": 797634991 } }, "json_reconciliation": { - "identical": 106, + "identical": 112, "mismatches": 2, "local_only": 40, "spark_only": 107, - "local_total": 148, - "spark_total": 215 + "local_total": 154, + "spark_total": 221 } }, "artifacts": [ + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv", + "size_bytes": 164324, + "sha256": "e5bfd39f1512737c0852975de319b502e2ce20677e2a51ad0ce50656923634a5", + "artifact_class": "nsight_csv_export", + "location": "benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep", + "size_bytes": 13364977, + "sha256": "975eea7110a19b472ef8cd29ec1c72f629b0f97e224080acc50a43b371bc290d", + "artifact_class": "nsight_compute_report", + "location": "benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-baseline-profile-20260825.json", + "size_bytes": 51924, + "sha256": "082e195a34a0a4738cd515dcc7db69e59d8bef22afcb49e73a91d21cb0ccaff0", + "artifact_class": "benchmark_json", + "location": "benchmarks/gb10-fc2-nvfp4-baseline-profile-20260825.json" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json", + "size_bytes": 61451, + "sha256": "d1051463765052c83494f417b40b9fe9eb389eb4ebf213ff7ff491991d569d8c", + "artifact_class": "benchmark_json", + "location": "benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json", + "size_bytes": 64494, + "sha256": "c762e9394f6ab6dd1edfd13097a7fa242caa4540fb3d71c3b1035b3edbcd3261", + "artifact_class": "benchmark_json", + "location": "benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv", + "size_bytes": 148497, + "sha256": "2a9da47628571fd430a8c0ba4fce3547d8e12dec3b443b101a3a8baa4bdf8e8f", + "artifact_class": "nsight_csv_export", + "location": "benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep", + "size_bytes": 5755084, + "sha256": "b368a14dcd96f859dbad9e7a8fa10bc33bf6373dbea8972b6685a2d83b68dfae", + "artifact_class": "nsight_compute_report", + "location": "benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-splitk1-profile-20260825.json", + "size_bytes": 52089, + "sha256": "7b2282b6381e90d07d7175fe3a8f1aa03c7d85a3c6d25cab2350abf6559800d8", + "artifact_class": "benchmark_json", + "location": "benchmarks/gb10-fc2-nvfp4-splitk1-profile-20260825.json" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json", + "size_bytes": 53013, + "sha256": "075c3aecbe6f77cc28f07a8f656de044e437c94a9fc71068c0ac9e8f3082972c", + "artifact_class": "benchmark_json", + "location": "benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json" + }, + { + "scope": "local", + "path": "benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json", + "size_bytes": 53007, + "sha256": "a9cc0b229dcc7e37705d94705b1e607092d9aee7e3873787eff539074f46228f", + "artifact_class": "benchmark_json", + "location": "benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv", + "size_bytes": 164324, + "sha256": "e5bfd39f1512737c0852975de319b502e2ce20677e2a51ad0ce50656923634a5", + "artifact_class": "nsight_csv_export", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep", + "size_bytes": 13364977, + "sha256": "975eea7110a19b472ef8cd29ec1c72f629b0f97e224080acc50a43b371bc290d", + "artifact_class": "nsight_compute_report", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-baseline-profile-20260825.json", + "size_bytes": 51924, + "sha256": "082e195a34a0a4738cd515dcc7db69e59d8bef22afcb49e73a91d21cb0ccaff0", + "artifact_class": "benchmark_json", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-baseline-profile-20260825.json" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json", + "size_bytes": 61451, + "sha256": "d1051463765052c83494f417b40b9fe9eb389eb4ebf213ff7ff491991d569d8c", + "artifact_class": "benchmark_json", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json", + "size_bytes": 64494, + "sha256": "c762e9394f6ab6dd1edfd13097a7fa242caa4540fb3d71c3b1035b3edbcd3261", + "artifact_class": "benchmark_json", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv", + "size_bytes": 148497, + "sha256": "2a9da47628571fd430a8c0ba4fce3547d8e12dec3b443b101a3a8baa4bdf8e8f", + "artifact_class": "nsight_csv_export", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep", + "size_bytes": 5755084, + "sha256": "b368a14dcd96f859dbad9e7a8fa10bc33bf6373dbea8972b6685a2d83b68dfae", + "artifact_class": "nsight_compute_report", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-splitk1-profile-20260825.json", + "size_bytes": 52089, + "sha256": "7b2282b6381e90d07d7175fe3a8f1aa03c7d85a3c6d25cab2350abf6559800d8", + "artifact_class": "benchmark_json", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-splitk1-profile-20260825.json" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json", + "size_bytes": 53013, + "sha256": "075c3aecbe6f77cc28f07a8f656de044e437c94a9fc71068c0ac9e8f3082972c", + "artifact_class": "benchmark_json", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json" + }, + { + "scope": "spark", + "path": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json", + "size_bytes": 53007, + "sha256": "a9cc0b229dcc7e37705d94705b1e607092d9aee7e3873787eff539074f46228f", + "artifact_class": "benchmark_json", + "location": "/home/daniel/StoryStudioAssets/H3-output/h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json" + }, { "scope": "local", "path": "benchmarks/audio-dialogue-format-sweep-sage2-seeds440420-440429.json", diff --git a/research/experiment_registry.json b/research/experiment_registry.json index ffd85da..ff061b6 100644 --- a/research/experiment_registry.json +++ b/research/experiment_registry.json @@ -500,6 +500,28 @@ "reproducer_commands": [], "timestamp": null, "evidence_missing": ["No accepted alternative FC2 reduction implementation"], "production_behavior": "FC2 retains the Comfy/CUBLAS fallback.", "source_recovery": "P0 artifacts and fallback policy are retained in the roadmap." }, + { + "id": "fc2-cublaslt-splitk1-schedule", + "name": "FC2 cuBLASLt public split-K-1 schedule", + "family": "nvfp4-library-scheduling", + "status": "research_retained", + "hypothesis": "A documented cuBLASLt schedule can preserve the exact FC2 reduction result while avoiding the production heuristic's traffic and synchronization regression.", + "implementation_strategy": "Reproduce the exact Comfy Kitchen descriptors in an isolated extension, enumerate checked cuBLASLt algorithms, and compare one selected public split-K-1 schedule against the accepted FC2 path.", + "source_locations": ["research/fc2_nvfp4_scheduling/README.md", "research/fc2_nvfp4_scheduling/RESULTS.md", "research/fc2_nvfp4_scheduling/fc2_nvfp4_lt.cpp", "tools/benchmark_fc2_nvfp4_algorithms.py"], + "active_source_location": "research/fc2_nvfp4_scheduling/fc2_nvfp4_lt.cpp", + "commit_hash": null, + "benchmark_artifacts": [{"path": "benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json", "exists": true}, {"path": "benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json", "exists": true}, {"path": "benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json", "exists": true}, {"path": "benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json", "exists": true}], + "profiler_artifacts": [{"path": "benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep", "exists": true}, {"path": "benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep", "exists": true}, {"path": "benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv", "exists": true}, {"path": "benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv", "exists": true}], + "environment": {"gpu": "NVIDIA GB10", "cuda": "13", "driver": null, "pytorch": null, "triton": null, "container_image": "sha256:1d340e14cb6fc45ccfdbe63dde8db2a2b3aea94b493702c8a08e1f8d5b4f7b83", "commit_context": "isolated extension; production source and dispatch unchanged"}, + "metrics": {"fc2_p50_ms_baseline": 53.618, "fc2_p50_ms_candidate": 15.636, "block_improvement_pct": [8.16, 8.63, 8.88], "two_step_improvement_pct": 7.50, "canonical_12_step_improvement_pct": 8.21, "trajectory_measurement_note": "single baseline-then-candidate paired runs", "l2_hit_rate_pct_baseline": 53.32, "l2_hit_rate_pct_candidate": 91.10, "l2_read_miss_reduction_bytes": 9853094784}, + "correctness_evidence": ["Selected FC2 output is byte-exact.", "Blocks 0, 24, and 49 are byte-exact against paired baseline and retained traversal.", "Two-step and canonical 12-step video and audio latents are byte-exact."], + "decision_rationale": "The library schedule passed all exactness and performance gates, explaining the prior traffic amplification without requiring a custom kernel. It is retained pending explicit production integration and deployment validation.", + "reproducer_commands": ["python tools/benchmark_fc2_nvfp4_algorithms.py --mode sweep --rounds 20 --workspace-bytes 67108864", "python tools/benchmark_fc2_nvfp4_algorithms.py --mode block-gate --candidate research/fc2_nvfp4_scheduling/candidate_splitk1.json --rounds 20 --workspace-bytes 0"], + "timestamp": "2026-08-25", + "evidence_missing": ["Production Nvfp4Linear integration", "Production deployment smoke", "Portable validation outside GB10/SM121"], + "production_behavior": "Not dispatched. Production still uses Comfy Kitchen's heuristic-selected FC2 path.", + "source_recovery": "The direct cuBLASLt extension, benchmark harness, selected candidate, raw NCU reports, and full gate artifacts are retained in this checkout." + }, { "id": "layout-direct-temporal-output", "name": "Direct-to-temporal output", diff --git a/research/fc2_nvfp4_scheduling/README.md b/research/fc2_nvfp4_scheduling/README.md new file mode 100644 index 0000000..800a43c --- /dev/null +++ b/research/fc2_nvfp4_scheduling/README.md @@ -0,0 +1,206 @@ +# FC2 NVFP4 Library Scheduling + +This directory contains the first isolated library-scheduling study for H3 FC2. +It does not register a PyTorch operator, alter `Nvfp4Linear`, or participate in +production dispatch. The extension is loaded only by +`tools/benchmark_fc2_nvfp4_algorithms.py`. + +The completed measurements and decision are in `RESULTS.md`. Algorithm 70 with +public split-K 1 is byte-exact and passes the block, two-step, and canonical +12-step gates, but remains research-only until production integration and +deployment validation are performed. + +## Exact Operation + +The extension reproduces Comfy Kitchen 0.2.31's +`cublas_gemm_nvfp4.cu` descriptors for row-major packed activation `[M,K]` +times packed weight `[N,K]` to BF16 `[M,N]`: + +- cuBLASLt sees column-major `weight.T @ activation`, so Lt `m=N`, `n=M`, and + `k=K`; the output storage is the row-major `[M,N]` view. +- A and B are `CUDA_R_4F_E2M1`, with + `CUBLASLT_MATMUL_MATRIX_SCALE_VEC16_UE4M3` block scales. +- Compute, scalar scale type, alpha, and beta are FP32. Alpha and zero beta are + device pointers. +- Output is BF16, epilogue is default, and there is no bias. +- The activation producer is the deployed, accepted + `vortex_native_quantize_swiglu_nvfp4`. Alpha is its FP32 tensor scale times + FC2's FP32 weight tensor scale. + +The canonical 37,810-row input pads to 37,824 rows in the accepted producer. +The report distinguishes logical `[M,N,K]` from descriptor/padded dimensions; +comparison slices back to the logical output exactly as Comfy Kitchen does. + +## Current Baseline + +The accepted production path fuses exact BF16 SwiGLU into NVFP4 production and +retains Comfy Kitchen's cuBLASLt FC2 GEMM. Existing validated metadata reports: + +- Producer tensor scale, QDATA, and SFA are byte-identical at blocks 0, 24, and + 49. +- Canonical block FC2 input is logically `37,810 x 14,336` BF16 before packing. +- Fully fused block-24 profiling attributes 26.87% of kernel time to all four + NVFP4 GEMMs; this is not claimed as an FC2-only percentage. +- Spark's accepted image is + `sha256:1d340e14cb6fc45ccfdbe63dde8db2a2b3aea94b493702c8a08e1f8d5b4f7b83` + and enables `H3_NVFP4_SCALE_BACKEND=vortex`, + `H3_FUSED_ELEMENTWISE=1`, `H3_NVFP4_MODULATE_FUSION=1`, and + `H3_NVFP4_SWIGLU_FUSION=1`. + +Every run also records a fresh profiler-derived baseline kernel list. Kernel +names and times in that list are run metadata, not hard-coded claims. + +## Commands + +Run inside image `1d340e14cb6f...` from the Spark checkout, without starting the +hot service. Preserve its production environment switches, for example: + +```bash +export H3_NVFP4_SCALE_BACKEND=vortex H3_NVFP4_SCALE_VERSION=1 +export H3_FUSED_ELEMENTWISE=1 H3_NVFP4_MODULATE_FUSION=1 +export H3_NVFP4_SWIGLU_FUSION=1 H3_SAGE_QKV_LAYOUT=strided_nhd +``` + +Compile only: + +```bash +python tools/benchmark_fc2_nvfp4_algorithms.py --mode compile --verbose-build \ + --build-directory /tmp/fc2-nvfp4-build +``` + +Default characterization enumerates workspace budgets 0, 4, 8, 16, 32, and +64 MiB, checks a bounded set of public split-K configurations, and times one +valid probe with alternating AB/BA order: + +```bash +python tools/benchmark_fc2_nvfp4_algorithms.py --mode characterize \ + --build-directory /tmp/fc2-nvfp4-build \ + --output /output/h3-blackwell-runtime/benchmarks/fc2-nvfp4-characterize.json +``` + +The same workload can come from an existing capture: + +```bash +python tools/benchmark_fc2_nvfp4_algorithms.py --mode characterize \ + --capture /artifacts/capture --build-directory /tmp/fc2-nvfp4-build \ + --output /output/h3-blackwell-runtime/benchmarks/fc2-nvfp4-capture.json +``` + +A bounded timing sweep defaults to eight candidates, not a full combinatorial +gate: + +```bash +python tools/benchmark_fc2_nvfp4_algorithms.py --mode sweep \ + --candidate-limit 8 --rounds 6 --workspace-bytes 33554432 \ + --build-directory /tmp/fc2-nvfp4-build \ + --output /output/h3-blackwell-runtime/benchmarks/fc2-nvfp4-sweep.json +``` + +Re-run one selected candidate. `--candidate` accepts an inline JSON object, a +JSON file containing one object, or `enumerate::`: + +```bash +python tools/benchmark_fc2_nvfp4_algorithms.py --mode selected \ + --candidate '{"algorithm_id":23,"tile_id":42,"stages_id":35,"split_k":1,"reduction_scheme":0}' \ + --workspace-bytes 33554432 --build-directory /tmp/fc2-nvfp4-build \ + --output /output/h3-blackwell-runtime/benchmarks/fc2-nvfp4-selected.json +``` + +NCU capture uses `cudaProfilerStart/Stop`; only the selected FC2 cuBLASLt call +is inside the range: + +```bash +ncu --target-processes all --profile-from-start off --set full \ + --export /output/h3-blackwell-runtime/benchmarks/fc2-nvfp4-selected \ + python tools/benchmark_fc2_nvfp4_algorithms.py --mode profile \ + --candidate /tmp/fc2-candidate.json --workspace-bytes 33554432 \ + --build-directory /tmp/fc2-nvfp4-build \ + --output /output/h3-blackwell-runtime/benchmarks/fc2-nvfp4-ncu.json +``` + +Complete block gate for blocks 0, 24, and 49: + +```bash +python tools/benchmark_fc2_nvfp4_algorithms.py --mode block-gate \ + --candidate /tmp/fc2-candidate.json --rounds 6 \ + --workspace-bytes 33554432 --build-directory /tmp/fc2-nvfp4-build \ + --output /output/h3-blackwell-runtime/benchmarks/fc2-nvfp4-block-gate.json +``` + +Block-gate monkeypatches only `block.mlp.fc2.forward_swiglu`. It still invokes +the accepted SwiGLU producer on the actual FC1 result. Norm, FC1, modulation, +gate, residual, attention, and all other code remain the accepted path. + +## Requested NCU Traffic Counters + +Use `ncu --query-metrics` on the installed NCU version before replacing `--set +full` with an explicit list. Request the available equivalents of: + +```text +gpu__time_duration.sum +dram__bytes_read.sum +dram__bytes_write.sum +lts__t_bytes.sum +lts__t_sector_hit_rate.pct +sm__throughput.avg.pct_of_peak_sustained_elapsed +smsp__inst_executed.sum +smsp__pipe_tensor_cycles_active.avg.pct_of_peak_sustained_active +launch__registers_per_thread +launch__shared_mem_per_block_allocated +``` + +Metric spelling and availability vary with NCU and GB10. Missing counters must +be reported as unavailable rather than silently substituted. Also retain +kernel duration, grid/cluster dimensions, achieved occupancy, waves, required +workspace, and the extension's tile/stage/inner/cluster configuration. + +## Validation Ladder + +1. Compile against the image's CUDA 13 headers and print `build_info()`. +2. Characterize heuristics and verify every executed candidate passes + `cublasLtMatmulAlgoCheck` within caller-supplied workspace. +3. Compare direct-probe candidate FC2 against a cloned + `fc2.forward_swiglu(gate_up)` output byte-for-byte. +4. Alternate baseline/candidate AB and BA each round; record p50, p95, dense + TFLOP/s, producer-plus-FC2 time, workspace, and output SHA-256. +5. NCU one FC2 call and inspect traffic/resource counters. +6. Gate complete blocks 0, 24, and 49 with only FC2 GEMM replaced. Require + BF16 exactness against both the paired baseline and retained traversal. +7. Only after those gates should a separate, explicitly authorized trajectory + experiment be considered. + +That trajectory authorization was granted for the retained candidate. Both the +two-step and canonical 12-step gates passed with byte-exact video and audio +latents; see `RESULTS.md` and the linked benchmark artifacts. + +## Prohibited Experiments + +- Do not alter production source, configuration, dispatch, or existing + research files. +- Do not expose this extension through `Nvfp4Linear` or any production operator. +- Do not start or perturb the hot service. +- Do not add bias, accumulation, host scalars, another output dtype, a different + producer, or approximate validation to this study. +- Do not claim a checkpoint hash unless `--checkpoint-sha256` actually computes + it. +- Do not label undocumented behavior Stream-K. CUDA 13 exposes no documented + public `cublasLtMatmulAlgoConfig` attribute that directly selects Stream-K; + the report states that limitation explicitly. +- Do not promote a direct-GEMM result without the complete block gate and later + separately authorized trajectory validation. + +## Limitations + +- Heuristics are library, driver, GPU, shape, and workspace specific. +- Explicit split-K is attempted only through documented `AlgoInit`, + `ConfigSet`, and `AlgoCheck`. Unsupported factors/reduction schemes are + recorded; they are not emulated. +- Negative `SPLITK_NUM` values returned by the heuristic are retained as raw + signed library sentinels. They are not interpreted or labeled as Stream-K. +- Characterization deliberately bounds explicit checks and timing candidates. + Increase limits consciously because each full canonical FC2 call is costly. +- Caller-owned output and workspace are reused during timing. The accepted + producer still owns its quantized activation allocations. +- A capture supplies packed-denoiser inputs, not pre-captured FC2 operands; the + harness traverses the loaded H3 model once to construct exact current + boundaries. diff --git a/research/fc2_nvfp4_scheduling/RESULTS.md b/research/fc2_nvfp4_scheduling/RESULTS.md new file mode 100644 index 0000000..761bd1b --- /dev/null +++ b/research/fc2_nvfp4_scheduling/RESULTS.md @@ -0,0 +1,158 @@ +# FC2 NVFP4 Scheduling Results + +Status: research retained; validated as a canonical-shape production-integration +candidate, but not wired into production dispatch. + +## Decision + +The canonical GB10 FC2 slowdown is caused by the cuBLASLt heuristic selecting a +`_stream_k` kernel with poor traversal locality and heavy synchronization +polling. A documented public split-K-1 configuration is byte-exact and removes +the regression. Because this library candidate passed the complete validation +ladder, the custom persistent-kernel branch is closed. + +The retained candidate is: + +| Field | Value | +| --- | ---: | +| Algorithm ID | `70` | +| Tile ID | `20` | +| Stages ID | `37` | +| Split-K | `1` | +| Reduction scheme | `0` | +| Custom option / CTA swizzle | `0 / 0` | +| Required workspace | `0 bytes` | +| Supplied workspace | `64 MiB` sweep; `0 bytes` block/trajectory gates | + +The baseline heuristic's negative split-K value is an undocumented library +sentinel. CUDA 13 has no documented public attribute that directly selects +Stream-K, so the sentinel itself is not interpreted as a public Stream-K +control. The launched baseline symbol does end in `_stream_k`. + +## Canonical Operation + +- Logical GEMM: `M=37,810, N=5,376, K=14,336`. +- Descriptor GEMM after accepted producer padding: `M=37,824`. +- Inputs: packed NVFP4 E2M1 activation and weight with vector-16 UE4M3 scales. +- Accumulation/scalars: FP32; output: BF16. +- Accepted activation producer: `vortex_native_quantize_swiglu_nvfp4`. +- Image: `sha256:1d340e14cb6fc45ccfdbe63dde8db2a2b3aea94b493702c8a08e1f8d5b4f7b83`. + +## Timing + +The 20-round isolated sweep alternated AB/BA order in one loaded model. + +| Measurement | Baseline | Split-K-1 | Result | +| --- | ---: | ---: | ---: | +| FC2 p50 | `53.618 ms` | `15.636 ms` | `3.43x` | +| Dense throughput p50 | `108.70 TFLOP/s` | `372.74 TFLOP/s` | `3.43x` | +| Producer + FC2 p50 | `76.737 ms` | `38.823 ms` | `49.41%` faster | + +The negative-sentinel candidate reproduced the production-equivalent baseline +within noise. Explicit split-K factors `2`, `4`, `8`, and `16` were rejected by +`cublasLtMatmulAlgoCheck` for the tested configuration space. + +Complete 20-round block gates replaced only +`block.mlp.fc2.forward_swiglu`: + +| Block | Baseline p50 | Candidate p50 | Improvement | +| ---: | ---: | ---: | ---: | +| 0 | `457.891 ms` | `420.536 ms` | `8.16%` | +| 24 | `459.928 ms` | `420.243 ms` | `8.63%` | +| 49 | `458.054 ms` | `417.364 ms` | `8.88%` | + +All block outputs were byte-exact against both the paired baseline and retained +traversal. + +| Trajectory | Baseline | Candidate | Improvement | Correctness | +| --- | ---: | ---: | ---: | --- | +| Two-step | `45.962 s` | `42.516 s` | `7.50%` | video/audio BF16 exact | +| Canonical 12-step | `278.201 s` | `255.371 s` | `8.21%` | video/audio BF16 exact | + +Trajectory values are single paired runs in baseline-then-candidate +order, not repeated medians. Their approximately `38 ms` per-FC2 savings agree +with the alternating isolated and complete-block gates, but the precise +end-to-end percentages retain run-order uncertainty. + +The 12-step saving of `22.829 s` matches approximately 600 FC2 invocations +multiplied by the isolated roughly `38 ms` saving. + +## Traffic Attribution + +The one-pass distinct-data footprint is: + +| Category | Bytes | +| --- | ---: | +| Activation QDATA + scales | `305,012,736` | +| Weight QDATA + scales | `43,352,064` | +| Padded BF16 output | `406,683,648` | +| Total | `755,048,448` | + +Both schedules issue the same nominal operand requests: + +| Requested category | Bytes | Geometric reuse | +| --- | ---: | ---: | +| Weight QDATA + scales | `12,832,210,944` | `296x` | +| Activation QDATA + scales | `12,832,210,944` | about `42.1x` | +| Total L2 operand reads | `25,664,424,960` | unchanged | + +The schedule does not eliminate CTA-level rereading. It changes whether those +rereads remain cache-resident: + +| NCU metric | Baseline `_stream_k` | Split-K-1 | +| --- | ---: | ---: | +| Main-kernel duration | `55.057 ms` | `17.031 ms` | +| Combined L2 hit rate | `53.32%` | `91.10%` | +| L2 read-miss bytes | `11.762 GB` | `1.909 GB` | +| L2 miss-byte proxy | `12.169 GB` | `2.316 GB` | +| Sysmem traffic proxy | `12.176 GB` | `2.327 GB` | +| Off-chip amplification | `16.12x` | `3.07x` | +| SM throughput | `25.11%` | `81.79%` | +| Tensor-pipe active share | `25.01%` | `82.42%` | +| Eligible warps/scheduler | `0.073` | `0.220` | +| Issue rate | `0.057` | `0.168` | + +GB10 does not expose the usual discrete-GPU DRAM byte counters in these +captures. L2 misses and the reported sysmem fill/write sectors are used as the +off-chip proxy. + +The `9.853 GB` reduction in L2 read misses is avoided operand rereading. NCU +aggregates the two TMA input descriptors, so it cannot defensibly assign exact +miss-byte totals separately to weights and activations. The request geometry +does prove that each input family accounts for `12.832 GB` of requested reads. + +There is no material global partial-accumulator or output-reduction traffic: + +- Candidate workspace is zero. +- Neither kernel issues global atomics or global reduction operations. +- No auxiliary reduction kernel is launched. +- Candidate writes exactly one padded BF16 output; baseline writes only + `73,728` bytes more. +- Baseline-only local stack traffic is about `22.35 MB` at L1 and almost none + reaches sysmem. + +The baseline's dominant scheduler stall is sleeping. Source-correlated samples +land in `NANOSLEEP.SYNCS` polling around synchronization phase checks. Launch +resources are otherwise the same: 12,432 CTAs, 384 threads/CTA, 168 registers +per thread, 89,088 bytes allocated shared memory, and 25% theoretical +occupancy. The gain therefore comes from traversal locality and reduced +synchronization waiting, not occupancy or a reduction workspace. + +## Evidence + +- `benchmarks/gb10-fc2-nvfp4-library-sweep-20260825.json` +- `benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json` +- `benchmarks/gb10-fc2-nvfp4-trajectory-2step-20260825.json` +- `benchmarks/gb10-fc2-nvfp4-trajectory-12step-20260825.json` +- `benchmarks/gb10-fc2-nvfp4-baseline-profile-20260825.json` +- `benchmarks/gb10-fc2-nvfp4-splitk1-profile-20260825.json` +- `benchmarks/gb10-fc2-nvfp4-baseline-20260825.csv` +- `benchmarks/gb10-fc2-nvfp4-splitk1-20260825.csv` +- `benchmarks/gb10-fc2-nvfp4-baseline-20260825.ncu-rep` +- `benchmarks/gb10-fc2-nvfp4-splitk1-20260825.ncu-rep` + +Production source and configuration were not changed by this study. The +candidate is validated only for the canonical descriptor shape and current +GB10/CUDA-library combination. Promotion still requires shape-specific +`AlgoCheck` with a safe fallback, integration behind the existing +`Nvfp4Linear` boundary, broader shape tests, and deployment validation. diff --git a/research/fc2_nvfp4_scheduling/__init__.py b/research/fc2_nvfp4_scheduling/__init__.py new file mode 100644 index 0000000..cba87aa --- /dev/null +++ b/research/fc2_nvfp4_scheduling/__init__.py @@ -0,0 +1 @@ +"""Isolated cuBLASLt NVFP4 FC2 scheduling research.""" diff --git a/research/fc2_nvfp4_scheduling/candidate_splitk1.json b/research/fc2_nvfp4_scheduling/candidate_splitk1.json new file mode 100644 index 0000000..e1ce3ec --- /dev/null +++ b/research/fc2_nvfp4_scheduling/candidate_splitk1.json @@ -0,0 +1,9 @@ +{ + "algorithm_id": 70, + "tile_id": 20, + "stages_id": 37, + "split_k": 1, + "reduction_scheme": 0, + "custom_option": 0, + "cta_swizzle": 0 +} diff --git a/research/fc2_nvfp4_scheduling/fc2_nvfp4_lt.cpp b/research/fc2_nvfp4_scheduling/fc2_nvfp4_lt.cpp new file mode 100644 index 0000000..3cf79d8 --- /dev/null +++ b/research/fc2_nvfp4_scheduling/fc2_nvfp4_lt.cpp @@ -0,0 +1,388 @@ +#include + +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +namespace py = pybind11; + +#define LT_CHECK(call) \ + do { \ + const cublasStatus_t status_ = (call); \ + if (status_ != CUBLAS_STATUS_SUCCESS) \ + throw std::runtime_error(std::string(#call) + " failed: " + \ + std::to_string(static_cast(status_))); \ + } while (0) + +namespace { + +thread_local cublasLtHandle_t handle = nullptr; + +cublasLtHandle_t get_handle() { + if (!handle) LT_CHECK(cublasLtCreate(&handle)); + return handle; +} + +void require_cuda_contiguous(const torch::Tensor& tensor, const char* name) { + TORCH_CHECK(tensor.is_cuda(), name, " must be CUDA"); + TORCH_CHECK(tensor.is_contiguous(), name, " must be contiguous"); +} + +void require_same_device(const torch::Tensor& tensor, + const torch::Tensor& reference, + const char* name) { + TORCH_CHECK(tensor.device() == reference.device(), name, + " must be on the same CUDA device as activation_qdata"); +} + +struct Problem { + cublasLtMatmulDesc_t operation = nullptr; + cublasLtMatrixLayout_t a = nullptr, b = nullptr, c = nullptr, d = nullptr; + + Problem(const torch::Tensor& activation_qdata, + const torch::Tensor& activation_block_scale, + const torch::Tensor& weight_qdata, + const torch::Tensor& weight_block_scale) { + require_cuda_contiguous(activation_qdata, "activation_qdata"); + require_cuda_contiguous(activation_block_scale, "activation_block_scale"); + require_cuda_contiguous(weight_qdata, "weight_qdata"); + require_cuda_contiguous(weight_block_scale, "weight_block_scale"); + require_same_device(activation_block_scale, activation_qdata, + "activation_block_scale"); + require_same_device(weight_qdata, activation_qdata, "weight_qdata"); + require_same_device(weight_block_scale, activation_qdata, + "weight_block_scale"); + TORCH_CHECK(activation_qdata.scalar_type() == at::kByte && + weight_qdata.scalar_type() == at::kByte, + "packed NVFP4 operands must use uint8 storage"); + TORCH_CHECK(activation_qdata.dim() == 2 && weight_qdata.dim() == 2, + "packed NVFP4 operands must be rank two"); + TORCH_CHECK(activation_block_scale.element_size() == 1 && + weight_block_scale.element_size() == 1, + "NVFP4 block scales must use one-byte E4M3 storage"); + TORCH_CHECK(activation_qdata.size(1) == weight_qdata.size(1), + "packed K dimensions differ"); + + // This is deliberately the same column-major reinterpretation used by + // Comfy Kitchen 0.2.31: weight is Lt A, activation is Lt B, and D is D^T. + const int64_t m = weight_qdata.size(0); // row-major N + const int64_t n = activation_qdata.size(0); // row-major padded M + const int64_t k = activation_qdata.size(1) * 2; + TORCH_CHECK(activation_block_scale.numel() >= n * (k / 16), + "activation_block_scale storage is too small"); + TORCH_CHECK(weight_block_scale.numel() >= m * (k / 16), + "weight_block_scale storage is too small"); + LT_CHECK(cublasLtMatmulDescCreate(&operation, CUBLAS_COMPUTE_32F, + CUDA_R_32F)); + cublasLtMatmulMatrixScale_t scale_mode = + CUBLASLT_MATMUL_MATRIX_SCALE_VEC16_UE4M3; + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_A_SCALE_MODE, &scale_mode, + sizeof(scale_mode))); + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_B_SCALE_MODE, &scale_mode, + sizeof(scale_mode))); + const cublasOperation_t transa = CUBLAS_OP_T; + const cublasOperation_t transb = CUBLAS_OP_N; + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_TRANSA, &transa, sizeof(transa))); + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_TRANSB, &transb, sizeof(transb))); + const void* a_scale = weight_block_scale.data_ptr(); + const void* b_scale = activation_block_scale.data_ptr(); + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_A_SCALE_POINTER, &a_scale, + sizeof(a_scale))); + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_B_SCALE_POINTER, &b_scale, + sizeof(b_scale))); + const cublasDataType_t scale_type = CUDA_R_32F; + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_SCALE_TYPE, &scale_type, + sizeof(scale_type))); + const cublasLtPointerMode_t pointer_mode = CUBLASLT_POINTER_MODE_DEVICE; + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_POINTER_MODE, &pointer_mode, + sizeof(pointer_mode))); + const cublasLtEpilogue_t epilogue = CUBLASLT_EPILOGUE_DEFAULT; + LT_CHECK(cublasLtMatmulDescSetAttribute( + operation, CUBLASLT_MATMUL_DESC_EPILOGUE, &epilogue, + sizeof(epilogue))); + + LT_CHECK(cublasLtMatrixLayoutCreate(&a, CUDA_R_4F_E2M1, k, m, k)); + LT_CHECK(cublasLtMatrixLayoutCreate(&b, CUDA_R_4F_E2M1, k, n, k)); + LT_CHECK(cublasLtMatrixLayoutCreate(&c, CUDA_R_16BF, m, n, m)); + LT_CHECK(cublasLtMatrixLayoutCreate(&d, CUDA_R_16BF, m, n, m)); + } + + ~Problem() { + if (d) cublasLtMatrixLayoutDestroy(d); + if (c) cublasLtMatrixLayoutDestroy(c); + if (b) cublasLtMatrixLayoutDestroy(b); + if (a) cublasLtMatrixLayoutDestroy(a); + if (operation) cublasLtMatmulDescDestroy(operation); + } +}; + +template +bool config_get(const cublasLtMatmulAlgo_t& algo, + cublasLtMatmulAlgoConfigAttributes_t attr, T* value) { + size_t written = 0; + return cublasLtMatmulAlgoConfigGetAttribute(&algo, attr, value, + sizeof(T), &written) == + CUBLAS_STATUS_SUCCESS && + written == sizeof(T); +} + +template +void put_config(py::dict& result, const char* name, + const cublasLtMatmulAlgo_t& algo, + cublasLtMatmulAlgoConfigAttributes_t attr) { + T value{}; + if (config_get(algo, attr, &value)) result[name] = value; +} + +template +void put_cap_scalar(py::dict& caps, const char* name, + const cublasLtMatmulAlgo_t& algo, + cublasLtMatmulAlgoCapAttributes_t attr) { + T value{}; + size_t written = 0; + if (cublasLtMatmulAlgoCapGetAttribute(&algo, attr, &value, sizeof(value), + &written) == CUBLAS_STATUS_SUCCESS && + written == sizeof(value)) + caps[name] = value; +} + +void put_cap_array(py::dict& caps, const char* name, + const cublasLtMatmulAlgo_t& algo, + cublasLtMatmulAlgoCapAttributes_t attr) { + size_t bytes = 0; + if (cublasLtMatmulAlgoCapGetAttribute(&algo, attr, nullptr, 0, &bytes) != + CUBLAS_STATUS_SUCCESS || + bytes == 0) + return; + std::vector values((bytes + sizeof(uint32_t) - 1) / + sizeof(uint32_t)); + size_t written = 0; + if (cublasLtMatmulAlgoCapGetAttribute(&algo, attr, values.data(), bytes, + &written) != CUBLAS_STATUS_SUCCESS) + return; + values.resize(written / sizeof(uint32_t)); + caps[name] = values; +} + +py::dict describe(const cublasLtMatmulAlgo_t& algo, + const cublasLtMatmulHeuristicResult_t& checked, + cublasStatus_t api_status) { + py::dict result; + for (const char* name : {"algorithm_id", "tile_id", "stages_id", "split_k", + "reduction_scheme", "custom_option", "cta_swizzle", + "inner_shape", "cluster_shape"}) + result[name] = py::none(); + put_config(result, "algorithm_id", algo, CUBLASLT_ALGO_CONFIG_ID); + put_config(result, "tile_id", algo, CUBLASLT_ALGO_CONFIG_TILE_ID); + put_config(result, "stages_id", algo, + CUBLASLT_ALGO_CONFIG_STAGES_ID); + // CUDA 13 documents SPLITK_NUM as int32_t. Preserve negative library + // sentinel values instead of wrapping them into fictitious huge factors. + put_config(result, "split_k", algo, + CUBLASLT_ALGO_CONFIG_SPLITK_NUM); + put_config(result, "reduction_scheme", algo, + CUBLASLT_ALGO_CONFIG_REDUCTION_SCHEME); + put_config(result, "custom_option", algo, + CUBLASLT_ALGO_CONFIG_CUSTOM_OPTION); + put_config(result, "cta_swizzle", algo, + CUBLASLT_ALGO_CONFIG_CTA_SWIZZLING); +#if CUDA_VERSION >= 12000 + put_config(result, "inner_shape", algo, + CUBLASLT_ALGO_CONFIG_INNER_SHAPE_ID); + put_config(result, "cluster_shape", algo, + CUBLASLT_ALGO_CONFIG_CLUSTER_SHAPE_ID); +#endif + result["required_workspace_bytes"] = checked.workspaceSize; + result["waves"] = checked.wavesCount; + result["state"] = static_cast(checked.state); + result["api_status"] = static_cast(api_status); + result["valid"] = api_status == CUBLAS_STATUS_SUCCESS && + checked.state == CUBLAS_STATUS_SUCCESS; + + py::dict caps; + put_cap_scalar(caps, "split_k_support", algo, + CUBLASLT_ALGO_CAP_SPLITK_SUPPORT); + put_cap_scalar(caps, "reduction_scheme_mask", algo, + CUBLASLT_ALGO_CAP_REDUCTION_SCHEME_MASK); + put_cap_scalar(caps, "cta_swizzle_support", algo, + CUBLASLT_ALGO_CAP_CTA_SWIZZLING_SUPPORT); + put_cap_scalar(caps, "custom_option_max", algo, + CUBLASLT_ALGO_CAP_CUSTOM_OPTION_MAX); + put_cap_scalar(caps, "strided_batch_support", algo, + CUBLASLT_ALGO_CAP_STRIDED_BATCH_SUPPORT); + put_cap_scalar(caps, "out_of_place_result_support", algo, + CUBLASLT_ALGO_CAP_OUT_OF_PLACE_RESULT_SUPPORT); + put_cap_array(caps, "tile_ids", algo, CUBLASLT_ALGO_CAP_TILE_IDS); + put_cap_array(caps, "stages_ids", algo, CUBLASLT_ALGO_CAP_STAGES_IDS); + caps["inner_cluster_shape_capability_note"] = + "This CUDA 13 cublasLt.h exposes config IDs but no public capability " + "attributes that enumerate inner/cluster shape IDs."; + result["capabilities"] = caps; + return result; +} + +cublasLtMatmulAlgo_t init_algo(int algorithm_id) { + cublasLtMatmulAlgo_t algo{}; + LT_CHECK(cublasLtMatmulAlgoInit( + get_handle(), CUBLAS_COMPUTE_32F, CUDA_R_32F, CUDA_R_4F_E2M1, + CUDA_R_4F_E2M1, CUDA_R_16BF, CUDA_R_16BF, algorithm_id, &algo)); + return algo; +} + +template +void maybe_set(cublasLtMatmulAlgo_t* algo, const py::dict& config, + const char* key, cublasLtMatmulAlgoConfigAttributes_t attr) { + if (!config.contains(key) || config[key].is_none()) return; + const T value = config[key].cast(); + LT_CHECK(cublasLtMatmulAlgoConfigSetAttribute(algo, attr, &value, + sizeof(value))); +} + +cublasLtMatmulAlgo_t configured_algo(const py::dict& config) { + TORCH_CHECK(config.contains("algorithm_id"), "algorithm_id is required"); + auto algo = init_algo(config["algorithm_id"].cast()); + maybe_set(&algo, config, "tile_id", CUBLASLT_ALGO_CONFIG_TILE_ID); + maybe_set(&algo, config, "stages_id", + CUBLASLT_ALGO_CONFIG_STAGES_ID); + maybe_set(&algo, config, "split_k", + CUBLASLT_ALGO_CONFIG_SPLITK_NUM); + maybe_set(&algo, config, "reduction_scheme", + CUBLASLT_ALGO_CONFIG_REDUCTION_SCHEME); + maybe_set(&algo, config, "custom_option", + CUBLASLT_ALGO_CONFIG_CUSTOM_OPTION); + maybe_set(&algo, config, "cta_swizzle", + CUBLASLT_ALGO_CONFIG_CTA_SWIZZLING); +#if CUDA_VERSION >= 12000 + maybe_set(&algo, config, "inner_shape", + CUBLASLT_ALGO_CONFIG_INNER_SHAPE_ID); + maybe_set(&algo, config, "cluster_shape", + CUBLASLT_ALGO_CONFIG_CLUSTER_SHAPE_ID); +#endif + return algo; +} + +py::list enumerate(torch::Tensor activation_qdata, + torch::Tensor activation_block_scale, + torch::Tensor weight_qdata, + torch::Tensor weight_block_scale, + int64_t max_workspace, int requested_count) { + c10::cuda::CUDAGuard guard(activation_qdata.device()); + Problem problem(activation_qdata, activation_block_scale, weight_qdata, + weight_block_scale); + cublasLtMatmulPreference_t preference = nullptr; + LT_CHECK(cublasLtMatmulPreferenceCreate(&preference)); + LT_CHECK(cublasLtMatmulPreferenceSetAttribute( + preference, CUBLASLT_MATMUL_PREF_MAX_WORKSPACE_BYTES, &max_workspace, + sizeof(max_workspace))); + std::vector found(requested_count); + int returned = 0; + const auto status = cublasLtMatmulAlgoGetHeuristic( + get_handle(), problem.operation, problem.a, problem.b, problem.c, + problem.d, preference, requested_count, found.data(), &returned); + cublasLtMatmulPreferenceDestroy(preference); + LT_CHECK(status); + py::list output; + for (int i = 0; i < returned; ++i) + output.append(describe(found[i].algo, found[i], CUBLAS_STATUS_SUCCESS)); + return output; +} + +py::dict check(torch::Tensor activation_qdata, + torch::Tensor activation_block_scale, + torch::Tensor weight_qdata, + torch::Tensor weight_block_scale, py::dict config) { + c10::cuda::CUDAGuard guard(activation_qdata.device()); + Problem problem(activation_qdata, activation_block_scale, weight_qdata, + weight_block_scale); + auto algo = configured_algo(config); + cublasLtMatmulHeuristicResult_t result{}; + const auto status = cublasLtMatmulAlgoCheck( + get_handle(), problem.operation, problem.a, problem.b, problem.c, + problem.d, &algo, &result); + return describe(algo, result, status); +} + +void run(torch::Tensor activation_qdata, + torch::Tensor activation_block_scale, + torch::Tensor weight_qdata, torch::Tensor weight_block_scale, + torch::Tensor alpha, torch::Tensor beta, torch::Tensor output, + torch::Tensor workspace, py::dict config) { + c10::cuda::CUDAGuard guard(activation_qdata.device()); + require_cuda_contiguous(alpha, "alpha"); + require_cuda_contiguous(beta, "beta"); + require_cuda_contiguous(output, "output"); + require_cuda_contiguous(workspace, "workspace"); + require_same_device(alpha, activation_qdata, "alpha"); + require_same_device(beta, activation_qdata, "beta"); + require_same_device(output, activation_qdata, "output"); + require_same_device(workspace, activation_qdata, "workspace"); + TORCH_CHECK(alpha.scalar_type() == at::kFloat && alpha.numel() == 1, + "alpha must be one device FP32 value"); + TORCH_CHECK(beta.scalar_type() == at::kFloat && beta.numel() == 1, + "beta must be one device FP32 value"); + TORCH_CHECK(output.scalar_type() == at::kBFloat16 && output.dim() == 2, + "output must be rank-two BF16"); + TORCH_CHECK(workspace.scalar_type() == at::kByte, + "workspace must use uint8 storage"); + TORCH_CHECK(output.size(0) == activation_qdata.size(0) && + output.size(1) == weight_qdata.size(0), + "output must be [packed activation rows, weight rows]"); + Problem problem(activation_qdata, activation_block_scale, weight_qdata, + weight_block_scale); + auto algo = configured_algo(config); + cublasLtMatmulHeuristicResult_t checked{}; + LT_CHECK(cublasLtMatmulAlgoCheck(get_handle(), problem.operation, problem.a, + problem.b, problem.c, problem.d, &algo, + &checked)); + TORCH_CHECK(checked.state == CUBLAS_STATUS_SUCCESS, + "selected algorithm failed AlgoCheck with state ", + static_cast(checked.state)); + TORCH_CHECK(checked.workspaceSize <= static_cast(workspace.numel()), + "selected algorithm requires ", checked.workspaceSize, + " workspace bytes but caller supplied ", workspace.numel()); + const auto stream = at::cuda::getCurrentCUDAStream( + activation_qdata.get_device()).stream(); + void* workspace_ptr = workspace.numel() ? workspace.data_ptr() : nullptr; + LT_CHECK(cublasLtMatmul( + get_handle(), problem.operation, alpha.data_ptr(), weight_qdata.data_ptr(), + problem.a, activation_qdata.data_ptr(), problem.b, beta.data_ptr(), + output.data_ptr(), problem.c, output.data_ptr(), problem.d, &algo, + workspace_ptr, workspace.numel(), stream)); +} + +py::dict build_info() { + py::dict result; + result["cuda_version"] = CUDA_VERSION; + result["cublas_version"] = CUBLAS_VERSION; + result["stream_k_public_control"] = false; + result["stream_k_note"] = + "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute " + "that directly selects Stream-K. Negative SPLITK_NUM values returned " + "by heuristics are preserved as undocumented library sentinels, not " + "claimed as public Stream-K control."; + return result; +} + +} // namespace + +PYBIND11_MODULE(TORCH_EXTENSION_NAME, module) { + module.def("enumerate", &enumerate); + module.def("check", &check); + module.def("run", &run); + module.def("build_info", &build_info); +} diff --git a/tools/benchmark_fc2_nvfp4_algorithms.py b/tools/benchmark_fc2_nvfp4_algorithms.py new file mode 100644 index 0000000..49de326 --- /dev/null +++ b/tools/benchmark_fc2_nvfp4_algorithms.py @@ -0,0 +1,729 @@ +"""Characterize isolated cuBLASLt NVFP4 scheduling at the real H3 FC2 boundary.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import platform +import statistics +import subprocess +import sys +import time +import types +from pathlib import Path +from typing import Any, Callable + +import torch + +from h3_blackwell_runtime.checkpoint import H3Checkpoint +from h3_blackwell_runtime.denoiser import H3PackedDenoiser +from h3_blackwell_runtime.nvfp4_quant import vortex_native_quantize_swiglu_nvfp4 +from h3_blackwell_runtime.packing import H3PromptPacker +from h3_blackwell_runtime.rope import h3_rope_rotation +from h3_blackwell_runtime.sampler import _audio_sigma, _model_sigma, beta_sigmas, sample_video_res_multistep +from h3_blackwell_runtime.t2v import random_av_latents + + +ROOT = Path(__file__).resolve().parents[1] +SOURCE = ROOT / "research" / "fc2_nvfp4_scheduling" / "fc2_nvfp4_lt.cpp" +DEFAULT_BUDGETS = (0, 4 << 20, 8 << 20, 16 << 20, 32 << 20, 64 << 20) +DEFAULT_BLOCKS = (0, 24, 49) +ENV_PREFIXES = ("H3_", "COMFY_KITCHEN_", "CUDA_", "TORCH_") + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--mode", choices=("compile", "characterize", "sweep", "selected", "profile", "block-gate", "trajectory"), default="characterize") + parser.add_argument("--model-path", default="/models/minimax_h3_fl2va_pruned_nvfp4.safetensors") + parser.add_argument("--capture", type=Path, help="Directory containing input.pt, or an input .pt file") + parser.add_argument("--output", type=Path, default=Path("/output/h3-blackwell-runtime/benchmarks/fc2-nvfp4-scheduling.json")) + parser.add_argument("--build-directory", type=Path) + parser.add_argument("--verbose-build", action="store_true") + parser.add_argument("--width", type=int, default=1344) + parser.add_argument("--height", type=int, default=768) + parser.add_argument("--frames", type=int, default=124) + parser.add_argument("--steps", type=int, default=12) + parser.add_argument("--sampler-step", type=int, default=1) + parser.add_argument("--seed", type=int, default=440420) + parser.add_argument("--text-tokens", type=int, default=100) + parser.add_argument("--blocks", type=int, nargs="+", default=list(DEFAULT_BLOCKS)) + parser.add_argument("--probe-block", type=int, default=24) + parser.add_argument("--workspace-budgets", type=int, nargs="+", default=list(DEFAULT_BUDGETS)) + parser.add_argument("--requested-count", type=int, default=32) + parser.add_argument("--explicit-split-k", type=int, nargs="+", default=[1, 2, 4, 8, 16]) + parser.add_argument("--max-explicit-checks", type=int, default=64) + parser.add_argument("--candidate-limit", type=int, default=8) + parser.add_argument("--candidate", help="JSON object, JSON file, or enumerate::") + parser.add_argument("--profile-target", choices=("baseline", "candidate"), default="candidate") + parser.add_argument("--workspace-bytes", type=int, default=32 << 20) + parser.add_argument("--warmup", type=int, default=2) + parser.add_argument("--rounds", type=int, default=6) + parser.add_argument("--checkpoint-sha256", action="store_true", help="Expensive: actually calculate and record the checkpoint SHA-256") + return parser.parse_args() + + +def load_extension(args: argparse.Namespace): + from torch.utils.cpp_extension import CUDA_HOME, load + + if CUDA_HOME is None: + raise RuntimeError("CUDA_HOME is unavailable; CUDA 12.9 or newer headers are required") + kwargs: dict[str, Any] = {} + if args.build_directory is not None: + args.build_directory.mkdir(parents=True, exist_ok=True) + kwargs["build_directory"] = str(args.build_directory) + return load( + name="h3_fc2_nvfp4_lt_schedule", + sources=[str(SOURCE)], + extra_include_paths=[str(Path(CUDA_HOME) / "include")], + extra_cflags=["-O2", "-std=c++17"], + extra_ldflags=["-L" + str(Path(CUDA_HOME) / "lib64"), "-lcublasLt", "-lcublas", "-lcudart"], + verbose=args.verbose_build, + **kwargs, + ) + + +def sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(8 << 20), b""): + digest.update(chunk) + return digest.hexdigest() + + +def tensor_sha256(tensor: torch.Tensor) -> str: + immutable = tensor.detach().contiguous().clone().view(torch.uint16).cpu() + return hashlib.sha256(immutable.numpy().tobytes()).hexdigest() + + +def compare(actual: torch.Tensor, expected: torch.Tensor) -> dict[str, Any]: + actual_copy = actual.detach().clone() + expected_copy = expected.detach().clone() + different = int(torch.count_nonzero(actual_copy != expected_copy)) + delta = (actual_copy.float() - expected_copy.float()).abs() + return { + "bf16_exact": different == 0, + "different_elements": different, + "max_abs": float(delta.max()), + "mean_abs": float(delta.mean()), + "actual_sha256": tensor_sha256(actual_copy), + "expected_sha256": tensor_sha256(expected_copy), + } + + +def percentile(values: list[float], fraction: float) -> float: + ordered = sorted(values) + if len(ordered) == 1: + return ordered[0] + position = (len(ordered) - 1) * fraction + lower = int(position) + weight = position - lower + return ordered[lower] * (1.0 - weight) + ordered[min(lower + 1, len(ordered) - 1)] * weight + + +def timing_summary(milliseconds: list[float], m: int | None = None, n: int | None = None, k: int | None = None) -> dict[str, Any]: + result: dict[str, Any] = { + "samples_ms": milliseconds, + "p50_ms": percentile(milliseconds, 0.50), + "p95_ms": percentile(milliseconds, 0.95), + "mean_ms": statistics.fmean(milliseconds), + } + if m is not None and n is not None and k is not None: + result["dense_tflop_s_p50"] = 2.0 * m * n * k / (result["p50_ms"] * 1.0e9) + return result + + +def timed_cuda(call: Callable[[], torch.Tensor]) -> tuple[float, torch.Tensor]: + start = torch.cuda.Event(enable_timing=True) + end = torch.cuda.Event(enable_timing=True) + start.record() + output = call() + end.record() + end.synchronize() + return float(start.elapsed_time(end)), output + + +def environment(args: argparse.Namespace, extension) -> dict[str, Any]: + checkpoint = Path(args.model_path) + try: + commit = subprocess.run( + ["git", "rev-parse", "HEAD"], cwd=ROOT, check=True, + capture_output=True, text=True, + ).stdout.strip() + except (OSError, subprocess.CalledProcessError): + commit = None + result = { + "platform": platform.platform(), + "python": sys.version, + "torch": torch.__version__, + "torch_cuda": torch.version.cuda, + "device": torch.cuda.get_device_name(), + "device_capability": list(torch.cuda.get_device_capability()), + "driver": torch.cuda.driver_version() if hasattr(torch.cuda, "driver_version") else None, + "git_commit": commit, + "checkpoint_path": str(checkpoint), + "checkpoint_sha256": sha256_file(checkpoint) if args.checkpoint_sha256 else None, + "checkpoint_hash_note": "calculated" if args.checkpoint_sha256 else "not calculated", + "environment_switches": {key: value for key, value in sorted(os.environ.items()) if key.startswith(ENV_PREFIXES)}, + "extension": dict(extension.build_info()), + } + try: + import comfy_kitchen + result["comfy_kitchen"] = getattr(comfy_kitchen, "__version__", "0.2.31 package without __version__") + except Exception as error: + result["comfy_kitchen"] = f"import error: {error}" + return result + + +def make_workload(args: argparse.Namespace, checkpoint: H3Checkpoint, model: H3PackedDenoiser): + torch.manual_seed(args.seed) + if args.capture is not None: + path = args.capture / "input.pt" if args.capture.is_dir() else args.capture + payload = torch.load(path, map_location="cuda", weights_only=False) + hidden = payload["hidden"].to("cuda").contiguous() + timesteps = payload["timesteps"].to("cuda") + positions = payload["position_ids"].to("cuda") + segments = payload["segments"] + metadata = {"capture": str(path)} + else: + packer = H3PromptPacker(checkpoint) + video, audio, aligned_frames = random_av_latents( + args.width, args.height, args.frames, args.seed, device="cuda", + ) + sigma = beta_sigmas(args.steps, device="cuda")[args.sampler_step - 1] + native_audio = audio.to(torch.bfloat16) * (_audio_sigma(sigma) / sigma) + text = torch.randn(1, args.text_tokens, 5376, device="cuda", dtype=torch.bfloat16) + hidden, timesteps, segments, positions, _, _ = packer( + text, video, native_audio, _model_sigma(sigma), + ) + metadata = { + "resolution": [args.width, args.height], + "frames": aligned_frames, + "steps": args.steps, + "sampler_step": args.sampler_step, + "seed": args.seed, + "text_tokens": args.text_tokens, + } + canonical = (args.width, args.height, args.frames, args.seed, args.sampler_step, args.text_tokens) == (1344, 768, 124, 440420, 1, 100) + if canonical and hidden.shape[0] != 37_810: + raise RuntimeError(f"canonical workload must contain 37,810 tokens, got {hidden.shape[0]}") + rotation = h3_rope_rotation(positions.to("cuda"), model.backbone.inv_freq, torch.bfloat16) + metadata.update({"tokens": hidden.shape[0], "hidden_shape": list(hidden.shape), "segments": segments}) + return hidden, timesteps, rotation, segments, metadata + + +def capture_boundaries(args: argparse.Namespace, model: H3PackedDenoiser, hidden, timesteps, rotation, segments): + wanted = set(args.blocks) + if wanted != set(DEFAULT_BLOCKS): + missing = set(DEFAULT_BLOCKS) - wanted + if missing: + raise ValueError(f"--blocks must retain required blocks 0,24,49; missing {sorted(missing)}") + block_inputs: dict[int, torch.Tensor] = {} + block_outputs: dict[int, torch.Tensor] = {} + gate_up: dict[int, torch.Tensor] = {} + hooks = [] + modulated_forwards = {} + for index in wanted: + fc1 = model.backbone.blocks[index].mlp.fc1 + hooks.append(fc1.register_forward_hook( + lambda _module, _inputs, output, index=index: gate_up.__setitem__(index, output.detach().clone()) + )) + original = fc1.forward_modulated + modulated_forwards[index] = original + + def capture_modulated(_self, *values, index=index, original=original, **kwargs): + output = original(*values, **kwargs) + gate_up[index] = output.detach().clone() + return output + + fc1.forward_modulated = types.MethodType(capture_modulated, fc1) + try: + with torch.inference_mode(): + for index, (block, adaln) in enumerate(zip(model.backbone.blocks, model.backbone.adaln, strict=True)): + if index in wanted: + block_inputs[index] = hidden.detach().clone() + hidden = block(hidden, rotation, *adaln(timesteps), segments) + if index in wanted: + block_outputs[index] = hidden.detach().clone() + finally: + for hook in hooks: + hook.remove() + for index, original in modulated_forwards.items(): + model.backbone.blocks[index].mlp.fc1.forward_modulated = original + if set(gate_up) != wanted: + raise RuntimeError(f"failed to capture FC1 boundaries: got {sorted(gate_up)}") + return block_inputs, block_outputs, gate_up + + +def packed_boundary(gate_up: torch.Tensor, fc2): + tensor_scale, qdata, block_scale = vortex_native_quantize_swiglu_nvfp4(gate_up) + alpha = (tensor_scale.float() * fc2.weight_scale_2.float()).reshape(1).contiguous() + beta = torch.zeros(1, device=gate_up.device, dtype=torch.float32) + return tensor_scale, qdata, block_scale, alpha, beta + + +def baseline_metadata(fc2, gate_up: torch.Tensor) -> dict[str, Any]: + from torch.profiler import ProfilerActivity, profile + + with torch.inference_mode(): + fc2.forward_swiglu(gate_up) + torch.cuda.synchronize() + with profile(activities=[ProfilerActivity.CPU, ProfilerActivity.CUDA]) as captured: + output = fc2.forward_swiglu(gate_up) + torch.cuda.synchronize() + events = [] + for event in captured.events(): + if "CUDA" not in str(getattr(event, "device_type", "")): + continue + device_us = max( + float(getattr(event, "device_time_total", 0.0) or 0.0), + float(getattr(event, "self_device_time_total", 0.0) or 0.0), + float(getattr(event, "cuda_time_total", 0.0) or 0.0), + float(getattr(event, "self_cuda_time_total", 0.0) or 0.0), + ) + events.append({"name": event.name, "device_time_total_us": device_us}) + events.sort(key=lambda row: row["device_time_total_us"], reverse=True) + return { + "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", + "descriptors": { + "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", + "block_scale_mode": "VEC16_UE4M3", + "compute_and_scale": "FP32", + "scalar_pointer_mode": "device", + "bias": None, + "beta": 0.0, + "comfy_kitchen_version": "0.2.31", + }, + "profiler_cuda_events_available": bool(events), + "profiler_note": None if events else "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", + "top_cuda_events": events[:10], + "output_sha256": tensor_sha256(output.detach().clone()), + } + + +def enumerate_all(extension, packed, fc2, args: argparse.Namespace): + _tensor_scale, qdata, block_scale, _alpha, _beta = packed + groups = [] + errors = [] + for budget in args.workspace_budgets: + try: + values = [dict(value) for value in extension.enumerate( + qdata, block_scale, fc2.weight, fc2.weight_scale, budget, args.requested_count, + )] + groups.append({"max_workspace_bytes": budget, "requested_count": args.requested_count, "returned_count": len(values), "algorithms": values}) + except Exception as error: + groups.append({"max_workspace_bytes": budget, "requested_count": args.requested_count, "returned_count": 0, "algorithms": []}) + errors.append({"operation": "heuristic", "max_workspace_bytes": budget, "error": repr(error)}) + return groups, errors + + +def explicit_checks(extension, packed, fc2, enumerated, args: argparse.Namespace): + _tensor_scale, qdata, block_scale, _alpha, _beta = packed + unique_bases: dict[int, dict[str, Any]] = {} + for group in enumerated: + for candidate in group["algorithms"]: + unique_bases.setdefault(candidate["algorithm_id"], candidate) + checked, errors = [], [] + attempts = 0 + for base in unique_bases.values(): + mask = int(base.get("capabilities", {}).get("reduction_scheme_mask", 0)) + reductions = [1 << bit for bit in range(32) if mask & (1 << bit)] + for split_k in args.explicit_split_k: + schemes = [0] if split_k == 1 else reductions + if not schemes: + errors.append({"algorithm_id": base["algorithm_id"], "split_k": split_k, "error": "capability reports no public reduction scheme"}) + for reduction in schemes: + if attempts >= args.max_explicit_checks: + return checked, errors + attempts += 1 + config = {key: base[key] for key in ("algorithm_id", "tile_id", "stages_id", "custom_option", "cta_swizzle", "inner_shape", "cluster_shape") if key in base} + config.update({"split_k": split_k, "reduction_scheme": reduction}) + try: + result = dict(extension.check(qdata, block_scale, fc2.weight, fc2.weight_scale, config)) + result["requested_config"] = config + checked.append(result) + except Exception as error: + errors.append({"config": config, "error": repr(error)}) + return checked, errors + + +def resolve_candidate(args: argparse.Namespace, enumerated, explicit) -> dict[str, Any]: + if args.candidate: + if args.candidate.startswith("enumerate:"): + _, group, result = args.candidate.split(":") + return dict(enumerated[int(group)]["algorithms"][int(result)]) + path = Path(args.candidate) + return json.loads(path.read_text()) if path.exists() else json.loads(args.candidate) + valid_explicit = [row for row in explicit if row.get("valid") and row.get("required_workspace_bytes", sys.maxsize) <= args.workspace_bytes] + if valid_explicit: + return valid_explicit[0]["requested_config"] + for group in enumerated: + for row in group["algorithms"]: + if row.get("valid") and row.get("required_workspace_bytes", sys.maxsize) <= args.workspace_bytes: + return row + raise RuntimeError("no valid candidate fits --workspace-bytes") + + +def candidate_runner(extension, packed, fc2, config, workspace_bytes: int): + _tensor_scale, qdata, block_scale, alpha, beta = packed + output = torch.empty((qdata.shape[0], fc2.out_features), device=qdata.device, dtype=torch.bfloat16) + workspace = torch.empty(workspace_bytes, device=qdata.device, dtype=torch.uint8) + + def run() -> torch.Tensor: + extension.run(qdata, block_scale, fc2.weight, fc2.weight_scale, alpha, beta, output, workspace, config) + return output + + return run, output, workspace + + +def comfy_packed_runner(packed, fc2, logical_rows: int): + import torch.nn.functional as functional + from comfy_kitchen.tensor import QuantizedTensor, TensorCoreNVFP4Layout + + tensor_scale, qdata, block_scale, _alpha, _beta = packed + activation = QuantizedTensor( + qdata, + "TensorCoreNVFP4Layout", + TensorCoreNVFP4Layout.Params( + scale=tensor_scale, + orig_dtype=torch.bfloat16, + orig_shape=(logical_rows, fc2.in_features), + block_scale=block_scale, + ), + ) + weight = fc2._packed_weight() + + def run(): + return functional.linear(activation, weight, None)[:logical_rows, :fc2.out_features] + + return run + + +def benchmark_pair(baseline, candidate, rounds: int, warmup: int, m: int, n: int, k: int, *, report_dense: bool = True): + with torch.inference_mode(): + for _ in range(warmup): + baseline() + candidate() + torch.cuda.synchronize() + samples = {"baseline": [], "candidate": []} + last = {} + for round_index in range(rounds): + order = (("baseline", baseline), ("candidate", candidate)) if round_index % 2 == 0 else (("candidate", candidate), ("baseline", baseline)) + for name, call in order: + elapsed, output = timed_cuda(call) + samples[name].append(elapsed) + last[name] = output.detach().clone() + return { + "order": "AB/BA alternates by round", + "baseline": timing_summary(samples["baseline"], m if report_dense else None, n if report_dense else None, k if report_dense else None), + "candidate": timing_summary(samples["candidate"], m if report_dense else None, n if report_dense else None, k if report_dense else None), + "parity": compare(last["candidate"][:m, :n], last["baseline"][:m, :n]), + } + + +def benchmark_fc2_candidate(extension, gate_up, fc2, packed, config, args): + run_candidate, output, workspace = candidate_runner(extension, packed, fc2, config, args.workspace_bytes) + m, n, k = gate_up.shape[0], fc2.out_features, fc2.in_features + run_comfy_packed = comfy_packed_runner(packed, fc2, m) + direct = benchmark_pair( + run_comfy_packed, run_candidate, + args.rounds, args.warmup, m, n, k, + ) + + boundary_output = torch.empty_like(output) + boundary_workspace = torch.empty_like(workspace) + + def candidate_boundary(): + fresh = packed_boundary(gate_up, fc2) + extension.run( + fresh[1], fresh[2], fc2.weight, fc2.weight_scale, fresh[3], fresh[4], + boundary_output, boundary_workspace, config, + ) + return boundary_output + + boundary = benchmark_pair( + lambda: fc2.forward_swiglu(gate_up), candidate_boundary, + args.rounds, args.warmup, m, n, k, + ) + if not direct["parity"]["bf16_exact"] or not boundary["parity"]["bf16_exact"]: + raise RuntimeError("FC2 library candidate is not byte-exact") + checked = dict(extension.check(packed[1], packed[2], fc2.weight, fc2.weight_scale, config)) + return { + "selected_config": config, + "supplied_workspace_bytes": workspace.numel(), + "required_workspace_bytes": checked.get("required_workspace_bytes"), + "checked": checked, + "fc2_only": direct, + "accepted_producer_plus_fc2": boundary, + "output_sha256": tensor_sha256(output[:m].detach().clone()), + } + + +def profile_one(extension, packed, fc2, config, logical_rows: int, args): + run_candidate, output, workspace = candidate_runner(extension, packed, fc2, config, args.workspace_bytes) + expected = None + if args.profile_target == "baseline": + run_profiled = comfy_packed_runner(packed, fc2, logical_rows) + else: + run_profiled = run_candidate + with torch.inference_mode(): + for _ in range(args.warmup): + run_profiled() + if args.profile_target == "candidate": + expected = comfy_packed_runner(packed, fc2, logical_rows)().detach().clone() + torch.cuda.synchronize() + torch.cuda.cudart().cudaProfilerStart() + output = run_profiled() + torch.cuda.cudart().cudaProfilerStop() + torch.cuda.synchronize() + parity = compare(output[:logical_rows], expected[:logical_rows]) if expected is not None else None + if parity is not None and not parity["bf16_exact"]: + raise RuntimeError("profiled FC2 library candidate is not byte-exact") + return { + "target": args.profile_target, + "selected_config": config if args.profile_target == "candidate" else None, + "supplied_workspace_bytes": workspace.numel() if args.profile_target == "candidate" else 32 << 20, + "output_sha256": tensor_sha256(output[:logical_rows, :fc2.out_features]), + "candidate_vs_baseline": parity, + } + + +def block_gate(extension, model, block_inputs, block_outputs, gate_up, timesteps, rotation, segments, config, args): + rows = [] + for index in DEFAULT_BLOCKS: + block = model.backbone.blocks[index] + fc2 = block.mlp.fc2 + packed = packed_boundary(gate_up[index], fc2) + _tensor_scale, qdata, _block_scale, _alpha, beta = packed + candidate_output = torch.empty( + (qdata.shape[0], fc2.out_features), device=qdata.device, dtype=torch.bfloat16, + ) + workspace = torch.empty(args.workspace_bytes, device=qdata.device, dtype=torch.uint8) + original = fc2.forward_swiglu + + def replacement(_self, actual_gate_up, expected=gate_up[index]): + if actual_gate_up.shape != expected.shape: + raise RuntimeError("block-gate FC2 received an unexpected boundary shape") + tensor_scale, actual_qdata, actual_block_scale = vortex_native_quantize_swiglu_nvfp4(actual_gate_up) + alpha = (tensor_scale.float() * fc2.weight_scale_2.float()).reshape(1).contiguous() + extension.run( + actual_qdata, actual_block_scale, fc2.weight, fc2.weight_scale, + alpha, beta, candidate_output, workspace, config, + ) + return candidate_output[:actual_gate_up.shape[0], :fc2.out_features] + + candidate_method = types.MethodType(replacement, fc2) + adaln_values = tuple(value.detach().clone() for value in model.backbone.adaln[index](timesteps)) + + def baseline(): + fc2.forward_swiglu = original + return block(block_inputs[index].detach().clone(), rotation, *adaln_values, segments) + + def candidate(): + fc2.forward_swiglu = candidate_method + return block(block_inputs[index].detach().clone(), rotation, *adaln_values, segments) + + try: + timing = benchmark_pair( + baseline, candidate, args.rounds, args.warmup, + block_outputs[index].shape[0], block_outputs[index].shape[1], 1, + report_dense=False, + ) + baseline_value = baseline().detach().clone() + candidate_value = candidate().detach().clone() + finally: + fc2.forward_swiglu = original + row = { + "block": index, + "only_monkeypatched_method": "block.mlp.fc2.forward_swiglu", + "accepted_gate_and_residual_path_preserved": True, + "candidate_vs_baseline": compare(candidate_value, baseline_value), + "baseline_vs_traversal": compare(baseline_value, block_outputs[index]), + "candidate_vs_traversal": compare(candidate_value, block_outputs[index]), + "timing": timing, + "supplied_workspace_bytes": workspace.numel(), + } + if not all( + row[name]["bf16_exact"] + for name in ("candidate_vs_baseline", "baseline_vs_traversal", "candidate_vs_traversal") + ): + raise RuntimeError(f"FC2 library candidate is not byte-exact in block {index}") + rows.append(row) + return rows + + +def trajectory_gate(extension, checkpoint, model, config, args): + packer = H3PromptPacker(checkpoint) + torch.manual_seed(args.seed) + video, audio, aligned_frames = random_av_latents( + args.width, args.height, args.frames, args.seed, device="cuda", + ) + text = torch.randn( + 1, args.text_tokens, 5376, device="cuda", dtype=torch.bfloat16, + ) + originals = [block.mlp.fc2.forward_swiglu for block in model.backbone.blocks] + shared_output = torch.empty( + (((37_810 + 15) // 16) * 16, 5376), device="cuda", dtype=torch.bfloat16, + ) + workspace = torch.empty(args.workspace_bytes, device="cuda", dtype=torch.uint8) + beta = torch.zeros(1, device="cuda", dtype=torch.float32) + + def candidate_method(fc2): + def replacement(_self, actual_gate_up): + tensor_scale, qdata, block_scale = vortex_native_quantize_swiglu_nvfp4(actual_gate_up) + if qdata.shape[0] > shared_output.shape[0] or fc2.out_features > shared_output.shape[1]: + raise RuntimeError("trajectory FC2 boundary exceeds the preallocated canonical output") + alpha = (tensor_scale.float() * fc2.weight_scale_2.float()).reshape(1).contiguous() + output = shared_output[:qdata.shape[0], :fc2.out_features] + extension.run( + qdata, block_scale, fc2.weight, fc2.weight_scale, + alpha, beta, output, workspace, config, + ) + return output[:actual_gate_up.shape[0], :fc2.out_features] + return types.MethodType(replacement, fc2) + + candidates = [candidate_method(block.mlp.fc2) for block in model.backbone.blocks] + + def run(candidate: bool): + for index, block in enumerate(model.backbone.blocks): + block.mlp.fc2.forward_swiglu = candidates[index] if candidate else originals[index] + torch.cuda.synchronize() + started = time.perf_counter() + result = sample_video_res_multistep( + model, + packer, + text, + video.detach().clone(), + audio.detach().clone(), + steps=args.steps, + seed=args.seed, + return_audio=True, + progress=True, + ) + torch.cuda.synchronize() + return result, time.perf_counter() - started + + try: + with torch.inference_mode(): + (reference_video, reference_audio), baseline_seconds = run(False) + (candidate_video, candidate_audio), candidate_seconds = run(True) + finally: + for block, original in zip(model.backbone.blocks, originals, strict=True): + block.mlp.fc2.forward_swiglu = original + + video_parity = compare(candidate_video, reference_video) + audio_parity = compare(candidate_audio, reference_audio) + result = { + "steps": args.steps, + "resolution": [args.width, args.height], + "frames": aligned_frames, + "seed": args.seed, + "baseline_seconds": baseline_seconds, + "candidate_seconds": candidate_seconds, + "improvement_percent": (1.0 - candidate_seconds / baseline_seconds) * 100.0, + "video_parity": video_parity, + "audio_parity": audio_parity, + "bf16_exact": video_parity["bf16_exact"] and audio_parity["bf16_exact"], + "selected_config": config, + "supplied_workspace_bytes": workspace.numel(), + "all_50_fc2_calls_replaced": True, + "accepted_swiglu_producer_preserved": True, + "accepted_gate_and_residual_path_preserved": True, + } + if not result["bf16_exact"]: + raise RuntimeError("FC2 library candidate trajectory is not byte-exact") + return result + + +def main() -> None: + args = parse_args() + if not torch.cuda.is_available(): + raise RuntimeError("CUDA is required") + extension = load_extension(args) + if args.mode == "compile": + result = {"mode": "compile", "extension": dict(extension.build_info())} + print(json.dumps(result, indent=2), flush=True) + return + + checkpoint = H3Checkpoint(args.model_path, device="cuda") + model = H3PackedDenoiser.from_checkpoint(checkpoint, attention_backend="sage2").eval() + hidden, timesteps, rotation, segments, workload = make_workload(args, checkpoint, model) + with torch.inference_mode(): + block_inputs, block_outputs, gate_ups = capture_boundaries( + args, model, hidden, timesteps, rotation, segments, + ) + probe = args.probe_block + if probe not in gate_ups: + raise ValueError("--probe-block must be one of the retained blocks") + fc2 = model.backbone.blocks[probe].mlp.fc2 + if fc2.bias is not None: + raise RuntimeError("this no-bias FC2 scheduling study refuses a biased module") + packed = packed_boundary(gate_ups[probe], fc2) + enumerated, errors = enumerate_all(extension, packed, fc2, args) + explicit, explicit_errors = explicit_checks(extension, packed, fc2, enumerated, args) + errors.extend(explicit_errors) + result: dict[str, Any] = { + "mode": args.mode, + "environment": environment(args, extension), + "workload": workload, + "retained_blocks": list(DEFAULT_BLOCKS), + "immutable_cloned_block_inputs": {str(index): list(value.shape) for index, value in block_inputs.items()}, + "fc2_boundary": { + "block": probe, + "gate_up_shape": list(gate_ups[probe].shape), + "activation_qdata_shape": list(packed[1].shape), + "weight_qdata_shape": list(fc2.weight.shape), + "logical_mnk": [gate_ups[probe].shape[0], fc2.out_features, fc2.in_features], + "descriptor_mnk_after_padding": [packed[1].shape[0], fc2.out_features, fc2.in_features], + "producer": "vortex_native_quantize_swiglu_nvfp4", + "no_bias": fc2.bias is None, + }, + "baseline_kernel_metadata": baseline_metadata(fc2, gate_ups[probe]), + "heuristics": enumerated, + "explicit_split_k_checks": explicit, + } + + if args.mode in {"characterize", "sweep", "selected", "profile", "block-gate", "trajectory"}: + config = resolve_candidate(args, enumerated, explicit) + result["selected"] = config + if args.mode == "characterize": + result["candidate_probe"] = benchmark_fc2_candidate(extension, gate_ups[probe], fc2, packed, config, args) + elif args.mode == "sweep": + candidates = [] + seen = set() + pool = [row for group in enumerated for row in group["algorithms"]] + [row for row in explicit if row.get("valid")] + for row in pool: + candidate = row.get("requested_config", row) + key = tuple(candidate.get(name) for name in ("algorithm_id", "tile_id", "stages_id", "split_k", "reduction_scheme", "custom_option", "cta_swizzle", "inner_shape", "cluster_shape")) + if key in seen or row.get("required_workspace_bytes", 0) > args.workspace_bytes: + continue + seen.add(key) + try: + candidates.append(benchmark_fc2_candidate(extension, gate_ups[probe], fc2, packed, candidate, args)) + except Exception as error: + errors.append({"config": candidate, "operation": "benchmark", "error": repr(error)}) + if len(candidates) >= args.candidate_limit: + break + result["candidates"] = candidates + elif args.mode == "selected": + result["candidate_probe"] = benchmark_fc2_candidate(extension, gate_ups[probe], fc2, packed, config, args) + elif args.mode == "profile": + result["profile"] = profile_one(extension, packed, fc2, config, gate_ups[probe].shape[0], args) + elif args.mode == "block-gate": + result["block_gate"] = block_gate(extension, model, block_inputs, block_outputs, gate_ups, timesteps, rotation, segments, config, args) + elif args.mode == "trajectory": + result["trajectory"] = trajectory_gate(extension, checkpoint, model, config, args) + + result["errors_and_unsupported"] = errors + [{ + "feature": "Stream-K", + "supported_public_control": False, + "reason": extension.build_info()["stream_k_note"], + }] + args.output.parent.mkdir(parents=True, exist_ok=True) + args.output.write_text(json.dumps(result, indent=2) + "\n", encoding="utf-8") + print(json.dumps(result, indent=2), flush=True) + + +if __name__ == "__main__": + main()