{ "mode": "block-gate", "environment": { "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", "torch": "2.9.1+cu130", "torch_cuda": "13.0", "device": "NVIDIA GB10", "device_capability": [ 12, 1 ], "driver": null, "git_commit": null, "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", "checkpoint_sha256": null, "checkpoint_hash_note": "not calculated", "environment_switches": { "CUDA_DEVICE_MAX_CONNECTIONS": "1", "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", "CUDA_HOME": "/usr/local/cuda", "CUDA_INC_PATH": "/usr/local/cuda/include", "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", "CUDA_MODULE_LOADING": "EAGER", "CUDA_VERSION": "13.0.2", "H3_FUSED_ELEMENTWISE": "1", "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", "H3_NVFP4_MODULATE_FUSION": "1", "H3_NVFP4_SCALE_BACKEND": "vortex", "H3_NVFP4_SCALE_VERSION": "1", "H3_NVFP4_SWIGLU_FUSION": "1", "H3_SAGE_QKV_LAYOUT": "strided_nhd", "TORCH_COMPILE_DISABLE": "0", "TORCH_CUDA_ARCH_LIST": "12.1a", "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" }, "extension": { "cuda_version": 13000, "cublas_version": 130100, "stream_k_public_control": false, "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." }, "comfy_kitchen": "0.2.31 package without __version__" }, "workload": { "resolution": [ 1344, 768 ], "frames": 124, "steps": 12, "sampler_step": 1, "seed": 440420, "text_tokens": 100, "tokens": 37810, "hidden_shape": [ 37810, 5376 ], "segments": [ [ 0, 100, 1 ], [ 100, 514, 2 ], [ 514, 37810, 0 ] ] }, "retained_blocks": [ 0, 24, 49 ], "immutable_cloned_block_inputs": { "0": [ 37810, 5376 ], "24": [ 37810, 5376 ], "49": [ 37810, 5376 ] }, "fc2_boundary": { "block": 24, "gate_up_shape": [ 37810, 28672 ], "activation_qdata_shape": [ 37824, 7168 ], "weight_qdata_shape": [ 5376, 7168 ], "logical_mnk": [ 37810, 5376, 14336 ], "descriptor_mnk_after_padding": [ 37824, 5376, 14336 ], "producer": "vortex_native_quantize_swiglu_nvfp4", "no_bias": true }, "baseline_kernel_metadata": { "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", "descriptors": { "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", "block_scale_mode": "VEC16_UE4M3", "compute_and_scale": "FP32", "scalar_pointer_mode": "device", "bias": null, "beta": 0.0, "comfy_kitchen_version": "0.2.31" }, "profiler_cuda_events_available": false, "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", "top_cuda_events": [], "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" }, "heuristics": [ { "max_workspace_bytes": 0, "requested_count": 32, "returned_count": 5, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 4194304, "requested_count": 32, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 8388608, "requested_count": 32, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 16777216, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 33554432, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 67108864, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] } ], "explicit_split_k_checks": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 1, "reduction_scheme": 0 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 4 } } ], "selected": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0 }, "block_gate": [ { "block": 0, "only_monkeypatched_method": "block.mlp.fc2.forward_swiglu", "accepted_gate_and_residual_path_preserved": true, "candidate_vs_baseline": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" }, "baseline_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" }, "candidate_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 457.58404541015625, 455.8675537109375, 455.042724609375, 456.5163879394531, 456.7074279785156, 459.7113037109375, 456.3467102050781, 459.0717468261719, 457.6736755371094, 458.6617736816406, 458.61083984375, 456.4378662109375, 457.4807434082031, 458.178466796875, 457.4196472167969, 458.90869140625, 458.4776306152344, 458.2261962890625, 458.10919189453125, 460.418701171875 ], "p50_ms": 457.8914337158203, "p95_ms": 459.7466735839844, "mean_ms": 457.7725662231445 }, "candidate": { "samples_ms": [ 428.5652160644531, 425.1132507324219, 420.88983154296875, 420.71435546875, 419.1617736816406, 421.9552307128906, 419.37896728515625, 419.3441467285156, 419.79351806640625, 419.0124206542969, 420.5487976074219, 420.9779052734375, 420.5229187011719, 420.88104248046875, 420.3728942871094, 420.11834716796875, 420.3209533691406, 423.73919677734375, 419.7261657714844, 423.4720458984375 ], "p50_ms": 420.5358581542969, "p95_ms": 425.2858489990234, "mean_ms": 421.2304489135742 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" } }, "supplied_workspace_bytes": 0 }, { "block": 24, "only_monkeypatched_method": "block.mlp.fc2.forward_swiglu", "accepted_gate_and_residual_path_preserved": true, "candidate_vs_baseline": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" }, "baseline_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" }, "candidate_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 458.4346008300781, 459.98040771484375, 457.2791748046875, 456.69989013671875, 458.1690368652344, 459.9689025878906, 459.0711975097656, 461.275390625, 458.7782897949219, 460.40740966796875, 459.8872985839844, 461.1615905761719, 459.39276123046875, 459.6435546875, 460.915283203125, 461.53076171875, 460.4534912109375, 460.2518615722656, 458.4383544921875, 461.34747314453125 ], "p50_ms": 459.9281005859375, "p95_ms": 461.3566375732422, "mean_ms": 459.6543365478516 }, "candidate": { "samples_ms": [ 418.2376708984375, 421.28668212890625, 417.4606018066406, 420.7315368652344, 421.40478515625, 419.4402770996094, 417.5429382324219, 419.78125, 419.68603515625, 421.91644287109375, 422.23394775390625, 420.081298828125, 419.1646728515625, 421.63360595703125, 422.85125732421875, 418.6084899902344, 420.3404235839844, 421.3265075683594, 422.9977111816406, 420.1449279785156 ], "p50_ms": 420.24267578125, "p95_ms": 422.8585800170899, "mean_ms": 420.3435531616211 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" } }, "supplied_workspace_bytes": 0 }, { "block": 49, "only_monkeypatched_method": "block.mlp.fc2.forward_swiglu", "accepted_gate_and_residual_path_preserved": true, "candidate_vs_baseline": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" }, "baseline_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" }, "candidate_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 455.2886962890625, 456.0704650878906, 455.62127685546875, 459.4961242675781, 456.5694580078125, 457.70477294921875, 458.9316101074219, 458.7506103515625, 457.3757629394531, 456.0811462402344, 458.1807861328125, 457.0354919433594, 459.8126220703125, 457.84747314453125, 458.5899963378906, 459.2362060546875, 460.5073547363281, 459.9343566894531, 457.9271240234375, 458.1864318847656 ], "p50_ms": 458.053955078125, "p95_ms": 459.9630065917969, "mean_ms": 457.95738830566404 }, "candidate": { "samples_ms": [ 414.79840087890625, 421.3484191894531, 416.3827209472656, 416.5849914550781, 419.1302490234375, 414.7456970214844, 419.312255859375, 415.2906494140625, 416.9346008300781, 416.1675720214844, 418.51458740234375, 417.76995849609375, 420.13006591796875, 418.7623596191406, 416.8761901855469, 418.8280029296875, 417.02777099609375, 417.6993408203125, 416.9612121582031, 419.1832580566406 ], "p50_ms": 417.3635559082031, "p95_ms": 420.190983581543, "mean_ms": 417.62241516113284 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" } }, "supplied_workspace_bytes": 0 } ], "errors_and_unsupported": [ { "feature": "Stream-K", "supported_public_control": false, "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." } ] }