{ "mode": "production-block-gate", "environment": { "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", "torch": "2.9.1+cu130", "torch_cuda": "13.0", "device": "NVIDIA GB10", "device_capability": [ 12, 1 ], "driver": null, "git_commit": null, "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", "checkpoint_sha256": null, "checkpoint_hash_note": "not calculated", "environment_switches": { "CUDA_DEVICE_MAX_CONNECTIONS": "1", "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", "CUDA_HOME": "/usr/local/cuda", "CUDA_INC_PATH": "/usr/local/cuda/include", "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", "CUDA_MODULE_LOADING": "EAGER", "CUDA_VERSION": "13.0.2", "H3_FUSED_ELEMENTWISE": "1", "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", "H3_NVFP4_MODULATE_FUSION": "1", "H3_NVFP4_SCALE_BACKEND": "vortex", "H3_NVFP4_SCALE_VERSION": "1", "H3_NVFP4_SWIGLU_FUSION": "1", "H3_SAGE_QKV_LAYOUT": "strided_nhd", "TORCH_COMPILE_DISABLE": "0", "TORCH_CUDA_ARCH_LIST": "12.1a", "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" }, "extension": { "cuda_version": 13000, "cublas_version": 130100, "cuda_runtime_version": 13000, "cublaslt_runtime_version": 130000, "stream_k_public_control": false, "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." }, "comfy_kitchen": "0.2.31 package without __version__" }, "workload": { "resolution": [ 1344, 768 ], "frames": 124, "steps": 12, "sampler_step": 1, "seed": 440420, "text_tokens": 100, "tokens": 37810, "hidden_shape": [ 37810, 5376 ], "segments": [ [ 0, 100, 1 ], [ 100, 514, 2 ], [ 514, 37810, 0 ] ] }, "retained_blocks": [ 0, 24, 49 ], "immutable_cloned_block_inputs": { "0": [ 37810, 5376 ], "24": [ 37810, 5376 ], "49": [ 37810, 5376 ] }, "fc2_boundary": { "block": 24, "gate_up_shape": [ 37810, 28672 ], "activation_qdata_shape": [ 37824, 7168 ], "weight_qdata_shape": [ 5376, 7168 ], "logical_mnk": [ 37810, 5376, 14336 ], "descriptor_mnk_after_padding": [ 37824, 5376, 14336 ], "producer": "vortex_native_quantize_swiglu_nvfp4", "no_bias": true }, "baseline_kernel_metadata": { "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", "descriptors": { "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", "block_scale_mode": "VEC16_UE4M3", "compute_and_scale": "FP32", "scalar_pointer_mode": "device", "bias": null, "beta": 0.0, "comfy_kitchen_version": "0.2.31" }, "profiler_cuda_events_available": false, "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", "top_cuda_events": [], "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" }, "heuristics": [ { "max_workspace_bytes": 0, "requested_count": 32, "returned_count": 5, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 4194304, "requested_count": 32, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 8388608, "requested_count": 32, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 16777216, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 33554432, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 67108864, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] } ], "explicit_split_k_checks": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 1, "reduction_scheme": 0 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 4 } } ], "selected": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0 }, "production_block_gate": [ { "block": 0, "production_method": "block.mlp.fc2.forward_swiglu", "candidate_vs_baseline": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" }, "baseline_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" }, "candidate_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 462.0343322753906, 519.793701171875, 463.6451721191406, 461.61883544921875, 461.3515930175781, 464.995361328125, 463.2869873046875, 462.8765563964844, 461.7970886230469, 465.5448913574219, 461.8421936035156, 465.0537109375, 464.8384094238281, 465.4224548339844, 463.8840637207031, 463.4743347167969, 465.61236572265625, 464.4220275878906, 464.6580810546875, 466.87091064453125 ], "p50_ms": 464.1530456542969, "p95_ms": 469.51705017089853, "mean_ms": 466.6511535644531 }, "candidate": { "samples_ms": [ 571.5564575195312, 422.9067077636719, 416.4256591796875, 419.8518981933594, 419.189453125, 420.7510070800781, 420.09149169921875, 419.6890869140625, 419.6440734863281, 420.3362731933594, 421.8302917480469, 417.80389404296875, 420.0198669433594, 420.9107971191406, 421.89117431640625, 420.67724609375, 421.85693359375, 422.25421142578125, 422.38531494140625, 423.3605041503906 ], "p50_ms": 420.71412658691406, "p95_ms": 430.77030181884777, "mean_ms": 428.17161712646487 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147", "expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147" } }, "dispatch_delta": { "attempts": 23, "successes": 23, "fallbacks": 0 } }, { "block": 24, "production_method": "block.mlp.fc2.forward_swiglu", "candidate_vs_baseline": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" }, "baseline_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" }, "candidate_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 462.13031005859375, 1074.8060302734375, 499.2489318847656, 460.1708984375, 459.4143371582031, 459.58917236328125, 458.5694274902344, 459.37249755859375, 457.447509765625, 462.1033020019531, 460.47076416015625, 458.94989013671875, 461.4758605957031, 461.214111328125, 458.8883972167969, 459.40771484375, 461.8358154296875, 465.3328857421875, 457.39892578125, 460.1976623535156 ], "p50_ms": 460.1842803955078, "p95_ms": 528.0267868041996, "mean_ms": 492.9012222290039 }, "candidate": { "samples_ms": [ 427.8656921386719, 427.14617919921875, 525.4254150390625, 420.90576171875, 421.28857421875, 422.8717041015625, 423.4268493652344, 422.6097106933594, 422.38580322265625, 422.58343505859375, 423.83770751953125, 422.4466552734375, 422.70172119140625, 422.8758544921875, 421.4642333984375, 422.4757995605469, 424.78662109375, 422.1180725097656, 421.7055358886719, 424.3546447753906 ], "p50_ms": 422.6557159423828, "p95_ms": 432.7436782836915, "mean_ms": 428.26379852294923 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75", "expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75" } }, "dispatch_delta": { "attempts": 23, "successes": 23, "fallbacks": 0 } }, { "block": 49, "production_method": "block.mlp.fc2.forward_swiglu", "candidate_vs_baseline": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" }, "baseline_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" }, "candidate_vs_traversal": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 457.59906005859375, 455.51739501953125, 454.50775146484375, 455.77972412109375, 454.094482421875, 453.72442626953125, 456.2005615234375, 457.22589111328125, 455.93505859375, 454.9208679199219, 454.7513732910156, 456.42205810546875, 456.1644592285156, 455.8465270996094, 459.75469970703125, 456.6866149902344, 456.8373107910156, 458.20733642578125, 456.2602233886719, 456.6114807128906 ], "p50_ms": 456.18251037597656, "p95_ms": 458.2847045898437, "mean_ms": 456.15236511230466 }, "candidate": { "samples_ms": [ 421.10394287109375, 427.0021057128906, 418.5732116699219, 420.71209716796875, 419.9263916015625, 420.857177734375, 417.8297424316406, 421.5742492675781, 421.7043762207031, 419.58203125, 419.05865478515625, 421.8150634765625, 421.2725524902344, 419.3066711425781, 419.87371826171875, 420.12255859375, 419.9741516113281, 420.2537841796875, 420.41552734375, 422.7670593261719 ], "p50_ms": 420.33465576171875, "p95_ms": 422.9788116455078, "mean_ms": 420.6862533569336 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6", "expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6" } }, "dispatch_delta": { "attempts": 23, "successes": 23, "fallbacks": 0 } } ], "errors_and_unsupported": [ { "feature": "Stream-K", "supported_public_control": false, "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." } ] }