{ "mode": "sweep", "environment": { "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", "torch": "2.9.1+cu130", "torch_cuda": "13.0", "device": "NVIDIA GB10", "device_capability": [ 12, 1 ], "driver": null, "git_commit": null, "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", "checkpoint_sha256": null, "checkpoint_hash_note": "not calculated", "environment_switches": { "CUDA_DEVICE_MAX_CONNECTIONS": "1", "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", "CUDA_HOME": "/usr/local/cuda", "CUDA_INC_PATH": "/usr/local/cuda/include", "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", "CUDA_MODULE_LOADING": "EAGER", "CUDA_VERSION": "13.0.2", "H3_FUSED_ELEMENTWISE": "1", "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", "H3_NVFP4_MODULATE_FUSION": "1", "H3_NVFP4_SCALE_BACKEND": "vortex", "H3_NVFP4_SCALE_VERSION": "1", "H3_NVFP4_SWIGLU_FUSION": "1", "H3_SAGE_QKV_LAYOUT": "strided_nhd", "TORCH_COMPILE_DISABLE": "0", "TORCH_CUDA_ARCH_LIST": "12.1a", "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" }, "extension": { "cuda_version": 13000, "cublas_version": 130100, "stream_k_public_control": false, "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." }, "comfy_kitchen": "0.2.31 package without __version__" }, "workload": { "resolution": [ 1344, 768 ], "frames": 124, "steps": 12, "sampler_step": 1, "seed": 440420, "text_tokens": 100, "tokens": 37810, "hidden_shape": [ 37810, 5376 ], "segments": [ [ 0, 100, 1 ], [ 100, 514, 2 ], [ 514, 37810, 0 ] ] }, "retained_blocks": [ 0, 24, 49 ], "immutable_cloned_block_inputs": { "0": [ 37810, 5376 ], "24": [ 37810, 5376 ], "49": [ 37810, 5376 ] }, "fc2_boundary": { "block": 24, "gate_up_shape": [ 37810, 28672 ], "activation_qdata_shape": [ 37824, 7168 ], "weight_qdata_shape": [ 5376, 7168 ], "logical_mnk": [ 37810, 5376, 14336 ], "descriptor_mnk_after_padding": [ 37824, 5376, 14336 ], "producer": "vortex_native_quantize_swiglu_nvfp4", "no_bias": true }, "baseline_kernel_metadata": { "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", "descriptors": { "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", "block_scale_mode": "VEC16_UE4M3", "compute_and_scale": "FP32", "scalar_pointer_mode": "device", "bias": null, "beta": 0.0, "comfy_kitchen_version": "0.2.31" }, "profiler_cuda_events_available": false, "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", "top_cuda_events": [], "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" }, "heuristics": [ { "max_workspace_bytes": 0, "requested_count": 64, "returned_count": 5, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 4194304, "requested_count": 64, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 8388608, "requested_count": 64, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 16777216, "requested_count": 64, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 33554432, "requested_count": 64, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 67108864, "requested_count": 64, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] } ], "explicit_split_k_checks": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 1, "reduction_scheme": 0 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 4 } } ], "selected": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 1, "reduction_scheme": 0 }, "candidates": [ { "selected_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 67108864, "required_workspace_bytes": 0, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "fc2_only": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 52.528385162353516, 53.54975891113281, 52.53424072265625, 52.453086853027344, 52.415489196777344, 53.50998306274414, 52.42780685424805, 52.526817321777344, 52.41756820678711, 52.83504104614258, 54.08259201049805, 52.47983932495117, 52.43289566040039, 52.59030532836914, 52.4370231628418, 52.52783966064453, 52.4318733215332, 52.5145263671875, 52.424705505371094, 52.57904052734375 ], "p50_ms": 52.52067184448242, "p95_ms": 53.57640056610108, "mean_ms": 52.68494091033936, "dense_tflop_s_p50": 110.9669507956279 }, "candidate": { "samples_ms": [ 52.557823181152344, 53.029598236083984, 52.549888610839844, 52.55766296386719, 52.51689529418945, 53.11779022216797, 54.02726364135742, 52.74710464477539, 52.530174255371094, 52.540096282958984, 52.56089782714844, 52.54924774169922, 52.527103424072266, 52.53104019165039, 53.064640045166016, 53.72809600830078, 53.41603088378906, 52.76732635498047, 52.54451370239258, 53.056480407714844 ], "p50_ms": 52.55936050415039, "p95_ms": 53.74305438995361, "mean_ms": 52.84598369598389, "dense_tflop_s_p50": 110.88526862612385 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" } }, "accepted_producer_plus_fc2": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 75.45340728759766, 75.004638671875, 76.25904083251953, 75.5995864868164, 75.02406311035156, 75.02108764648438, 74.97523498535156, 76.13423919677734, 76.12108612060547, 76.02950286865234, 76.11692810058594, 75.5208969116211, 75.1124496459961, 75.58025360107422, 74.99574279785156, 76.13922882080078, 76.13350677490234, 76.16307067871094, 76.70272064208984, 75.0445785522461 ], "p50_ms": 75.58992004394531, "p95_ms": 76.28122482299804, "mean_ms": 75.65656318664551, "dense_tflop_s_p50": 77.10100506696888 }, "candidate": { "samples_ms": [ 75.16223907470703, 75.09487915039062, 76.4610595703125, 75.77107238769531, 76.33715057373047, 76.33900451660156, 75.62342071533203, 76.34333038330078, 75.10733032226562, 75.13775634765625, 76.2429428100586, 75.65090942382812, 76.9865951538086, 76.45164489746094, 75.12268829345703, 76.36873626708984, 75.21382141113281, 75.16233825683594, 75.15340423583984, 75.63337707519531 ], "p50_ms": 75.64214324951172, "p95_ms": 76.48733634948731, "mean_ms": 75.76818504333497, "dense_tflop_s_p50": 77.04777466571349 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" } }, "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" }, { "selected_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 67108864, "required_workspace_bytes": 0, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "fc2_only": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 53.98486328125, 53.7977294921875, 53.85007858276367, 53.532447814941406, 53.588897705078125, 53.95724868774414, 53.63395309448242, 53.011390686035156, 53.56748962402344, 54.085472106933594, 54.20851135253906, 53.808990478515625, 53.888832092285156, 54.48992156982422, 53.129215240478516, 53.52640151977539, 53.18348693847656, 53.60192108154297, 53.001953125, 53.25020980834961 ], "p50_ms": 53.617937088012695, "p95_ms": 54.22258186340332, "mean_ms": 53.65495071411133, "dense_tflop_s_p50": 108.69606562358724 }, "candidate": { "samples_ms": [ 15.689727783203125, 15.966943740844727, 15.744000434875488, 15.626208305358887, 15.685664176940918, 15.68841552734375, 15.331328392028809, 15.71504020690918, 14.792703628540039, 15.66592025756836, 14.791680335998535, 15.645536422729492, 14.812159538269043, 14.813887596130371, 14.793631553649902, 15.76848030090332, 15.557696342468262, 15.463264465332031, 14.7957763671875, 15.723199844360352 ], "p50_ms": 15.63587236404419, "p95_ms": 15.778403472900392, "mean_ms": 15.403563261032104, "dense_tflop_s_p50": 372.73640207770177 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" } }, "accepted_producer_plus_fc2": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 76.28396606445312, 76.32662200927734, 76.3351058959961, 76.94000244140625, 76.97782135009766, 76.59494018554688, 76.87884521484375, 77.03215789794922, 76.31446075439453, 76.24060821533203, 76.16102600097656, 76.39116668701172, 77.31603240966797, 76.89494323730469, 75.6058578491211, 76.90121459960938, 76.15897369384766, 77.05766296386719, 77.53421020507812, 76.94409942626953 ], "p50_ms": 76.73689270019531, "p95_ms": 77.32694129943847, "mean_ms": 76.64448585510254, "dense_tflop_s_p50": 75.94859008807853 }, "candidate": { "samples_ms": [ 38.138526916503906, 39.03964614868164, 39.00892639160156, 38.78895950317383, 38.25788879394531, 39.111358642578125, 37.374977111816406, 39.55583953857422, 38.676639556884766, 39.315616607666016, 39.011199951171875, 38.85724639892578, 38.27609634399414, 39.127777099609375, 38.3109130859375, 39.1360969543457, 38.23001480102539, 39.10319900512695, 38.683841705322266, 38.4785270690918 ], "p50_ms": 38.823102951049805, "p95_ms": 39.32762775421143, "mean_ms": 38.72416458129883, "dense_tflop_s_p50": 150.11831526368 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" } }, "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" } ], "errors_and_unsupported": [ { "feature": "Stream-K", "supported_public_control": false, "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." } ] }