{ "mode": "shape-gate", "environment": { "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", "torch": "2.9.1+cu130", "torch_cuda": "13.0", "device": "NVIDIA GB10", "device_capability": [ 12, 1 ], "driver": null, "git_commit": null, "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", "checkpoint_sha256": null, "checkpoint_hash_note": "not calculated", "environment_switches": { "CUDA_DEVICE_MAX_CONNECTIONS": "1", "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", "CUDA_HOME": "/usr/local/cuda", "CUDA_INC_PATH": "/usr/local/cuda/include", "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", "CUDA_MODULE_LOADING": "EAGER", "CUDA_VERSION": "13.0.2", "H3_FUSED_ELEMENTWISE": "1", "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", "H3_NVFP4_MODULATE_FUSION": "1", "H3_NVFP4_SCALE_BACKEND": "vortex", "H3_NVFP4_SCALE_VERSION": "1", "H3_NVFP4_SWIGLU_FUSION": "1", "H3_SAGE_QKV_LAYOUT": "strided_nhd", "TORCH_COMPILE_DISABLE": "0", "TORCH_CUDA_ARCH_LIST": "12.1a", "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" }, "extension": { "cuda_version": 13000, "cublas_version": 130100, "stream_k_public_control": false, "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." }, "comfy_kitchen": "0.2.31 package without __version__" }, "workload": { "resolution": [ 1344, 768 ], "frames": 124, "steps": 12, "sampler_step": 1, "seed": 440420, "text_tokens": 100, "tokens": 37810, "hidden_shape": [ 37810, 5376 ], "segments": [ [ 0, 100, 1 ], [ 100, 514, 2 ], [ 514, 37810, 0 ] ] }, "retained_blocks": [ 0, 24, 49 ], "immutable_cloned_block_inputs": { "0": [ 37810, 5376 ], "24": [ 37810, 5376 ], "49": [ 37810, 5376 ] }, "fc2_boundary": { "block": 24, "gate_up_shape": [ 37810, 28672 ], "activation_qdata_shape": [ 37824, 7168 ], "weight_qdata_shape": [ 5376, 7168 ], "logical_mnk": [ 37810, 5376, 14336 ], "descriptor_mnk_after_padding": [ 37824, 5376, 14336 ], "producer": "vortex_native_quantize_swiglu_nvfp4", "no_bias": true }, "baseline_kernel_metadata": { "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", "descriptors": { "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", "block_scale_mode": "VEC16_UE4M3", "compute_and_scale": "FP32", "scalar_pointer_mode": "device", "bias": null, "beta": 0.0, "comfy_kitchen_version": "0.2.31" }, "profiler_cuda_events_available": false, "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", "top_cuda_events": [], "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" }, "heuristics": [ { "max_workspace_bytes": 0, "requested_count": 32, "returned_count": 5, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 4194304, "requested_count": 32, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 8388608, "requested_count": 32, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 16777216, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 33554432, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 67108864, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] } ], "explicit_split_k_checks": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 1, "reduction_scheme": 0 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 4 } } ], "selected": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0 }, "shape_gate": [ { "text_tokens": 1, "logical_rows": 37711, "packed_rows": 37712, "parity": { "bf16_exact": false, "different_elements": 2, "max_abs": 0.03125, "mean_abs": 1.553468603754382e-10, "actual_sha256": "c1ee129ef17b93c7814dc0dfb4dfc9fc0218a3d4252cddb02c80eb98e0471ab5", "expected_sha256": "e017519cd7e89a93770f7648e81c637ccac517fe2715dd213f46627dfdfed711" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 57.960289001464844 ], "p50_ms": 57.960289001464844, "p95_ms": 57.960289001464844, "mean_ms": 57.960289001464844, "dense_tflop_s_p50": 100.28933571475277 }, "candidate": { "samples_ms": [ 18.854400634765625 ], "p50_ms": 18.854400634765625, "p95_ms": 18.854400634765625, "mean_ms": 18.854400634765625, "dense_tflop_s_p50": 308.2993193150771 }, "parity": { "bf16_exact": false, "different_elements": 2, "max_abs": 0.03125, "mean_abs": 1.553468603754382e-10, "actual_sha256": "c1ee129ef17b93c7814dc0dfb4dfc9fc0218a3d4252cddb02c80eb98e0471ab5", "expected_sha256": "e017519cd7e89a93770f7648e81c637ccac517fe2715dd213f46627dfdfed711" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": false }, { "text_tokens": 15, "logical_rows": 37725, "packed_rows": 37728, "parity": { "bf16_exact": false, "different_elements": 2, "max_abs": 0.03125, "mean_abs": 1.5528919816709674e-10, "actual_sha256": "c47d817c29526fcf16d800bba07d8ac659831fe16076c07b5b5ba5347ed63c7e", "expected_sha256": "310f6ecf71f385122512a2289282a9a8723e231aa252223201e23e9857219eda" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 57.9911994934082 ], "p50_ms": 57.9911994934082, "p95_ms": 57.9911994934082, "mean_ms": 57.9911994934082, "dense_tflop_s_p50": 100.27309146900781 }, "candidate": { "samples_ms": [ 18.6856632232666 ], "p50_ms": 18.6856632232666, "p95_ms": 18.6856632232666, "mean_ms": 18.6856632232666, "dense_tflop_s_p50": 311.1988470368801 }, "parity": { "bf16_exact": false, "different_elements": 2, "max_abs": 0.03125, "mean_abs": 1.5528919816709674e-10, "actual_sha256": "c47d817c29526fcf16d800bba07d8ac659831fe16076c07b5b5ba5347ed63c7e", "expected_sha256": "310f6ecf71f385122512a2289282a9a8723e231aa252223201e23e9857219eda" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": false }, { "text_tokens": 32, "logical_rows": 37742, "packed_rows": 37744, "parity": { "bf16_exact": false, "different_elements": 2, "max_abs": 0.03125, "mean_abs": 1.5521925411654536e-10, "actual_sha256": "c50b1f08167249b90c80632f4bf1c379b252446cd766ac1ba331087c2de2744d", "expected_sha256": "42816a3c87083bd5ed98a975cd6c7b31f35746761089fa8f2d19c92f47fed3df" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 57.83795166015625 ], "p50_ms": 57.83795166015625, "p95_ms": 57.83795166015625, "mean_ms": 57.83795166015625, "dense_tflop_s_p50": 100.58408148350189 }, "candidate": { "samples_ms": [ 18.732128143310547 ], "p50_ms": 18.732128143310547, "p95_ms": 18.732128143310547, "mean_ms": 18.732128143310547, "dense_tflop_s_p50": 310.56680789905454 }, "parity": { "bf16_exact": false, "different_elements": 2, "max_abs": 0.03125, "mean_abs": 1.5521925411654536e-10, "actual_sha256": "c50b1f08167249b90c80632f4bf1c379b252446cd766ac1ba331087c2de2744d", "expected_sha256": "42816a3c87083bd5ed98a975cd6c7b31f35746761089fa8f2d19c92f47fed3df" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": false }, { "text_tokens": 64, "logical_rows": 37774, "packed_rows": 37776, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "9b272d355084277500dc89e4d633d44c7daa81481b092d3b96f577e2c207ddac", "expected_sha256": "9b272d355084277500dc89e4d633d44c7daa81481b092d3b96f577e2c207ddac" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 58.0667839050293 ], "p50_ms": 58.0667839050293, "p95_ms": 58.0667839050293, "mean_ms": 58.0667839050293, "dense_tflop_s_p50": 100.27264044192569 }, "candidate": { "samples_ms": [ 19.07366371154785 ], "p50_ms": 19.07366371154785, "p95_ms": 19.07366371154785, "mean_ms": 19.07366371154785, "dense_tflop_s_p50": 305.26435991439087 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "9b272d355084277500dc89e4d633d44c7daa81481b092d3b96f577e2c207ddac", "expected_sha256": "9b272d355084277500dc89e4d633d44c7daa81481b092d3b96f577e2c207ddac" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": true }, { "text_tokens": 99, "logical_rows": 37809, "packed_rows": 37824, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "dcc4ac6d680a76bf563145e569ea114364a0ff3e76f3f2f656d1dd70adfba812", "expected_sha256": "dcc4ac6d680a76bf563145e569ea114364a0ff3e76f3f2f656d1dd70adfba812" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 57.88643264770508 ], "p50_ms": 57.88643264770508, "p95_ms": 57.88643264770508, "mean_ms": 57.88643264770508, "dense_tflop_s_p50": 100.6782487895987 }, "candidate": { "samples_ms": [ 18.773759841918945 ], "p50_ms": 18.773759841918945, "p95_ms": 18.773759841918945, "mean_ms": 18.773759841918945, "dense_tflop_s_p50": 310.4282102637308 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "dcc4ac6d680a76bf563145e569ea114364a0ff3e76f3f2f656d1dd70adfba812", "expected_sha256": "dcc4ac6d680a76bf563145e569ea114364a0ff3e76f3f2f656d1dd70adfba812" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": true }, { "text_tokens": 100, "logical_rows": 37810, "packed_rows": 37824, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 57.96451187133789 ], "p50_ms": 57.96451187133789, "p95_ms": 57.96451187133789, "mean_ms": 57.96451187133789, "dense_tflop_s_p50": 100.54529263105621 }, "candidate": { "samples_ms": [ 18.34467124938965 ], "p50_ms": 18.34467124938965, "p95_ms": 18.34467124938965, "mean_ms": 18.34467124938965, "dense_tflop_s_p50": 317.6976424973496 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f", "expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": true }, { "text_tokens": 128, "logical_rows": 37838, "packed_rows": 37840, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "adb567fdd1568c406d514983f0dbf799b3d9121656788a853fe8620bca7cb67b", "expected_sha256": "adb567fdd1568c406d514983f0dbf799b3d9121656788a853fe8620bca7cb67b" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 57.80854415893555 ], "p50_ms": 57.80854415893555, "p95_ms": 57.80854415893555, "mean_ms": 57.80854415893555, "dense_tflop_s_p50": 100.89122346864156 }, "candidate": { "samples_ms": [ 19.78188705444336 ], "p50_ms": 19.78188705444336, "p95_ms": 19.78188705444336, "mean_ms": 19.78188705444336, "dense_tflop_s_p50": 294.8340939913488 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "adb567fdd1568c406d514983f0dbf799b3d9121656788a853fe8620bca7cb67b", "expected_sha256": "adb567fdd1568c406d514983f0dbf799b3d9121656788a853fe8620bca7cb67b" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": true }, { "text_tokens": 256, "logical_rows": 37966, "packed_rows": 37968, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "e264df4d576726d22bf75e6e33a7bc3880543af28a6a528a8332b6fe25aa7676", "expected_sha256": "e264df4d576726d22bf75e6e33a7bc3880543af28a6a528a8332b6fe25aa7676" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 58.237281799316406 ], "p50_ms": 58.237281799316406, "p95_ms": 58.237281799316406, "mean_ms": 58.237281799316406, "dense_tflop_s_p50": 100.4872578586024 }, "candidate": { "samples_ms": [ 18.483808517456055 ], "p50_ms": 18.483808517456055, "p95_ms": 18.483808517456055, "mean_ms": 18.483808517456055, "dense_tflop_s_p50": 316.60708601397215 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "e264df4d576726d22bf75e6e33a7bc3880543af28a6a528a8332b6fe25aa7676", "expected_sha256": "e264df4d576726d22bf75e6e33a7bc3880543af28a6a528a8332b6fe25aa7676" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": true }, { "text_tokens": 512, "logical_rows": 38222, "packed_rows": 38224, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "b6276cda42be6a858bab99ae69a211c71400104f39bc28b2d6723ad37be015b9", "expected_sha256": "b6276cda42be6a858bab99ae69a211c71400104f39bc28b2d6723ad37be015b9" }, "timing": { "order": "AB/BA alternates by round", "baseline": { "samples_ms": [ 58.96633529663086 ], "p50_ms": 58.96633529663086, "p95_ms": 58.96633529663086, "mean_ms": 58.96633529663086, "dense_tflop_s_p50": 99.91403968970451 }, "candidate": { "samples_ms": [ 20.196256637573242 ], "p50_ms": 20.196256637573242, "p95_ms": 20.196256637573242, "mean_ms": 20.196256637573242, "dense_tflop_s_p50": 291.71568132201764 }, "parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "b6276cda42be6a858bab99ae69a211c71400104f39bc28b2d6723ad37be015b9", "expected_sha256": "b6276cda42be6a858bab99ae69a211c71400104f39bc28b2d6723ad37be015b9" } }, "checked": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, "supplied_workspace_bytes": 0, "passed": true } ], "errors_and_unsupported": [ { "feature": "Stream-K", "supported_public_control": false, "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." } ] }