{ "mode": "production-trajectory", "environment": { "platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39", "python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]", "torch": "2.9.1+cu130", "torch_cuda": "13.0", "device": "NVIDIA GB10", "device_capability": [ 12, 1 ], "driver": null, "git_commit": null, "checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors", "checkpoint_sha256": null, "checkpoint_hash_note": "not calculated", "environment_switches": { "CUDA_DEVICE_MAX_CONNECTIONS": "1", "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", "CUDA_HOME": "/usr/local/cuda", "CUDA_INC_PATH": "/usr/local/cuda/include", "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", "CUDA_MODULE_LOADING": "EAGER", "CUDA_VERSION": "13.0.2", "H3_FUSED_ELEMENTWISE": "1", "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", "H3_NVFP4_MODULATE_FUSION": "1", "H3_NVFP4_SCALE_BACKEND": "vortex", "H3_NVFP4_SCALE_VERSION": "1", "H3_NVFP4_SWIGLU_FUSION": "1", "TORCH_COMPILE_DISABLE": "0", "TORCH_CUDA_ARCH_LIST": "12.1a", "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" }, "extension": { "cuda_version": 13000, "cublas_version": 130100, "cuda_runtime_version": 13000, "cublaslt_runtime_version": 130000, "stream_k_public_control": false, "stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." }, "comfy_kitchen": "0.2.31 package without __version__" }, "workload": { "resolution": [ 1344, 768 ], "frames": 124, "steps": 2, "sampler_step": 1, "seed": 440420, "text_tokens": 100, "tokens": 37810, "hidden_shape": [ 37810, 5376 ], "segments": [ [ 0, 100, 1 ], [ 100, 514, 2 ], [ 514, 37810, 0 ] ] }, "retained_blocks": [ 0, 24, 49 ], "immutable_cloned_block_inputs": { "0": [ 37810, 5376 ], "24": [ 37810, 5376 ], "49": [ 37810, 5376 ] }, "fc2_boundary": { "block": 24, "gate_up_shape": [ 37810, 28672 ], "activation_qdata_shape": [ 37824, 7168 ], "weight_qdata_shape": [ 5376, 7168 ], "logical_mnk": [ 37810, 5376, 14336 ], "descriptor_mnk_after_padding": [ 37824, 5376, 14336 ], "producer": "vortex_native_quantize_swiglu_nvfp4", "no_bias": true }, "baseline_kernel_metadata": { "path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4", "descriptors": { "packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]", "block_scale_mode": "VEC16_UE4M3", "compute_and_scale": "FP32", "scalar_pointer_mode": "device", "bias": null, "beta": 0.0, "comfy_kitchen_version": "0.2.31" }, "profiler_cuda_events_available": false, "profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.", "top_cuda_events": [], "output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f" }, "heuristics": [ { "max_workspace_bytes": 0, "requested_count": 32, "returned_count": 5, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 4194304, "requested_count": 32, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 8388608, "requested_count": 32, "returned_count": 7, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 16777216, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 33554432, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] }, { "max_workspace_bytes": 67108864, "requested_count": 32, "returned_count": 6, "algorithms": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": -2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." } } ] } ], "explicit_split_k_checks": [ { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 1.0, "state": 0, "api_status": 0, "valid": true, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 1, "reduction_scheme": 0 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 2, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 2, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 4, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 4, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 8, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 8, "reduction_scheme": 4 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 2, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 2 } }, { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 16, "reduction_scheme": 4, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "required_workspace_bytes": 0, "waves": 0.0, "state": 15, "api_status": 15, "valid": false, "capabilities": { "split_k_support": 1, "reduction_scheme_mask": 6, "cta_swizzle_support": 0, "custom_option_max": 0, "strided_batch_support": 1, "out_of_place_result_support": 1, "tile_ids": [ 20 ], "stages_ids": [ 37 ], "inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs." }, "requested_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "custom_option": 0, "cta_swizzle": 0, "inner_shape": null, "cluster_shape": null, "split_k": 16, "reduction_scheme": 4 } } ], "selected": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0 }, "trajectory": { "steps": 2, "resolution": [ 1344, 768 ], "frames": 124, "seed": 440420, "baseline_seconds": 47.30056222799976, "candidate_seconds": 43.727866895999796, "improvement_percent": 7.553177306389591, "video_parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "7d5e50d8f0ee9a639feede8a0e6aea1bd160087634ecc7ae80d0ef5b96fd8e78", "expected_sha256": "7d5e50d8f0ee9a639feede8a0e6aea1bd160087634ecc7ae80d0ef5b96fd8e78" }, "audio_parity": { "bf16_exact": true, "different_elements": 0, "max_abs": 0.0, "mean_abs": 0.0, "actual_sha256": "df518aa8c645ec45e7cc46be58a941790f9766b420adf2cdcbaf14ef8b0e5605", "expected_sha256": "df518aa8c645ec45e7cc46be58a941790f9766b420adf2cdcbaf14ef8b0e5605" }, "bf16_exact": true, "selected_config": { "algorithm_id": 70, "tile_id": 20, "stages_id": 37, "split_k": 1, "reduction_scheme": 0, "custom_option": 0, "cta_swizzle": 0 }, "supplied_workspace_bytes": 0, "all_50_fc2_calls_replaced": true, "accepted_swiglu_producer_preserved": true, "accepted_gate_and_residual_path_preserved": true, "production_dispatch": true, "dispatch_delta": { "attempts": 100, "successes": 100, "fallbacks": 0 } }, "errors_and_unsupported": [ { "feature": "Stream-K", "supported_public_control": false, "reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control." } ] }