2026 lines
60 KiB
JSON
2026 lines
60 KiB
JSON
|
|
{
|
||
|
|
"mode": "production-block-gate",
|
||
|
|
"environment": {
|
||
|
|
"platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
|
||
|
|
"python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]",
|
||
|
|
"torch": "2.9.1+cu130",
|
||
|
|
"torch_cuda": "13.0",
|
||
|
|
"device": "NVIDIA GB10",
|
||
|
|
"device_capability": [
|
||
|
|
12,
|
||
|
|
1
|
||
|
|
],
|
||
|
|
"driver": null,
|
||
|
|
"git_commit": null,
|
||
|
|
"checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors",
|
||
|
|
"checkpoint_sha256": null,
|
||
|
|
"checkpoint_hash_note": "not calculated",
|
||
|
|
"environment_switches": {
|
||
|
|
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
|
||
|
|
"CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4",
|
||
|
|
"CUDA_HOME": "/usr/local/cuda",
|
||
|
|
"CUDA_INC_PATH": "/usr/local/cuda/include",
|
||
|
|
"CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1",
|
||
|
|
"CUDA_MODULE_LOADING": "EAGER",
|
||
|
|
"CUDA_VERSION": "13.0.2",
|
||
|
|
"H3_FUSED_ELEMENTWISE": "1",
|
||
|
|
"H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors",
|
||
|
|
"H3_NVFP4_MODULATE_FUSION": "1",
|
||
|
|
"H3_NVFP4_SCALE_BACKEND": "vortex",
|
||
|
|
"H3_NVFP4_SCALE_VERSION": "1",
|
||
|
|
"H3_NVFP4_SWIGLU_FUSION": "1",
|
||
|
|
"H3_SAGE_QKV_LAYOUT": "strided_nhd",
|
||
|
|
"TORCH_COMPILE_DISABLE": "0",
|
||
|
|
"TORCH_CUDA_ARCH_LIST": "12.1a",
|
||
|
|
"TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions"
|
||
|
|
},
|
||
|
|
"extension": {
|
||
|
|
"cuda_version": 13000,
|
||
|
|
"cublas_version": 130100,
|
||
|
|
"cuda_runtime_version": 13000,
|
||
|
|
"cublaslt_runtime_version": 130000,
|
||
|
|
"stream_k_public_control": false,
|
||
|
|
"stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control."
|
||
|
|
},
|
||
|
|
"comfy_kitchen": "0.2.31 package without __version__"
|
||
|
|
},
|
||
|
|
"workload": {
|
||
|
|
"resolution": [
|
||
|
|
1344,
|
||
|
|
768
|
||
|
|
],
|
||
|
|
"frames": 124,
|
||
|
|
"steps": 12,
|
||
|
|
"sampler_step": 1,
|
||
|
|
"seed": 440420,
|
||
|
|
"text_tokens": 100,
|
||
|
|
"tokens": 37810,
|
||
|
|
"hidden_shape": [
|
||
|
|
37810,
|
||
|
|
5376
|
||
|
|
],
|
||
|
|
"segments": [
|
||
|
|
[
|
||
|
|
0,
|
||
|
|
100,
|
||
|
|
1
|
||
|
|
],
|
||
|
|
[
|
||
|
|
100,
|
||
|
|
514,
|
||
|
|
2
|
||
|
|
],
|
||
|
|
[
|
||
|
|
514,
|
||
|
|
37810,
|
||
|
|
0
|
||
|
|
]
|
||
|
|
]
|
||
|
|
},
|
||
|
|
"retained_blocks": [
|
||
|
|
0,
|
||
|
|
24,
|
||
|
|
49
|
||
|
|
],
|
||
|
|
"immutable_cloned_block_inputs": {
|
||
|
|
"0": [
|
||
|
|
37810,
|
||
|
|
5376
|
||
|
|
],
|
||
|
|
"24": [
|
||
|
|
37810,
|
||
|
|
5376
|
||
|
|
],
|
||
|
|
"49": [
|
||
|
|
37810,
|
||
|
|
5376
|
||
|
|
]
|
||
|
|
},
|
||
|
|
"fc2_boundary": {
|
||
|
|
"block": 24,
|
||
|
|
"gate_up_shape": [
|
||
|
|
37810,
|
||
|
|
28672
|
||
|
|
],
|
||
|
|
"activation_qdata_shape": [
|
||
|
|
37824,
|
||
|
|
7168
|
||
|
|
],
|
||
|
|
"weight_qdata_shape": [
|
||
|
|
5376,
|
||
|
|
7168
|
||
|
|
],
|
||
|
|
"logical_mnk": [
|
||
|
|
37810,
|
||
|
|
5376,
|
||
|
|
14336
|
||
|
|
],
|
||
|
|
"descriptor_mnk_after_padding": [
|
||
|
|
37824,
|
||
|
|
5376,
|
||
|
|
14336
|
||
|
|
],
|
||
|
|
"producer": "vortex_native_quantize_swiglu_nvfp4",
|
||
|
|
"no_bias": true
|
||
|
|
},
|
||
|
|
"baseline_kernel_metadata": {
|
||
|
|
"path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4",
|
||
|
|
"descriptors": {
|
||
|
|
"packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]",
|
||
|
|
"block_scale_mode": "VEC16_UE4M3",
|
||
|
|
"compute_and_scale": "FP32",
|
||
|
|
"scalar_pointer_mode": "device",
|
||
|
|
"bias": null,
|
||
|
|
"beta": 0.0,
|
||
|
|
"comfy_kitchen_version": "0.2.31"
|
||
|
|
},
|
||
|
|
"profiler_cuda_events_available": false,
|
||
|
|
"profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.",
|
||
|
|
"top_cuda_events": [],
|
||
|
|
"output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
||
|
|
},
|
||
|
|
"heuristics": [
|
||
|
|
{
|
||
|
|
"max_workspace_bytes": 0,
|
||
|
|
"requested_count": 32,
|
||
|
|
"returned_count": 5,
|
||
|
|
"algorithms": [
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
}
|
||
|
|
]
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"max_workspace_bytes": 4194304,
|
||
|
|
"requested_count": 32,
|
||
|
|
"returned_count": 7,
|
||
|
|
"algorithms": [
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
}
|
||
|
|
]
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"max_workspace_bytes": 8388608,
|
||
|
|
"requested_count": 32,
|
||
|
|
"returned_count": 7,
|
||
|
|
"algorithms": [
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
}
|
||
|
|
]
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"max_workspace_bytes": 16777216,
|
||
|
|
"requested_count": 32,
|
||
|
|
"returned_count": 6,
|
||
|
|
"algorithms": [
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
}
|
||
|
|
]
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"max_workspace_bytes": 33554432,
|
||
|
|
"requested_count": 32,
|
||
|
|
"returned_count": 6,
|
||
|
|
"algorithms": [
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
}
|
||
|
|
]
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"max_workspace_bytes": 67108864,
|
||
|
|
"requested_count": 32,
|
||
|
|
"returned_count": 6,
|
||
|
|
"algorithms": [
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": -2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
}
|
||
|
|
}
|
||
|
|
]
|
||
|
|
}
|
||
|
|
],
|
||
|
|
"explicit_split_k_checks": [
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 1.0,
|
||
|
|
"state": 0,
|
||
|
|
"api_status": 0,
|
||
|
|
"valid": true,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 2,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 0.0,
|
||
|
|
"state": 15,
|
||
|
|
"api_status": 15,
|
||
|
|
"valid": false,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 2,
|
||
|
|
"reduction_scheme": 2
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 2,
|
||
|
|
"reduction_scheme": 4,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 0.0,
|
||
|
|
"state": 15,
|
||
|
|
"api_status": 15,
|
||
|
|
"valid": false,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 2,
|
||
|
|
"reduction_scheme": 4
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 4,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 0.0,
|
||
|
|
"state": 15,
|
||
|
|
"api_status": 15,
|
||
|
|
"valid": false,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 4,
|
||
|
|
"reduction_scheme": 2
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 4,
|
||
|
|
"reduction_scheme": 4,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 0.0,
|
||
|
|
"state": 15,
|
||
|
|
"api_status": 15,
|
||
|
|
"valid": false,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 4,
|
||
|
|
"reduction_scheme": 4
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 8,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 0.0,
|
||
|
|
"state": 15,
|
||
|
|
"api_status": 15,
|
||
|
|
"valid": false,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 8,
|
||
|
|
"reduction_scheme": 2
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 8,
|
||
|
|
"reduction_scheme": 4,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 0.0,
|
||
|
|
"state": 15,
|
||
|
|
"api_status": 15,
|
||
|
|
"valid": false,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 8,
|
||
|
|
"reduction_scheme": 4
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 16,
|
||
|
|
"reduction_scheme": 2,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 0.0,
|
||
|
|
"state": 15,
|
||
|
|
"api_status": 15,
|
||
|
|
"valid": false,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 16,
|
||
|
|
"reduction_scheme": 2
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 16,
|
||
|
|
"reduction_scheme": 4,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"required_workspace_bytes": 0,
|
||
|
|
"waves": 0.0,
|
||
|
|
"state": 15,
|
||
|
|
"api_status": 15,
|
||
|
|
"valid": false,
|
||
|
|
"capabilities": {
|
||
|
|
"split_k_support": 1,
|
||
|
|
"reduction_scheme_mask": 6,
|
||
|
|
"cta_swizzle_support": 0,
|
||
|
|
"custom_option_max": 0,
|
||
|
|
"strided_batch_support": 1,
|
||
|
|
"out_of_place_result_support": 1,
|
||
|
|
"tile_ids": [
|
||
|
|
20
|
||
|
|
],
|
||
|
|
"stages_ids": [
|
||
|
|
37
|
||
|
|
],
|
||
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
||
|
|
},
|
||
|
|
"requested_config": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0,
|
||
|
|
"inner_shape": null,
|
||
|
|
"cluster_shape": null,
|
||
|
|
"split_k": 16,
|
||
|
|
"reduction_scheme": 4
|
||
|
|
}
|
||
|
|
}
|
||
|
|
],
|
||
|
|
"selected": {
|
||
|
|
"algorithm_id": 70,
|
||
|
|
"tile_id": 20,
|
||
|
|
"stages_id": 37,
|
||
|
|
"split_k": 1,
|
||
|
|
"reduction_scheme": 0,
|
||
|
|
"custom_option": 0,
|
||
|
|
"cta_swizzle": 0
|
||
|
|
},
|
||
|
|
"production_block_gate": [
|
||
|
|
{
|
||
|
|
"block": 0,
|
||
|
|
"production_method": "block.mlp.fc2.forward_swiglu",
|
||
|
|
"candidate_vs_baseline": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147",
|
||
|
|
"expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147"
|
||
|
|
},
|
||
|
|
"baseline_vs_traversal": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147",
|
||
|
|
"expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147"
|
||
|
|
},
|
||
|
|
"candidate_vs_traversal": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147",
|
||
|
|
"expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147"
|
||
|
|
},
|
||
|
|
"timing": {
|
||
|
|
"order": "AB/BA alternates by round",
|
||
|
|
"baseline": {
|
||
|
|
"samples_ms": [
|
||
|
|
462.0343322753906,
|
||
|
|
519.793701171875,
|
||
|
|
463.6451721191406,
|
||
|
|
461.61883544921875,
|
||
|
|
461.3515930175781,
|
||
|
|
464.995361328125,
|
||
|
|
463.2869873046875,
|
||
|
|
462.8765563964844,
|
||
|
|
461.7970886230469,
|
||
|
|
465.5448913574219,
|
||
|
|
461.8421936035156,
|
||
|
|
465.0537109375,
|
||
|
|
464.8384094238281,
|
||
|
|
465.4224548339844,
|
||
|
|
463.8840637207031,
|
||
|
|
463.4743347167969,
|
||
|
|
465.61236572265625,
|
||
|
|
464.4220275878906,
|
||
|
|
464.6580810546875,
|
||
|
|
466.87091064453125
|
||
|
|
],
|
||
|
|
"p50_ms": 464.1530456542969,
|
||
|
|
"p95_ms": 469.51705017089853,
|
||
|
|
"mean_ms": 466.6511535644531
|
||
|
|
},
|
||
|
|
"candidate": {
|
||
|
|
"samples_ms": [
|
||
|
|
571.5564575195312,
|
||
|
|
422.9067077636719,
|
||
|
|
416.4256591796875,
|
||
|
|
419.8518981933594,
|
||
|
|
419.189453125,
|
||
|
|
420.7510070800781,
|
||
|
|
420.09149169921875,
|
||
|
|
419.6890869140625,
|
||
|
|
419.6440734863281,
|
||
|
|
420.3362731933594,
|
||
|
|
421.8302917480469,
|
||
|
|
417.80389404296875,
|
||
|
|
420.0198669433594,
|
||
|
|
420.9107971191406,
|
||
|
|
421.89117431640625,
|
||
|
|
420.67724609375,
|
||
|
|
421.85693359375,
|
||
|
|
422.25421142578125,
|
||
|
|
422.38531494140625,
|
||
|
|
423.3605041503906
|
||
|
|
],
|
||
|
|
"p50_ms": 420.71412658691406,
|
||
|
|
"p95_ms": 430.77030181884777,
|
||
|
|
"mean_ms": 428.17161712646487
|
||
|
|
},
|
||
|
|
"parity": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147",
|
||
|
|
"expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147"
|
||
|
|
}
|
||
|
|
},
|
||
|
|
"dispatch_delta": {
|
||
|
|
"attempts": 23,
|
||
|
|
"successes": 23,
|
||
|
|
"fallbacks": 0
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"block": 24,
|
||
|
|
"production_method": "block.mlp.fc2.forward_swiglu",
|
||
|
|
"candidate_vs_baseline": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75",
|
||
|
|
"expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75"
|
||
|
|
},
|
||
|
|
"baseline_vs_traversal": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75",
|
||
|
|
"expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75"
|
||
|
|
},
|
||
|
|
"candidate_vs_traversal": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75",
|
||
|
|
"expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75"
|
||
|
|
},
|
||
|
|
"timing": {
|
||
|
|
"order": "AB/BA alternates by round",
|
||
|
|
"baseline": {
|
||
|
|
"samples_ms": [
|
||
|
|
462.13031005859375,
|
||
|
|
1074.8060302734375,
|
||
|
|
499.2489318847656,
|
||
|
|
460.1708984375,
|
||
|
|
459.4143371582031,
|
||
|
|
459.58917236328125,
|
||
|
|
458.5694274902344,
|
||
|
|
459.37249755859375,
|
||
|
|
457.447509765625,
|
||
|
|
462.1033020019531,
|
||
|
|
460.47076416015625,
|
||
|
|
458.94989013671875,
|
||
|
|
461.4758605957031,
|
||
|
|
461.214111328125,
|
||
|
|
458.8883972167969,
|
||
|
|
459.40771484375,
|
||
|
|
461.8358154296875,
|
||
|
|
465.3328857421875,
|
||
|
|
457.39892578125,
|
||
|
|
460.1976623535156
|
||
|
|
],
|
||
|
|
"p50_ms": 460.1842803955078,
|
||
|
|
"p95_ms": 528.0267868041996,
|
||
|
|
"mean_ms": 492.9012222290039
|
||
|
|
},
|
||
|
|
"candidate": {
|
||
|
|
"samples_ms": [
|
||
|
|
427.8656921386719,
|
||
|
|
427.14617919921875,
|
||
|
|
525.4254150390625,
|
||
|
|
420.90576171875,
|
||
|
|
421.28857421875,
|
||
|
|
422.8717041015625,
|
||
|
|
423.4268493652344,
|
||
|
|
422.6097106933594,
|
||
|
|
422.38580322265625,
|
||
|
|
422.58343505859375,
|
||
|
|
423.83770751953125,
|
||
|
|
422.4466552734375,
|
||
|
|
422.70172119140625,
|
||
|
|
422.8758544921875,
|
||
|
|
421.4642333984375,
|
||
|
|
422.4757995605469,
|
||
|
|
424.78662109375,
|
||
|
|
422.1180725097656,
|
||
|
|
421.7055358886719,
|
||
|
|
424.3546447753906
|
||
|
|
],
|
||
|
|
"p50_ms": 422.6557159423828,
|
||
|
|
"p95_ms": 432.7436782836915,
|
||
|
|
"mean_ms": 428.26379852294923
|
||
|
|
},
|
||
|
|
"parity": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75",
|
||
|
|
"expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75"
|
||
|
|
}
|
||
|
|
},
|
||
|
|
"dispatch_delta": {
|
||
|
|
"attempts": 23,
|
||
|
|
"successes": 23,
|
||
|
|
"fallbacks": 0
|
||
|
|
}
|
||
|
|
},
|
||
|
|
{
|
||
|
|
"block": 49,
|
||
|
|
"production_method": "block.mlp.fc2.forward_swiglu",
|
||
|
|
"candidate_vs_baseline": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6",
|
||
|
|
"expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6"
|
||
|
|
},
|
||
|
|
"baseline_vs_traversal": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6",
|
||
|
|
"expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6"
|
||
|
|
},
|
||
|
|
"candidate_vs_traversal": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6",
|
||
|
|
"expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6"
|
||
|
|
},
|
||
|
|
"timing": {
|
||
|
|
"order": "AB/BA alternates by round",
|
||
|
|
"baseline": {
|
||
|
|
"samples_ms": [
|
||
|
|
457.59906005859375,
|
||
|
|
455.51739501953125,
|
||
|
|
454.50775146484375,
|
||
|
|
455.77972412109375,
|
||
|
|
454.094482421875,
|
||
|
|
453.72442626953125,
|
||
|
|
456.2005615234375,
|
||
|
|
457.22589111328125,
|
||
|
|
455.93505859375,
|
||
|
|
454.9208679199219,
|
||
|
|
454.7513732910156,
|
||
|
|
456.42205810546875,
|
||
|
|
456.1644592285156,
|
||
|
|
455.8465270996094,
|
||
|
|
459.75469970703125,
|
||
|
|
456.6866149902344,
|
||
|
|
456.8373107910156,
|
||
|
|
458.20733642578125,
|
||
|
|
456.2602233886719,
|
||
|
|
456.6114807128906
|
||
|
|
],
|
||
|
|
"p50_ms": 456.18251037597656,
|
||
|
|
"p95_ms": 458.2847045898437,
|
||
|
|
"mean_ms": 456.15236511230466
|
||
|
|
},
|
||
|
|
"candidate": {
|
||
|
|
"samples_ms": [
|
||
|
|
421.10394287109375,
|
||
|
|
427.0021057128906,
|
||
|
|
418.5732116699219,
|
||
|
|
420.71209716796875,
|
||
|
|
419.9263916015625,
|
||
|
|
420.857177734375,
|
||
|
|
417.8297424316406,
|
||
|
|
421.5742492675781,
|
||
|
|
421.7043762207031,
|
||
|
|
419.58203125,
|
||
|
|
419.05865478515625,
|
||
|
|
421.8150634765625,
|
||
|
|
421.2725524902344,
|
||
|
|
419.3066711425781,
|
||
|
|
419.87371826171875,
|
||
|
|
420.12255859375,
|
||
|
|
419.9741516113281,
|
||
|
|
420.2537841796875,
|
||
|
|
420.41552734375,
|
||
|
|
422.7670593261719
|
||
|
|
],
|
||
|
|
"p50_ms": 420.33465576171875,
|
||
|
|
"p95_ms": 422.9788116455078,
|
||
|
|
"mean_ms": 420.6862533569336
|
||
|
|
},
|
||
|
|
"parity": {
|
||
|
|
"bf16_exact": true,
|
||
|
|
"different_elements": 0,
|
||
|
|
"max_abs": 0.0,
|
||
|
|
"mean_abs": 0.0,
|
||
|
|
"actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6",
|
||
|
|
"expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6"
|
||
|
|
}
|
||
|
|
},
|
||
|
|
"dispatch_delta": {
|
||
|
|
"attempts": 23,
|
||
|
|
"successes": 23,
|
||
|
|
"fallbacks": 0
|
||
|
|
}
|
||
|
|
}
|
||
|
|
],
|
||
|
|
"errors_and_unsupported": [
|
||
|
|
{
|
||
|
|
"feature": "Stream-K",
|
||
|
|
"supported_public_control": false,
|
||
|
|
"reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control."
|
||
|
|
}
|
||
|
|
]
|
||
|
|
}
|