2133 lines
63 KiB
JSON
2133 lines
63 KiB
JSON
{
|
|
"mode": "sweep",
|
|
"environment": {
|
|
"platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
|
|
"python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]",
|
|
"torch": "2.9.1+cu130",
|
|
"torch_cuda": "13.0",
|
|
"device": "NVIDIA GB10",
|
|
"device_capability": [
|
|
12,
|
|
1
|
|
],
|
|
"driver": null,
|
|
"git_commit": null,
|
|
"checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors",
|
|
"checkpoint_sha256": null,
|
|
"checkpoint_hash_note": "not calculated",
|
|
"environment_switches": {
|
|
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
|
|
"CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4",
|
|
"CUDA_HOME": "/usr/local/cuda",
|
|
"CUDA_INC_PATH": "/usr/local/cuda/include",
|
|
"CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1",
|
|
"CUDA_MODULE_LOADING": "EAGER",
|
|
"CUDA_VERSION": "13.0.2",
|
|
"H3_FUSED_ELEMENTWISE": "1",
|
|
"H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors",
|
|
"H3_NVFP4_MODULATE_FUSION": "1",
|
|
"H3_NVFP4_SCALE_BACKEND": "vortex",
|
|
"H3_NVFP4_SCALE_VERSION": "1",
|
|
"H3_NVFP4_SWIGLU_FUSION": "1",
|
|
"H3_SAGE_QKV_LAYOUT": "strided_nhd",
|
|
"TORCH_COMPILE_DISABLE": "0",
|
|
"TORCH_CUDA_ARCH_LIST": "12.1a",
|
|
"TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions"
|
|
},
|
|
"extension": {
|
|
"cuda_version": 13000,
|
|
"cublas_version": 130100,
|
|
"stream_k_public_control": false,
|
|
"stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control."
|
|
},
|
|
"comfy_kitchen": "0.2.31 package without __version__"
|
|
},
|
|
"workload": {
|
|
"resolution": [
|
|
1344,
|
|
768
|
|
],
|
|
"frames": 124,
|
|
"steps": 12,
|
|
"sampler_step": 1,
|
|
"seed": 440420,
|
|
"text_tokens": 100,
|
|
"tokens": 37810,
|
|
"hidden_shape": [
|
|
37810,
|
|
5376
|
|
],
|
|
"segments": [
|
|
[
|
|
0,
|
|
100,
|
|
1
|
|
],
|
|
[
|
|
100,
|
|
514,
|
|
2
|
|
],
|
|
[
|
|
514,
|
|
37810,
|
|
0
|
|
]
|
|
]
|
|
},
|
|
"retained_blocks": [
|
|
0,
|
|
24,
|
|
49
|
|
],
|
|
"immutable_cloned_block_inputs": {
|
|
"0": [
|
|
37810,
|
|
5376
|
|
],
|
|
"24": [
|
|
37810,
|
|
5376
|
|
],
|
|
"49": [
|
|
37810,
|
|
5376
|
|
]
|
|
},
|
|
"fc2_boundary": {
|
|
"block": 24,
|
|
"gate_up_shape": [
|
|
37810,
|
|
28672
|
|
],
|
|
"activation_qdata_shape": [
|
|
37824,
|
|
7168
|
|
],
|
|
"weight_qdata_shape": [
|
|
5376,
|
|
7168
|
|
],
|
|
"logical_mnk": [
|
|
37810,
|
|
5376,
|
|
14336
|
|
],
|
|
"descriptor_mnk_after_padding": [
|
|
37824,
|
|
5376,
|
|
14336
|
|
],
|
|
"producer": "vortex_native_quantize_swiglu_nvfp4",
|
|
"no_bias": true
|
|
},
|
|
"baseline_kernel_metadata": {
|
|
"path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4",
|
|
"descriptors": {
|
|
"packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]",
|
|
"block_scale_mode": "VEC16_UE4M3",
|
|
"compute_and_scale": "FP32",
|
|
"scalar_pointer_mode": "device",
|
|
"bias": null,
|
|
"beta": 0.0,
|
|
"comfy_kitchen_version": "0.2.31"
|
|
},
|
|
"profiler_cuda_events_available": false,
|
|
"profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.",
|
|
"top_cuda_events": [],
|
|
"output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
},
|
|
"heuristics": [
|
|
{
|
|
"max_workspace_bytes": 0,
|
|
"requested_count": 64,
|
|
"returned_count": 5,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 4194304,
|
|
"requested_count": 64,
|
|
"returned_count": 7,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 8388608,
|
|
"requested_count": 64,
|
|
"returned_count": 7,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 16777216,
|
|
"requested_count": 64,
|
|
"returned_count": 6,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 33554432,
|
|
"requested_count": 64,
|
|
"returned_count": 6,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 67108864,
|
|
"requested_count": 64,
|
|
"returned_count": 6,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"explicit_split_k_checks": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 2,
|
|
"reduction_scheme": 2
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 2,
|
|
"reduction_scheme": 4,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 2,
|
|
"reduction_scheme": 4
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 4,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 4,
|
|
"reduction_scheme": 2
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 4,
|
|
"reduction_scheme": 4,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 4,
|
|
"reduction_scheme": 4
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 8,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 8,
|
|
"reduction_scheme": 2
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 8,
|
|
"reduction_scheme": 4,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 8,
|
|
"reduction_scheme": 4
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 16,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 16,
|
|
"reduction_scheme": 2
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 16,
|
|
"reduction_scheme": 4,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 16,
|
|
"reduction_scheme": 4
|
|
}
|
|
}
|
|
],
|
|
"selected": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0
|
|
},
|
|
"candidates": [
|
|
{
|
|
"selected_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 67108864,
|
|
"required_workspace_bytes": 0,
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"fc2_only": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
52.528385162353516,
|
|
53.54975891113281,
|
|
52.53424072265625,
|
|
52.453086853027344,
|
|
52.415489196777344,
|
|
53.50998306274414,
|
|
52.42780685424805,
|
|
52.526817321777344,
|
|
52.41756820678711,
|
|
52.83504104614258,
|
|
54.08259201049805,
|
|
52.47983932495117,
|
|
52.43289566040039,
|
|
52.59030532836914,
|
|
52.4370231628418,
|
|
52.52783966064453,
|
|
52.4318733215332,
|
|
52.5145263671875,
|
|
52.424705505371094,
|
|
52.57904052734375
|
|
],
|
|
"p50_ms": 52.52067184448242,
|
|
"p95_ms": 53.57640056610108,
|
|
"mean_ms": 52.68494091033936,
|
|
"dense_tflop_s_p50": 110.9669507956279
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
52.557823181152344,
|
|
53.029598236083984,
|
|
52.549888610839844,
|
|
52.55766296386719,
|
|
52.51689529418945,
|
|
53.11779022216797,
|
|
54.02726364135742,
|
|
52.74710464477539,
|
|
52.530174255371094,
|
|
52.540096282958984,
|
|
52.56089782714844,
|
|
52.54924774169922,
|
|
52.527103424072266,
|
|
52.53104019165039,
|
|
53.064640045166016,
|
|
53.72809600830078,
|
|
53.41603088378906,
|
|
52.76732635498047,
|
|
52.54451370239258,
|
|
53.056480407714844
|
|
],
|
|
"p50_ms": 52.55936050415039,
|
|
"p95_ms": 53.74305438995361,
|
|
"mean_ms": 52.84598369598389,
|
|
"dense_tflop_s_p50": 110.88526862612385
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f",
|
|
"expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
}
|
|
},
|
|
"accepted_producer_plus_fc2": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
75.45340728759766,
|
|
75.004638671875,
|
|
76.25904083251953,
|
|
75.5995864868164,
|
|
75.02406311035156,
|
|
75.02108764648438,
|
|
74.97523498535156,
|
|
76.13423919677734,
|
|
76.12108612060547,
|
|
76.02950286865234,
|
|
76.11692810058594,
|
|
75.5208969116211,
|
|
75.1124496459961,
|
|
75.58025360107422,
|
|
74.99574279785156,
|
|
76.13922882080078,
|
|
76.13350677490234,
|
|
76.16307067871094,
|
|
76.70272064208984,
|
|
75.0445785522461
|
|
],
|
|
"p50_ms": 75.58992004394531,
|
|
"p95_ms": 76.28122482299804,
|
|
"mean_ms": 75.65656318664551,
|
|
"dense_tflop_s_p50": 77.10100506696888
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
75.16223907470703,
|
|
75.09487915039062,
|
|
76.4610595703125,
|
|
75.77107238769531,
|
|
76.33715057373047,
|
|
76.33900451660156,
|
|
75.62342071533203,
|
|
76.34333038330078,
|
|
75.10733032226562,
|
|
75.13775634765625,
|
|
76.2429428100586,
|
|
75.65090942382812,
|
|
76.9865951538086,
|
|
76.45164489746094,
|
|
75.12268829345703,
|
|
76.36873626708984,
|
|
75.21382141113281,
|
|
75.16233825683594,
|
|
75.15340423583984,
|
|
75.63337707519531
|
|
],
|
|
"p50_ms": 75.64214324951172,
|
|
"p95_ms": 76.48733634948731,
|
|
"mean_ms": 75.76818504333497,
|
|
"dense_tflop_s_p50": 77.04777466571349
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f",
|
|
"expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
}
|
|
},
|
|
"output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
},
|
|
{
|
|
"selected_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 67108864,
|
|
"required_workspace_bytes": 0,
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"fc2_only": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
53.98486328125,
|
|
53.7977294921875,
|
|
53.85007858276367,
|
|
53.532447814941406,
|
|
53.588897705078125,
|
|
53.95724868774414,
|
|
53.63395309448242,
|
|
53.011390686035156,
|
|
53.56748962402344,
|
|
54.085472106933594,
|
|
54.20851135253906,
|
|
53.808990478515625,
|
|
53.888832092285156,
|
|
54.48992156982422,
|
|
53.129215240478516,
|
|
53.52640151977539,
|
|
53.18348693847656,
|
|
53.60192108154297,
|
|
53.001953125,
|
|
53.25020980834961
|
|
],
|
|
"p50_ms": 53.617937088012695,
|
|
"p95_ms": 54.22258186340332,
|
|
"mean_ms": 53.65495071411133,
|
|
"dense_tflop_s_p50": 108.69606562358724
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
15.689727783203125,
|
|
15.966943740844727,
|
|
15.744000434875488,
|
|
15.626208305358887,
|
|
15.685664176940918,
|
|
15.68841552734375,
|
|
15.331328392028809,
|
|
15.71504020690918,
|
|
14.792703628540039,
|
|
15.66592025756836,
|
|
14.791680335998535,
|
|
15.645536422729492,
|
|
14.812159538269043,
|
|
14.813887596130371,
|
|
14.793631553649902,
|
|
15.76848030090332,
|
|
15.557696342468262,
|
|
15.463264465332031,
|
|
14.7957763671875,
|
|
15.723199844360352
|
|
],
|
|
"p50_ms": 15.63587236404419,
|
|
"p95_ms": 15.778403472900392,
|
|
"mean_ms": 15.403563261032104,
|
|
"dense_tflop_s_p50": 372.73640207770177
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f",
|
|
"expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
}
|
|
},
|
|
"accepted_producer_plus_fc2": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
76.28396606445312,
|
|
76.32662200927734,
|
|
76.3351058959961,
|
|
76.94000244140625,
|
|
76.97782135009766,
|
|
76.59494018554688,
|
|
76.87884521484375,
|
|
77.03215789794922,
|
|
76.31446075439453,
|
|
76.24060821533203,
|
|
76.16102600097656,
|
|
76.39116668701172,
|
|
77.31603240966797,
|
|
76.89494323730469,
|
|
75.6058578491211,
|
|
76.90121459960938,
|
|
76.15897369384766,
|
|
77.05766296386719,
|
|
77.53421020507812,
|
|
76.94409942626953
|
|
],
|
|
"p50_ms": 76.73689270019531,
|
|
"p95_ms": 77.32694129943847,
|
|
"mean_ms": 76.64448585510254,
|
|
"dense_tflop_s_p50": 75.94859008807853
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
38.138526916503906,
|
|
39.03964614868164,
|
|
39.00892639160156,
|
|
38.78895950317383,
|
|
38.25788879394531,
|
|
39.111358642578125,
|
|
37.374977111816406,
|
|
39.55583953857422,
|
|
38.676639556884766,
|
|
39.315616607666016,
|
|
39.011199951171875,
|
|
38.85724639892578,
|
|
38.27609634399414,
|
|
39.127777099609375,
|
|
38.3109130859375,
|
|
39.1360969543457,
|
|
38.23001480102539,
|
|
39.10319900512695,
|
|
38.683841705322266,
|
|
38.4785270690918
|
|
],
|
|
"p50_ms": 38.823102951049805,
|
|
"p95_ms": 39.32762775421143,
|
|
"mean_ms": 38.72416458129883,
|
|
"dense_tflop_s_p50": 150.11831526368
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f",
|
|
"expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
}
|
|
},
|
|
"output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
}
|
|
],
|
|
"errors_and_unsupported": [
|
|
{
|
|
"feature": "Stream-K",
|
|
"supported_public_control": false,
|
|
"reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control."
|
|
}
|
|
]
|
|
}
|