2404 lines
72 KiB
JSON
2404 lines
72 KiB
JSON
{
|
|
"mode": "shape-gate",
|
|
"environment": {
|
|
"platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
|
|
"python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]",
|
|
"torch": "2.9.1+cu130",
|
|
"torch_cuda": "13.0",
|
|
"device": "NVIDIA GB10",
|
|
"device_capability": [
|
|
12,
|
|
1
|
|
],
|
|
"driver": null,
|
|
"git_commit": null,
|
|
"checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors",
|
|
"checkpoint_sha256": null,
|
|
"checkpoint_hash_note": "not calculated",
|
|
"environment_switches": {
|
|
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
|
|
"CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4",
|
|
"CUDA_HOME": "/usr/local/cuda",
|
|
"CUDA_INC_PATH": "/usr/local/cuda/include",
|
|
"CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1",
|
|
"CUDA_MODULE_LOADING": "EAGER",
|
|
"CUDA_VERSION": "13.0.2",
|
|
"H3_FUSED_ELEMENTWISE": "1",
|
|
"H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors",
|
|
"H3_NVFP4_MODULATE_FUSION": "1",
|
|
"H3_NVFP4_SCALE_BACKEND": "vortex",
|
|
"H3_NVFP4_SCALE_VERSION": "1",
|
|
"H3_NVFP4_SWIGLU_FUSION": "1",
|
|
"H3_SAGE_QKV_LAYOUT": "strided_nhd",
|
|
"TORCH_COMPILE_DISABLE": "0",
|
|
"TORCH_CUDA_ARCH_LIST": "12.1a",
|
|
"TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions"
|
|
},
|
|
"extension": {
|
|
"cuda_version": 13000,
|
|
"cublas_version": 130100,
|
|
"stream_k_public_control": false,
|
|
"stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control."
|
|
},
|
|
"comfy_kitchen": "0.2.31 package without __version__"
|
|
},
|
|
"workload": {
|
|
"resolution": [
|
|
1344,
|
|
768
|
|
],
|
|
"frames": 124,
|
|
"steps": 12,
|
|
"sampler_step": 1,
|
|
"seed": 440420,
|
|
"text_tokens": 100,
|
|
"tokens": 37810,
|
|
"hidden_shape": [
|
|
37810,
|
|
5376
|
|
],
|
|
"segments": [
|
|
[
|
|
0,
|
|
100,
|
|
1
|
|
],
|
|
[
|
|
100,
|
|
514,
|
|
2
|
|
],
|
|
[
|
|
514,
|
|
37810,
|
|
0
|
|
]
|
|
]
|
|
},
|
|
"retained_blocks": [
|
|
0,
|
|
24,
|
|
49
|
|
],
|
|
"immutable_cloned_block_inputs": {
|
|
"0": [
|
|
37810,
|
|
5376
|
|
],
|
|
"24": [
|
|
37810,
|
|
5376
|
|
],
|
|
"49": [
|
|
37810,
|
|
5376
|
|
]
|
|
},
|
|
"fc2_boundary": {
|
|
"block": 24,
|
|
"gate_up_shape": [
|
|
37810,
|
|
28672
|
|
],
|
|
"activation_qdata_shape": [
|
|
37824,
|
|
7168
|
|
],
|
|
"weight_qdata_shape": [
|
|
5376,
|
|
7168
|
|
],
|
|
"logical_mnk": [
|
|
37810,
|
|
5376,
|
|
14336
|
|
],
|
|
"descriptor_mnk_after_padding": [
|
|
37824,
|
|
5376,
|
|
14336
|
|
],
|
|
"producer": "vortex_native_quantize_swiglu_nvfp4",
|
|
"no_bias": true
|
|
},
|
|
"baseline_kernel_metadata": {
|
|
"path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4",
|
|
"descriptors": {
|
|
"packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]",
|
|
"block_scale_mode": "VEC16_UE4M3",
|
|
"compute_and_scale": "FP32",
|
|
"scalar_pointer_mode": "device",
|
|
"bias": null,
|
|
"beta": 0.0,
|
|
"comfy_kitchen_version": "0.2.31"
|
|
},
|
|
"profiler_cuda_events_available": false,
|
|
"profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.",
|
|
"top_cuda_events": [],
|
|
"output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
},
|
|
"heuristics": [
|
|
{
|
|
"max_workspace_bytes": 0,
|
|
"requested_count": 32,
|
|
"returned_count": 5,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 4194304,
|
|
"requested_count": 32,
|
|
"returned_count": 7,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 8388608,
|
|
"requested_count": 32,
|
|
"returned_count": 7,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 16777216,
|
|
"requested_count": 32,
|
|
"returned_count": 6,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 33554432,
|
|
"requested_count": 32,
|
|
"returned_count": 6,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"max_workspace_bytes": 67108864,
|
|
"requested_count": 32,
|
|
"returned_count": 6,
|
|
"algorithms": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": -2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
}
|
|
]
|
|
}
|
|
],
|
|
"explicit_split_k_checks": [
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 2,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 2,
|
|
"reduction_scheme": 2
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 2,
|
|
"reduction_scheme": 4,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 2,
|
|
"reduction_scheme": 4
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 4,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 4,
|
|
"reduction_scheme": 2
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 4,
|
|
"reduction_scheme": 4,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 4,
|
|
"reduction_scheme": 4
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 8,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 8,
|
|
"reduction_scheme": 2
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 8,
|
|
"reduction_scheme": 4,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 8,
|
|
"reduction_scheme": 4
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 16,
|
|
"reduction_scheme": 2,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 16,
|
|
"reduction_scheme": 2
|
|
}
|
|
},
|
|
{
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 16,
|
|
"reduction_scheme": 4,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 0.0,
|
|
"state": 15,
|
|
"api_status": 15,
|
|
"valid": false,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
},
|
|
"requested_config": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"split_k": 16,
|
|
"reduction_scheme": 4
|
|
}
|
|
}
|
|
],
|
|
"selected": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0
|
|
},
|
|
"shape_gate": [
|
|
{
|
|
"text_tokens": 1,
|
|
"logical_rows": 37711,
|
|
"packed_rows": 37712,
|
|
"parity": {
|
|
"bf16_exact": false,
|
|
"different_elements": 2,
|
|
"max_abs": 0.03125,
|
|
"mean_abs": 1.553468603754382e-10,
|
|
"actual_sha256": "c1ee129ef17b93c7814dc0dfb4dfc9fc0218a3d4252cddb02c80eb98e0471ab5",
|
|
"expected_sha256": "e017519cd7e89a93770f7648e81c637ccac517fe2715dd213f46627dfdfed711"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
57.960289001464844
|
|
],
|
|
"p50_ms": 57.960289001464844,
|
|
"p95_ms": 57.960289001464844,
|
|
"mean_ms": 57.960289001464844,
|
|
"dense_tflop_s_p50": 100.28933571475277
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
18.854400634765625
|
|
],
|
|
"p50_ms": 18.854400634765625,
|
|
"p95_ms": 18.854400634765625,
|
|
"mean_ms": 18.854400634765625,
|
|
"dense_tflop_s_p50": 308.2993193150771
|
|
},
|
|
"parity": {
|
|
"bf16_exact": false,
|
|
"different_elements": 2,
|
|
"max_abs": 0.03125,
|
|
"mean_abs": 1.553468603754382e-10,
|
|
"actual_sha256": "c1ee129ef17b93c7814dc0dfb4dfc9fc0218a3d4252cddb02c80eb98e0471ab5",
|
|
"expected_sha256": "e017519cd7e89a93770f7648e81c637ccac517fe2715dd213f46627dfdfed711"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": false
|
|
},
|
|
{
|
|
"text_tokens": 15,
|
|
"logical_rows": 37725,
|
|
"packed_rows": 37728,
|
|
"parity": {
|
|
"bf16_exact": false,
|
|
"different_elements": 2,
|
|
"max_abs": 0.03125,
|
|
"mean_abs": 1.5528919816709674e-10,
|
|
"actual_sha256": "c47d817c29526fcf16d800bba07d8ac659831fe16076c07b5b5ba5347ed63c7e",
|
|
"expected_sha256": "310f6ecf71f385122512a2289282a9a8723e231aa252223201e23e9857219eda"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
57.9911994934082
|
|
],
|
|
"p50_ms": 57.9911994934082,
|
|
"p95_ms": 57.9911994934082,
|
|
"mean_ms": 57.9911994934082,
|
|
"dense_tflop_s_p50": 100.27309146900781
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
18.6856632232666
|
|
],
|
|
"p50_ms": 18.6856632232666,
|
|
"p95_ms": 18.6856632232666,
|
|
"mean_ms": 18.6856632232666,
|
|
"dense_tflop_s_p50": 311.1988470368801
|
|
},
|
|
"parity": {
|
|
"bf16_exact": false,
|
|
"different_elements": 2,
|
|
"max_abs": 0.03125,
|
|
"mean_abs": 1.5528919816709674e-10,
|
|
"actual_sha256": "c47d817c29526fcf16d800bba07d8ac659831fe16076c07b5b5ba5347ed63c7e",
|
|
"expected_sha256": "310f6ecf71f385122512a2289282a9a8723e231aa252223201e23e9857219eda"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": false
|
|
},
|
|
{
|
|
"text_tokens": 32,
|
|
"logical_rows": 37742,
|
|
"packed_rows": 37744,
|
|
"parity": {
|
|
"bf16_exact": false,
|
|
"different_elements": 2,
|
|
"max_abs": 0.03125,
|
|
"mean_abs": 1.5521925411654536e-10,
|
|
"actual_sha256": "c50b1f08167249b90c80632f4bf1c379b252446cd766ac1ba331087c2de2744d",
|
|
"expected_sha256": "42816a3c87083bd5ed98a975cd6c7b31f35746761089fa8f2d19c92f47fed3df"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
57.83795166015625
|
|
],
|
|
"p50_ms": 57.83795166015625,
|
|
"p95_ms": 57.83795166015625,
|
|
"mean_ms": 57.83795166015625,
|
|
"dense_tflop_s_p50": 100.58408148350189
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
18.732128143310547
|
|
],
|
|
"p50_ms": 18.732128143310547,
|
|
"p95_ms": 18.732128143310547,
|
|
"mean_ms": 18.732128143310547,
|
|
"dense_tflop_s_p50": 310.56680789905454
|
|
},
|
|
"parity": {
|
|
"bf16_exact": false,
|
|
"different_elements": 2,
|
|
"max_abs": 0.03125,
|
|
"mean_abs": 1.5521925411654536e-10,
|
|
"actual_sha256": "c50b1f08167249b90c80632f4bf1c379b252446cd766ac1ba331087c2de2744d",
|
|
"expected_sha256": "42816a3c87083bd5ed98a975cd6c7b31f35746761089fa8f2d19c92f47fed3df"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": false
|
|
},
|
|
{
|
|
"text_tokens": 64,
|
|
"logical_rows": 37774,
|
|
"packed_rows": 37776,
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "9b272d355084277500dc89e4d633d44c7daa81481b092d3b96f577e2c207ddac",
|
|
"expected_sha256": "9b272d355084277500dc89e4d633d44c7daa81481b092d3b96f577e2c207ddac"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
58.0667839050293
|
|
],
|
|
"p50_ms": 58.0667839050293,
|
|
"p95_ms": 58.0667839050293,
|
|
"mean_ms": 58.0667839050293,
|
|
"dense_tflop_s_p50": 100.27264044192569
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
19.07366371154785
|
|
],
|
|
"p50_ms": 19.07366371154785,
|
|
"p95_ms": 19.07366371154785,
|
|
"mean_ms": 19.07366371154785,
|
|
"dense_tflop_s_p50": 305.26435991439087
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "9b272d355084277500dc89e4d633d44c7daa81481b092d3b96f577e2c207ddac",
|
|
"expected_sha256": "9b272d355084277500dc89e4d633d44c7daa81481b092d3b96f577e2c207ddac"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": true
|
|
},
|
|
{
|
|
"text_tokens": 99,
|
|
"logical_rows": 37809,
|
|
"packed_rows": 37824,
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "dcc4ac6d680a76bf563145e569ea114364a0ff3e76f3f2f656d1dd70adfba812",
|
|
"expected_sha256": "dcc4ac6d680a76bf563145e569ea114364a0ff3e76f3f2f656d1dd70adfba812"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
57.88643264770508
|
|
],
|
|
"p50_ms": 57.88643264770508,
|
|
"p95_ms": 57.88643264770508,
|
|
"mean_ms": 57.88643264770508,
|
|
"dense_tflop_s_p50": 100.6782487895987
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
18.773759841918945
|
|
],
|
|
"p50_ms": 18.773759841918945,
|
|
"p95_ms": 18.773759841918945,
|
|
"mean_ms": 18.773759841918945,
|
|
"dense_tflop_s_p50": 310.4282102637308
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "dcc4ac6d680a76bf563145e569ea114364a0ff3e76f3f2f656d1dd70adfba812",
|
|
"expected_sha256": "dcc4ac6d680a76bf563145e569ea114364a0ff3e76f3f2f656d1dd70adfba812"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": true
|
|
},
|
|
{
|
|
"text_tokens": 100,
|
|
"logical_rows": 37810,
|
|
"packed_rows": 37824,
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f",
|
|
"expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
57.96451187133789
|
|
],
|
|
"p50_ms": 57.96451187133789,
|
|
"p95_ms": 57.96451187133789,
|
|
"mean_ms": 57.96451187133789,
|
|
"dense_tflop_s_p50": 100.54529263105621
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
18.34467124938965
|
|
],
|
|
"p50_ms": 18.34467124938965,
|
|
"p95_ms": 18.34467124938965,
|
|
"mean_ms": 18.34467124938965,
|
|
"dense_tflop_s_p50": 317.6976424973496
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f",
|
|
"expected_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": true
|
|
},
|
|
{
|
|
"text_tokens": 128,
|
|
"logical_rows": 37838,
|
|
"packed_rows": 37840,
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "adb567fdd1568c406d514983f0dbf799b3d9121656788a853fe8620bca7cb67b",
|
|
"expected_sha256": "adb567fdd1568c406d514983f0dbf799b3d9121656788a853fe8620bca7cb67b"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
57.80854415893555
|
|
],
|
|
"p50_ms": 57.80854415893555,
|
|
"p95_ms": 57.80854415893555,
|
|
"mean_ms": 57.80854415893555,
|
|
"dense_tflop_s_p50": 100.89122346864156
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
19.78188705444336
|
|
],
|
|
"p50_ms": 19.78188705444336,
|
|
"p95_ms": 19.78188705444336,
|
|
"mean_ms": 19.78188705444336,
|
|
"dense_tflop_s_p50": 294.8340939913488
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "adb567fdd1568c406d514983f0dbf799b3d9121656788a853fe8620bca7cb67b",
|
|
"expected_sha256": "adb567fdd1568c406d514983f0dbf799b3d9121656788a853fe8620bca7cb67b"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": true
|
|
},
|
|
{
|
|
"text_tokens": 256,
|
|
"logical_rows": 37966,
|
|
"packed_rows": 37968,
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "e264df4d576726d22bf75e6e33a7bc3880543af28a6a528a8332b6fe25aa7676",
|
|
"expected_sha256": "e264df4d576726d22bf75e6e33a7bc3880543af28a6a528a8332b6fe25aa7676"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
58.237281799316406
|
|
],
|
|
"p50_ms": 58.237281799316406,
|
|
"p95_ms": 58.237281799316406,
|
|
"mean_ms": 58.237281799316406,
|
|
"dense_tflop_s_p50": 100.4872578586024
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
18.483808517456055
|
|
],
|
|
"p50_ms": 18.483808517456055,
|
|
"p95_ms": 18.483808517456055,
|
|
"mean_ms": 18.483808517456055,
|
|
"dense_tflop_s_p50": 316.60708601397215
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "e264df4d576726d22bf75e6e33a7bc3880543af28a6a528a8332b6fe25aa7676",
|
|
"expected_sha256": "e264df4d576726d22bf75e6e33a7bc3880543af28a6a528a8332b6fe25aa7676"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": true
|
|
},
|
|
{
|
|
"text_tokens": 512,
|
|
"logical_rows": 38222,
|
|
"packed_rows": 38224,
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "b6276cda42be6a858bab99ae69a211c71400104f39bc28b2d6723ad37be015b9",
|
|
"expected_sha256": "b6276cda42be6a858bab99ae69a211c71400104f39bc28b2d6723ad37be015b9"
|
|
},
|
|
"timing": {
|
|
"order": "AB/BA alternates by round",
|
|
"baseline": {
|
|
"samples_ms": [
|
|
58.96633529663086
|
|
],
|
|
"p50_ms": 58.96633529663086,
|
|
"p95_ms": 58.96633529663086,
|
|
"mean_ms": 58.96633529663086,
|
|
"dense_tflop_s_p50": 99.91403968970451
|
|
},
|
|
"candidate": {
|
|
"samples_ms": [
|
|
20.196256637573242
|
|
],
|
|
"p50_ms": 20.196256637573242,
|
|
"p95_ms": 20.196256637573242,
|
|
"mean_ms": 20.196256637573242,
|
|
"dense_tflop_s_p50": 291.71568132201764
|
|
},
|
|
"parity": {
|
|
"bf16_exact": true,
|
|
"different_elements": 0,
|
|
"max_abs": 0.0,
|
|
"mean_abs": 0.0,
|
|
"actual_sha256": "b6276cda42be6a858bab99ae69a211c71400104f39bc28b2d6723ad37be015b9",
|
|
"expected_sha256": "b6276cda42be6a858bab99ae69a211c71400104f39bc28b2d6723ad37be015b9"
|
|
}
|
|
},
|
|
"checked": {
|
|
"algorithm_id": 70,
|
|
"tile_id": 20,
|
|
"stages_id": 37,
|
|
"split_k": 1,
|
|
"reduction_scheme": 0,
|
|
"custom_option": 0,
|
|
"cta_swizzle": 0,
|
|
"inner_shape": null,
|
|
"cluster_shape": null,
|
|
"required_workspace_bytes": 0,
|
|
"waves": 1.0,
|
|
"state": 0,
|
|
"api_status": 0,
|
|
"valid": true,
|
|
"capabilities": {
|
|
"split_k_support": 1,
|
|
"reduction_scheme_mask": 6,
|
|
"cta_swizzle_support": 0,
|
|
"custom_option_max": 0,
|
|
"strided_batch_support": 1,
|
|
"out_of_place_result_support": 1,
|
|
"tile_ids": [
|
|
20
|
|
],
|
|
"stages_ids": [
|
|
37
|
|
],
|
|
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
|
|
}
|
|
},
|
|
"supplied_workspace_bytes": 0,
|
|
"passed": true
|
|
}
|
|
],
|
|
"errors_and_unsupported": [
|
|
{
|
|
"feature": "Stream-K",
|
|
"supported_public_control": false,
|
|
"reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control."
|
|
}
|
|
]
|
|
}
|