h3-blackwell-runtime/benchmarks/gb10-fc2-nvfp4-block-gate-20260825.json
2026-08-25 22:32:48 +07:00

2014 lines
60 KiB
JSON

{
"mode": "block-gate",
"environment": {
"platform": "Linux-6.17.0-1026-nvidia-aarch64-with-glibc2.39",
"python": "3.12.3 (main, Mar 23 2026, 19:04:32) [GCC 13.3.0]",
"torch": "2.9.1+cu130",
"torch_cuda": "13.0",
"device": "NVIDIA GB10",
"device_capability": [
12,
1
],
"driver": null,
"git_commit": null,
"checkpoint_path": "/models/minimax_h3_fl2va_pruned_nvfp4.safetensors",
"checkpoint_sha256": null,
"checkpoint_hash_note": "not calculated",
"environment_switches": {
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
"CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4",
"CUDA_HOME": "/usr/local/cuda",
"CUDA_INC_PATH": "/usr/local/cuda/include",
"CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1",
"CUDA_MODULE_LOADING": "EAGER",
"CUDA_VERSION": "13.0.2",
"H3_FUSED_ELEMENTWISE": "1",
"H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors",
"H3_NVFP4_MODULATE_FUSION": "1",
"H3_NVFP4_SCALE_BACKEND": "vortex",
"H3_NVFP4_SCALE_VERSION": "1",
"H3_NVFP4_SWIGLU_FUSION": "1",
"H3_SAGE_QKV_LAYOUT": "strided_nhd",
"TORCH_COMPILE_DISABLE": "0",
"TORCH_CUDA_ARCH_LIST": "12.1a",
"TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions"
},
"extension": {
"cuda_version": 13000,
"cublas_version": 130100,
"stream_k_public_control": false,
"stream_k_note": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control."
},
"comfy_kitchen": "0.2.31 package without __version__"
},
"workload": {
"resolution": [
1344,
768
],
"frames": 124,
"steps": 12,
"sampler_step": 1,
"seed": 440420,
"text_tokens": 100,
"tokens": 37810,
"hidden_shape": [
37810,
5376
],
"segments": [
[
0,
100,
1
],
[
100,
514,
2
],
[
514,
37810,
0
]
]
},
"retained_blocks": [
0,
24,
49
],
"immutable_cloned_block_inputs": {
"0": [
37810,
5376
],
"24": [
37810,
5376
],
"49": [
37810,
5376
]
},
"fc2_boundary": {
"block": 24,
"gate_up_shape": [
37810,
28672
],
"activation_qdata_shape": [
37824,
7168
],
"weight_qdata_shape": [
5376,
7168
],
"logical_mnk": [
37810,
5376,
14336
],
"descriptor_mnk_after_padding": [
37824,
5376,
14336
],
"producer": "vortex_native_quantize_swiglu_nvfp4",
"no_bias": true
},
"baseline_kernel_metadata": {
"path": "fc2.forward_swiglu -> accepted producer -> Comfy Kitchen 0.2.31 scaled_mm_nvfp4",
"descriptors": {
"packed_input_output": "row-major [M,K] @ [N,K].T -> BF16 [M,N]",
"block_scale_mode": "VEC16_UE4M3",
"compute_and_scale": "FP32",
"scalar_pointer_mode": "device",
"bias": null,
"beta": 0.0,
"comfy_kitchen_version": "0.2.31"
},
"profiler_cuda_events_available": false,
"profiler_note": "Torch profiler returned no CUDA kernel events on this build; use --mode profile with NCU for kernel metadata.",
"top_cuda_events": [],
"output_sha256": "ad26c744afbb0249bff9afea503984e51217f85a59434e44fde347ae7d9b852f"
},
"heuristics": [
{
"max_workspace_bytes": 0,
"requested_count": 32,
"returned_count": 5,
"algorithms": [
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
}
]
},
{
"max_workspace_bytes": 4194304,
"requested_count": 32,
"returned_count": 7,
"algorithms": [
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
}
]
},
{
"max_workspace_bytes": 8388608,
"requested_count": 32,
"returned_count": 7,
"algorithms": [
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
}
]
},
{
"max_workspace_bytes": 16777216,
"requested_count": 32,
"returned_count": 6,
"algorithms": [
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
}
]
},
{
"max_workspace_bytes": 33554432,
"requested_count": 32,
"returned_count": 6,
"algorithms": [
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
}
]
},
{
"max_workspace_bytes": 67108864,
"requested_count": 32,
"returned_count": 6,
"algorithms": [
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": -2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
}
}
]
}
],
"explicit_split_k_checks": [
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 1.0,
"state": 0,
"api_status": 0,
"valid": true,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 1,
"reduction_scheme": 0
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 2,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 0.0,
"state": 15,
"api_status": 15,
"valid": false,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 2,
"reduction_scheme": 2
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 2,
"reduction_scheme": 4,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 0.0,
"state": 15,
"api_status": 15,
"valid": false,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 2,
"reduction_scheme": 4
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 4,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 0.0,
"state": 15,
"api_status": 15,
"valid": false,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 4,
"reduction_scheme": 2
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 4,
"reduction_scheme": 4,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 0.0,
"state": 15,
"api_status": 15,
"valid": false,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 4,
"reduction_scheme": 4
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 8,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 0.0,
"state": 15,
"api_status": 15,
"valid": false,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 8,
"reduction_scheme": 2
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 8,
"reduction_scheme": 4,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 0.0,
"state": 15,
"api_status": 15,
"valid": false,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 8,
"reduction_scheme": 4
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 16,
"reduction_scheme": 2,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 0.0,
"state": 15,
"api_status": 15,
"valid": false,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 16,
"reduction_scheme": 2
}
},
{
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 16,
"reduction_scheme": 4,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"required_workspace_bytes": 0,
"waves": 0.0,
"state": 15,
"api_status": 15,
"valid": false,
"capabilities": {
"split_k_support": 1,
"reduction_scheme_mask": 6,
"cta_swizzle_support": 0,
"custom_option_max": 0,
"strided_batch_support": 1,
"out_of_place_result_support": 1,
"tile_ids": [
20
],
"stages_ids": [
37
],
"inner_cluster_shape_capability_note": "This CUDA 13 cublasLt.h exposes config IDs but no public capability attributes that enumerate inner/cluster shape IDs."
},
"requested_config": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"custom_option": 0,
"cta_swizzle": 0,
"inner_shape": null,
"cluster_shape": null,
"split_k": 16,
"reduction_scheme": 4
}
}
],
"selected": {
"algorithm_id": 70,
"tile_id": 20,
"stages_id": 37,
"split_k": 1,
"reduction_scheme": 0,
"custom_option": 0,
"cta_swizzle": 0
},
"block_gate": [
{
"block": 0,
"only_monkeypatched_method": "block.mlp.fc2.forward_swiglu",
"accepted_gate_and_residual_path_preserved": true,
"candidate_vs_baseline": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147",
"expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147"
},
"baseline_vs_traversal": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147",
"expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147"
},
"candidate_vs_traversal": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147",
"expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147"
},
"timing": {
"order": "AB/BA alternates by round",
"baseline": {
"samples_ms": [
457.58404541015625,
455.8675537109375,
455.042724609375,
456.5163879394531,
456.7074279785156,
459.7113037109375,
456.3467102050781,
459.0717468261719,
457.6736755371094,
458.6617736816406,
458.61083984375,
456.4378662109375,
457.4807434082031,
458.178466796875,
457.4196472167969,
458.90869140625,
458.4776306152344,
458.2261962890625,
458.10919189453125,
460.418701171875
],
"p50_ms": 457.8914337158203,
"p95_ms": 459.7466735839844,
"mean_ms": 457.7725662231445
},
"candidate": {
"samples_ms": [
428.5652160644531,
425.1132507324219,
420.88983154296875,
420.71435546875,
419.1617736816406,
421.9552307128906,
419.37896728515625,
419.3441467285156,
419.79351806640625,
419.0124206542969,
420.5487976074219,
420.9779052734375,
420.5229187011719,
420.88104248046875,
420.3728942871094,
420.11834716796875,
420.3209533691406,
423.73919677734375,
419.7261657714844,
423.4720458984375
],
"p50_ms": 420.5358581542969,
"p95_ms": 425.2858489990234,
"mean_ms": 421.2304489135742
},
"parity": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147",
"expected_sha256": "6bdb03e2bdf7733fde92b2cf5c78a373e9818cb5ec9d213d873a632fad0af147"
}
},
"supplied_workspace_bytes": 0
},
{
"block": 24,
"only_monkeypatched_method": "block.mlp.fc2.forward_swiglu",
"accepted_gate_and_residual_path_preserved": true,
"candidate_vs_baseline": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75",
"expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75"
},
"baseline_vs_traversal": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75",
"expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75"
},
"candidate_vs_traversal": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75",
"expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75"
},
"timing": {
"order": "AB/BA alternates by round",
"baseline": {
"samples_ms": [
458.4346008300781,
459.98040771484375,
457.2791748046875,
456.69989013671875,
458.1690368652344,
459.9689025878906,
459.0711975097656,
461.275390625,
458.7782897949219,
460.40740966796875,
459.8872985839844,
461.1615905761719,
459.39276123046875,
459.6435546875,
460.915283203125,
461.53076171875,
460.4534912109375,
460.2518615722656,
458.4383544921875,
461.34747314453125
],
"p50_ms": 459.9281005859375,
"p95_ms": 461.3566375732422,
"mean_ms": 459.6543365478516
},
"candidate": {
"samples_ms": [
418.2376708984375,
421.28668212890625,
417.4606018066406,
420.7315368652344,
421.40478515625,
419.4402770996094,
417.5429382324219,
419.78125,
419.68603515625,
421.91644287109375,
422.23394775390625,
420.081298828125,
419.1646728515625,
421.63360595703125,
422.85125732421875,
418.6084899902344,
420.3404235839844,
421.3265075683594,
422.9977111816406,
420.1449279785156
],
"p50_ms": 420.24267578125,
"p95_ms": 422.8585800170899,
"mean_ms": 420.3435531616211
},
"parity": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75",
"expected_sha256": "d8f5c6335e823dc4e3ed8cdd34d90dda0d69297ea47de9b328962dfae00c9f75"
}
},
"supplied_workspace_bytes": 0
},
{
"block": 49,
"only_monkeypatched_method": "block.mlp.fc2.forward_swiglu",
"accepted_gate_and_residual_path_preserved": true,
"candidate_vs_baseline": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6",
"expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6"
},
"baseline_vs_traversal": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6",
"expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6"
},
"candidate_vs_traversal": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6",
"expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6"
},
"timing": {
"order": "AB/BA alternates by round",
"baseline": {
"samples_ms": [
455.2886962890625,
456.0704650878906,
455.62127685546875,
459.4961242675781,
456.5694580078125,
457.70477294921875,
458.9316101074219,
458.7506103515625,
457.3757629394531,
456.0811462402344,
458.1807861328125,
457.0354919433594,
459.8126220703125,
457.84747314453125,
458.5899963378906,
459.2362060546875,
460.5073547363281,
459.9343566894531,
457.9271240234375,
458.1864318847656
],
"p50_ms": 458.053955078125,
"p95_ms": 459.9630065917969,
"mean_ms": 457.95738830566404
},
"candidate": {
"samples_ms": [
414.79840087890625,
421.3484191894531,
416.3827209472656,
416.5849914550781,
419.1302490234375,
414.7456970214844,
419.312255859375,
415.2906494140625,
416.9346008300781,
416.1675720214844,
418.51458740234375,
417.76995849609375,
420.13006591796875,
418.7623596191406,
416.8761901855469,
418.8280029296875,
417.02777099609375,
417.6993408203125,
416.9612121582031,
419.1832580566406
],
"p50_ms": 417.3635559082031,
"p95_ms": 420.190983581543,
"mean_ms": 417.62241516113284
},
"parity": {
"bf16_exact": true,
"different_elements": 0,
"max_abs": 0.0,
"mean_abs": 0.0,
"actual_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6",
"expected_sha256": "3723ebf82eb1e279cc004832b68d071422a45dcc74adac93010e5fde33b715d6"
}
},
"supplied_workspace_bytes": 0
}
],
"errors_and_unsupported": [
{
"feature": "Stream-K",
"supported_public_control": false,
"reason": "CUDA 13 cuBLASLt exposes no documented MatmulAlgoConfig attribute that directly selects Stream-K. Negative SPLITK_NUM values returned by heuristics are preserved as undocumented library sentinels, not claimed as public Stream-K control."
}
]
}