h3-blackwell-runtime/benchmarks/gb10-cute-qkv-runtime-block-gate-summary.json
2026-08-25 20:30:22 +07:00

38 lines
1.1 KiB
JSON

{
"device": "NVIDIA GB10",
"workload": {
"width": 1344,
"height": 768,
"frames": 124,
"tokens": 37810,
"seed": 440420
},
"method": "Alternating baseline and ring execution inside one loaded block",
"capacity_2048": {
"block_0": {
"equal": true,
"baseline_p50_ms": 461.54195550479926,
"ring_p50_ms": 465.29679899686016,
"improvement_percent": -0.8135432645455021
},
"block_24": {
"equal": true,
"baseline_p50_ms": 467.8640030033421,
"ring_p50_ms": 470.3063364722766,
"improvement_percent": -0.5220178199768499
},
"block_49": {
"equal": true,
"baseline_p50_ms": 462.3519679880701,
"ring_p50_ms": 464.76577199064195,
"improvement_percent": -0.5220706668721542
}
},
"block_24_capacity_sweep": {
"3072": -1.0036016035203765,
"4096": -0.2438296625217884,
"8192": -4.798436757705438,
"37888": -14.49047057045394
},
"decision": "Reject runtime dispatch. Keep opt-in and disabled until launch fusion or a different persistent scheduler passes the alternating block gate. Skip trajectory validation."
}