h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json
2026-08-26 16:14:12 +07:00

67 lines
1.9 KiB
JSON

{
"device": "NVIDIA GB10",
"torch": "2.9.1+cu130",
"attention": "sage2",
"resolution": [
1344,
768
],
"frames": 124,
"steps": 1,
"seed": 440420,
"text_tokens": 100,
"warmup_runs": 1,
"cuda_profiler_capture": true,
"argv": [
"tools/profile_sampling_stages.py",
"--attention",
"sage2",
"--steps",
"1",
"--warmup-runs",
"1",
"--uninstrumented",
"--cuda-profiler-capture",
"--output",
"/output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json"
],
"environment_switches": {
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
"CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4",
"CUDA_HOME": "/usr/local/cuda",
"CUDA_INC_PATH": "/usr/local/cuda/include",
"CUDA_INJECTION64_PATH": "/opt/nsys/target-linux-sbsa-armv8/libToolsInjection64.so",
"CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1",
"CUDA_MODULE_LOADING": "EAGER",
"CUDA_VERSION": "13.0.2",
"H3_FUSED_ELEMENTWISE": "1",
"H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors",
"H3_NVFP4_FC2_LT_SPLITK1": "1",
"H3_NVFP4_MODULATE_FUSION": "1",
"H3_NVFP4_SCALE_BACKEND": "vortex",
"H3_NVFP4_SCALE_VERSION": "1",
"H3_NVFP4_SWIGLU_FUSION": "1",
"H3_SAGE_QKV_LAYOUT": "strided_nhd",
"TORCH_COMPILE_DISABLE": "0",
"TORCH_CUDA_ARCH_LIST": "12.1a",
"TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions"
},
"elapsed_seconds": 20.90812512299999,
"stage_trace": [],
"checksums": [
-197605.3125,
528.5980224609375
],
"sha256": {
"video": "b5fc9bd43ff65f189797fd2377d5a48d8d57f64fb2774fb729dc2be478c39a4b",
"audio": "de6da2540b639a821be7d069efcbf067b209b20de3f7daa0bb9ee28e2624764c"
},
"fc2_dispatch_delta": {
"attempts": 50,
"successes": 50,
"fallbacks": 0
},
"peak_allocated_bytes": 15570401280,
"peak_reserved_bytes": 18538823680,
"measurement_policy": "uninstrumented sampling wall time"
}