{ "device": "NVIDIA GB10", "torch": "2.9.1+cu130", "attention": "sage2", "resolution": [ 1344, 768 ], "frames": 124, "steps": 1, "seed": 440420, "text_tokens": 100, "warmup_runs": 1, "cuda_profiler_capture": true, "argv": [ "tools/profile_sampling_stages.py", "--attention", "sage2", "--steps", "1", "--warmup-runs", "1", "--uninstrumented", "--cuda-profiler-capture", "--output", "/output/h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-capture-20260826.json" ], "environment_switches": { "CUDA_DEVICE_MAX_CONNECTIONS": "1", "CUDA_DEVICE_MAX_COPY_CONNECTIONS": "4", "CUDA_HOME": "/usr/local/cuda", "CUDA_INC_PATH": "/usr/local/cuda/include", "CUDA_INJECTION64_PATH": "/opt/nsys/target-linux-sbsa-armv8/libToolsInjection64.so", "CUDA_MANAGED_FORCE_DEVICE_ALLOC": "1", "CUDA_MODULE_LOADING": "EAGER", "CUDA_VERSION": "13.0.2", "H3_FUSED_ELEMENTWISE": "1", "H3_MODEL_PATH": "/models/minimax_h3_ref2va_pruned_nvfp4.safetensors", "H3_NVFP4_FC2_LT_SPLITK1": "1", "H3_NVFP4_MODULATE_FUSION": "1", "H3_NVFP4_SCALE_BACKEND": "vortex", "H3_NVFP4_SCALE_VERSION": "1", "H3_NVFP4_SWIGLU_FUSION": "1", "H3_SAGE_QKV_LAYOUT": "strided_nhd", "TORCH_COMPILE_DISABLE": "0", "TORCH_CUDA_ARCH_LIST": "12.1a", "TORCH_EXTENSIONS_DIR": "/opt/h3-blackwell-runtime/.torch_extensions" }, "elapsed_seconds": 20.90812512299999, "stage_trace": [], "checksums": [ -197605.3125, 528.5980224609375 ], "sha256": { "video": "b5fc9bd43ff65f189797fd2377d5a48d8d57f64fb2774fb729dc2be478c39a4b", "audio": "de6da2540b639a821be7d069efcbf067b209b20de3f7daa0bb9ee28e2624764c" }, "fc2_dispatch_delta": { "attempts": 50, "successes": 50, "fallbacks": 0 }, "peak_allocated_bytes": 15570401280, "peak_reserved_bytes": 18538823680, "measurement_policy": "uninstrumented sampling wall time" }