h3-blackwell-runtime/benchmarks/gb10-post-fc2-warmed-step-stats_cuda_gpu_trace.csv

2805 lines
833 KiB
CSV
Raw Normal View History

2026-08-26 16:14:12 +07:00
Start (ns),Duration (ns),CorrId,GrdX,GrdY,GrdZ,BlkX,BlkY,BlkZ,Reg/Trd,StcSMem (MB),DymSMem (MB),Bytes (MB),Throughput (MB/s),SrcMemKd,DstMemKd,Device,Ctx,GreenCtx,Strm,Name
6687024,129408,4655,,,,,,,,,,14.322,110663.498,Device,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Device]
6818800,4288,4667,,,,,,,,,,0.053,12358.158,Device,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Device]
7204912,1888,4680,,,,,,,,,,0.000,4.237,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device]
7275376,2144,4697,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7292688,2816,4708,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7305872,3392,4719,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7313232,1088,4730,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7320496,1088,4741,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7326704,1056,4752,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7332624,2080,4763,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7339504,1824,4774,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7357872,2752,4788,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7370736,5728,4800,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl<at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7381104,1120,4811,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7392976,1984,4822,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7417616,38432,4839,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7459632,48832,4854,6993,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 12)]::operator ()() const::[lambda(c10::BFloat16) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7510000,71520,4869,6993,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7647600,4915936,4896,2336,3,1,256,1,1,212,0.000,0.049,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2<cutlass_80_simt_sgemm_256x128_8x4_tn_align1>(T1::Params)
12565488,5035136,4909,195804,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17601648,3808,4927,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17607760,44096,4953,56,6,1,128,1,1,130,0.000,0.031,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2<cutlass_80_simt_sgemm_128x64_8x5_tt_align1>(T1::Params)
17653744,30368,4966,2174,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17685488,3744,4978,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17691728,1792,4989,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17696080,1344,5000,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17700080,2048,5011,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17704272,1536,5022,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17708368,1280,5033,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17712368,1376,5044,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17716560,2080,5055,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17720400,2848,5066,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17729712,6688,5072,,,,,,,,,,0.000,0.598,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host]
17749968,1024,5083,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17754544,608,5089,,,,,,,,,,0.000,6.579,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host]
17778576,960,5102,,,,,,,,,,0.000,4.167,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device]
17807664,3766816,5114,1536,3,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::CatArrayBatchedCopy_alignedK_contig<at::native::<unnamed>::OpaqueType<(unsigned int)2>, unsigned int, (int)2, (int)128, (int)1, (int)8>(T1 *, at::native::<unnamed>::CatArrInputTensorMetadata<T1, T2, T4, T5>, at::native::<unnamed>::TensorSizeStride<T2, (unsigned int)4>, int, T2)"
26302704,20288,5127,,,,,,,,,,0.907,44727.718,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device]
26372208,8192,5141,222,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
26413968,27680,5153,7090,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
26449776,52384,5165,96,3,1,512,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::CatArrayBatchedCopy<at::native::<unnamed>::OpaqueType<(unsigned int)4>, unsigned int, (int)2, (int)64, (int)64>(T1 *, at::native::<unnamed>::CatArrInputTensorMetadata<T1, T2, T4, T5>, at::native::<unnamed>::TensorSizeStride<T2, (unsigned int)4>, int, T2)"
26503152,31872,5176,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::cos_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
26536016,39456,5187,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::sin_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
26577904,39712,5198,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
26618864,873696,5210,1536,4,1,128,1,1,37,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::CatArrayBatchedCopy_alignedK_contig<at::native::<unnamed>::OpaqueType<(unsigned int)4>, unsigned int, (int)3, (int)128, (int)1, (int)16>(T1 *, at::native::<unnamed>::CatArrInputTensorMetadata<T1, T2, T4, T5>, at::native::<unnamed>::TensorSizeStride<T2, (unsigned int)4>, int, T2)"
27495408,217120,5224,7090,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
27714544,2240,5236,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
27718736,1536,5247,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
27722736,3424,5258,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
27728848,3168,5272,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
27733072,3136,5284,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
27737488,4736,5296,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
27743504,1088,5309,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
27747568,1632,5321,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
27751536,1504,5334,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
27755760,2016,5345,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
27759856,30048,5366,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
27792368,3544672,5389,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
31338480,2464,5404,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
31342576,1984,5422,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
31346672,49568,5462,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
31398896,3452992,5470,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
34853840,2912,5473,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
34857968,2466752,5476,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
37326064,1920,5487,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
37330160,24059776,5516,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
61391216,11842304,5533,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
73250640,448,5547,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
73252304,2395136,5550,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
75649264,4001984,5583,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
79653872,3999008,5585,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
83654864,5297792,5601,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
88955088,5731328,5624,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
94689520,240315488,5627,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
335006992,2176224,5634,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
337185008,1952,5637,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
337189104,1376,5651,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
337193200,1504,5666,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
337197296,40608,5689,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
337240272,2775488,5701,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
340017392,1888,5710,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
340021488,7702240,5739,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
347725040,5461056,5753,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
353188080,3717632,5773,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
356907248,5856,5788,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
356915440,3776,5806,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
356921584,60928,5846,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
356984048,4091840,5854,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
361077968,1952,5857,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
361082096,2437024,5860,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
363520432,1920,5871,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
363524336,34563520,5900,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
398089488,140224,5941,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
398231792,11923520,5949,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
410157296,2464,5952,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
410161392,10756480,5955,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
420920560,1920,5966,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
420924656,3040,5981,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
420929776,16202336,6002,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
437133584,5455232,6015,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
442590448,1984,6023,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
442594544,1440,6034,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
442598640,1376,6045,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
442602736,3168,6059,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
442608848,1344,6071,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
442613040,4608,6083,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
442618992,1120,6096,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
442623216,1632,6108,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
442627312,1472,6121,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
442631408,1248,6132,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
442635504,29472,6153,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
442666256,3479584,6176,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
446148848,2368,6191,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
446152944,2112,6209,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
446157040,48416,6249,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
446208240,3449888,6257,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
449660144,4032,6260,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
449666288,2435424,6263,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
452104432,1888,6274,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
452108528,25770432,6303,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
477881584,11819968,6320,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
489703312,416,6334,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
489704624,2346688,6337,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
492053744,3809824,6370,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
495865072,3947776,6372,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
499815664,5406624,6388,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
505224432,5664096,6411,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
510891248,240444576,6414,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
751338768,2172448,6421,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
753512656,1952,6424,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
753516784,2912,6438,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
753520976,1536,6453,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
753525008,46304,6476,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
753573072,2749568,6488,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
756324592,1920,6497,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
756328656,7743936,6526,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
764074288,5461696,6540,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
769538256,3462944,6560,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
773002480,2304,6575,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
773006576,1984,6593,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
773010672,46624,6633,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
773058800,3966080,6641,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
777027824,1952,6644,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
777031920,2729952,6647,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
779763952,1856,6658,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
779768048,33135360,6687,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
812905712,186080,6728,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
813094096,12695904,6736,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
825791728,2176,6739,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
825795920,10475328,6742,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
836272528,2176,6753,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
836276464,832,6768,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
836278352,15614944,6789,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
851894576,5572000,6802,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
857469168,2336,6810,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
857473264,3104,6821,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
857479408,3264,6832,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
857485552,11680,6846,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
857499856,6144,6858,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
857509360,5600,6870,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
857516272,2368,6883,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
857520368,3360,6895,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
857526512,2944,6908,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
857530736,8192,6919,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
857540848,56608,6940,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
857600240,3673856,6963,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
861275376,2464,6978,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
861279440,2048,6996,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
861283536,48960,7036,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
861334736,3813056,7044,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
865150160,3776,7047,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
865156304,2434944,7050,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
867593456,1888,7061,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
867597584,23937632,7090,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
891537648,12480864,7107,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
904019856,416,7121,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
904021456,2345600,7124,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
906369264,3709440,7157,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
910080208,3724512,7159,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
913807568,5534656,7175,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
919343504,5795488,7198,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
925141232,239349632,7201,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
1164493200,2170784,7208,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
1166665936,1920,7211,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
1166670064,2912,7225,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1166674288,1504,7240,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
1166678288,47104,7263,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1166728400,2757664,7275,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
1169489136,1952,7284,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
1169493232,7799744,7313,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
1177295120,5451936,7327,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
1182748880,3475680,7347,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
1186227440,2400,7362,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1186231504,2080,7380,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1186235600,46016,7420,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1186282896,3462240,7428,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
1189746928,1984,7431,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
1189750992,2452064,7434,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
1192204528,1888,7445,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
1192208624,34645696,7474,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
1226856784,139968,7515,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1226999024,12761440,7523,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
1239763184,1920,7526,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
1239767280,10456096,7529,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
1250225392,1952,7540,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
1250229456,864,7555,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1250231632,15792128,7576,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
1266026768,5824160,7589,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
1271852272,1952,7597,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1271856368,1408,7608,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
1271860432,2848,7619,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1271864560,3200,7633,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
1271870704,1376,7645,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1271874800,4416,7657,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
1271880912,1120,7670,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
1271885040,1600,7682,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
1271889136,1472,7695,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
1271893200,1280,7706,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1271897392,26016,7727,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
1271925968,3545984,7750,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
1275473232,2496,7765,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1275477232,2112,7783,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1275481328,47552,7823,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1275530768,4259520,7831,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
1279793424,3712,7834,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
1279799536,2444704,7837,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
1282245872,1824,7848,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
1282249936,24882560,7877,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
1307135248,12475168,7894,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
1319611344,416,7908,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
1319612656,2339296,7911,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
1321954544,3722080,7944,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
1325678832,3736032,7946,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
1329416432,5150368,7962,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
1334569200,5873152,7985,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
1340443888,238904576,7988,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
1579351312,2158912,7995,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
1581512912,1984,7998,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
1581517040,2880,8012,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1581521200,1536,8027,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
1581525136,54208,8050,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1581580624,2747456,8062,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
1584329936,1920,8071,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
1584334064,7804864,8100,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
1592141072,5612512,8114,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
1597755632,3661824,8134,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
1601419504,2400,8149,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1601423568,2112,8167,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1601427696,46240,8207,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1601475824,3859104,8215,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
1605337328,1952,8218,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
1605341456,2438848,8221,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
1607782640,1920,8232,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
1607786736,34043744,8261,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
1641831824,139840,8302,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1641972944,12366720,8310,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
1654341872,1920,8313,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
1654345968,10667840,8316,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
1665016048,1984,8327,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
1665020144,2560,8342,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1665024240,15704512,8363,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
1680730352,5454720,8376,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
1686186480,1952,8384,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1686189744,1440,8395,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
1686193392,1312,8406,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1686197488,3264,8420,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
1686203600,1344,8432,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1686207728,4640,8444,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
1686213872,1088,8457,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
1686217968,1632,8469,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
1686222160,1504,8482,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
1686226160,1280,8493,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1686230256,27264,8514,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
1686258928,3493056,8537,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
1689753872,2464,8552,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1689757936,2016,8570,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
1689762032,47456,8610,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1689811152,3918208,8618,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
1693731056,3392,8621,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
1693737200,2845728,8624,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
1696585968,1856,8635,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
1696590096,25301536,8664,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
1721894096,11901664,8681,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
1733796816,384,8695,,,,,,,,,,0.000,583.333,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
1733798384,2551328,8698,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
1736351984,3738560,8731,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
1740092624,3951712,8733,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
1744046320,5119744,8749,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
1749167344,5697824,8772,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
1754867920,240193088,8775,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
1995062544,2167328,8782,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
1997232368,1920,8785,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
1997236432,2944,8799,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
1997240656,1504,8814,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
1997244624,47296,8837,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
1997293808,2750016,8849,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
2000046288,1952,8858,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2000050416,7806304,8887,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2007859472,5472288,8901,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
2013334768,3708736,8921,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
2017045712,6624,8936,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2017053936,2144,8954,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2017058000,47936,8994,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2017107216,4063680,9002,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
2021173488,1952,9005,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2021177584,2438432,9008,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
2023617776,1920,9019,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2023621872,34049568,9048,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2057673968,139872,9089,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2057816272,11888096,9097,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
2069705936,1984,9100,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2069710064,10741088,9103,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
2080453872,1952,9114,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2080457968,2496,9129,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2080462064,16236448,9150,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2096700624,5467584,9163,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
2102170832,1952,9171,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2102174960,1440,9182,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
2102179056,1312,9193,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2102181904,3456,9207,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
2102187248,1344,9219,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2102191280,4608,9231,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
2102197488,1088,9244,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
2102201520,1632,9256,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
2102205680,1728,9269,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
2102209776,1280,9280,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2102213840,26368,9301,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
2102242544,3495232,9324,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
2105740528,2496,9339,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2105744624,1984,9357,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2105748720,47232,9397,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2105797840,3515456,9405,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
2109316304,3360,9408,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2109322480,2434016,9411,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
2111758576,1888,9422,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2111762672,25849184,9451,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2137614608,11818912,9468,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
2149435280,416,9482,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
2149436720,2354816,9485,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
2151792912,3750336,9518,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2155544784,4064384,9520,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2159612112,5416480,9536,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2165031152,5667744,9559,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2170700176,242080160,9562,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
2412782864,2159488,9569,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
2414944496,1952,9572,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
2414948592,2944,9586,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2414952560,1504,9601,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
2414955312,49536,9624,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2415006928,2752224,9636,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
2417760496,1888,9645,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2417764560,7814080,9674,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2425580816,5456192,9688,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
2431038672,3483520,9708,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
2434524624,6336,9723,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2434532592,36384,9741,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2434571504,54208,9781,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2434628848,4283872,9789,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
2438914256,1952,9792,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2438918384,2440160,9795,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
2441360624,1888,9806,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2441364720,33763776,9835,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2475131248,139488,9876,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2475272432,12292416,9884,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
2487567568,2176,9887,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2487571728,10786304,9890,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
2498360560,1920,9901,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2498364656,832,9916,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2498366768,15820672,9937,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2514190576,5490432,9950,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
2519683344,2272,9958,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2519687440,1408,9969,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
2519691504,3712,9980,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2519697648,8064,9994,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
2519707856,1568,10006,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2519711920,8096,10018,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
2519722224,6880,10031,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
2519730384,1632,10043,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
2519734512,14048,10056,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
2519750896,3648,10067,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2519757040,93376,10088,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
2519853264,3592384,10111,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
2523448560,2912,10126,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2523452720,2240,10144,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2523456752,47712,10184,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2523505904,3622240,10192,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
2527129808,4000,10195,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2527135984,2437120,10198,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
2529576208,1856,10209,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2529580272,25405312,10238,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2554986864,12071296,10255,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
2567060336,448,10269,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
2567062000,2347552,10272,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
2569410832,3711616,10305,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2573123824,4098912,10307,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2577223984,5356480,10323,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2582582512,5691456,10346,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2588277008,239531264,10349,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
2827810064,2158528,10356,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
2829970640,1984,10359,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
2829974768,2720,10373,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2829978864,1504,10388,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
2829982928,46080,10411,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2830032080,2749120,10423,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
2832783600,1920,10432,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2832787696,7819616,10461,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2840610032,5449728,10475,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
2846062832,3479168,10495,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
2849543408,2336,10510,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2849547504,1984,10528,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2849551600,46464,10568,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2849600720,3413024,10576,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
2853015792,1984,10579,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2853019888,2757152,10582,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
2855779536,1856,10593,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2855783632,33791744,10622,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2889576720,141536,10663,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2889721072,12811136,10671,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
2902534352,2208,10674,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2902538480,10362432,10677,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
2912903408,1952,10688,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2912907472,864,10703,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2912909616,15744352,10724,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2928656624,5465664,10737,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
2934124752,1984,10745,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2934128912,1408,10756,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
2934132976,3680,10767,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2934139120,3488,10781,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
2934145136,1344,10793,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
2934147728,4608,10805,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
2934154352,1120,10818,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
2934158576,1632,10830,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
2934162672,1472,10843,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
2934166768,1280,10854,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2934170864,25920,10875,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
2934199536,3495488,10898,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
2937696496,2528,10913,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2937700560,1920,10931,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
2937704656,48864,10971,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
2937754864,3453632,10979,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
2941210832,3808,10982,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
2941217008,2435392,10985,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
2943654160,1888,10996,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
2943658224,24871360,11025,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
2968532176,12473728,11042,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
2981008304,416,11056,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
2981009616,2342880,11059,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
2983354608,3711520,11092,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2987067632,3730304,11094,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2990800112,5451680,11110,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
2996253392,5726816,11133,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3001983216,240209920,11136,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
3242196240,2180576,11143,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
3244379376,1984,11146,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
3244383504,2752,11160,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
3244387536,1504,11175,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
3244391664,45632,11198,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3244438736,2899360,11210,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
3247340784,2016,11219,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
3247344848,8260608,11248,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
3255608560,5606272,11262,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
3261217040,3532864,11282,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
3264751824,2528,11297,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3264755920,1984,11315,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3264760048,46304,11355,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3264808144,3517760,11363,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
3268327664,2144,11366,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
3268331760,2417696,11369,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
3270751472,1888,11380,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
3270755568,34255616,11409,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
3305013552,141536,11450,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3305157872,12819968,11458,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
3317979344,2464,11461,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
3317983472,10346848,11464,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
3328333040,2144,11475,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
3328337136,864,11490,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3328339280,15735488,11511,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
3344077040,5454016,11524,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
3349532912,2016,11532,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
3349537008,1408,11543,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
3349541104,1312,11554,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
3349545200,3456,11568,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
3349551344,1312,11580,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
3349555568,4224,11592,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
3349561520,1088,11605,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
3349565680,1632,11617,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
3349569744,1504,11630,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
3349573872,1280,11641,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3349577936,26720,11662,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
3349606640,3496640,11685,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
3353104752,2400,11700,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3353108720,2112,11718,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3353112816,47680,11758,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3353162960,3506560,11766,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
3356672208,3680,11769,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
3356678352,2438752,11772,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
3359119600,1888,11783,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
3359123664,24909184,11812,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
3384035568,12198048,11829,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
3396235152,416,11843,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
3396236784,2509216,11846,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
3398747472,3777248,11879,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3402526000,3770368,11881,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3406299344,5082752,11897,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3411384560,5710464,11920,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3417097904,240101408,11923,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
3657200912,2166688,11930,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
3659370736,1920,11933,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
3659374832,2528,11947,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
3659378896,1536,11962,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
3659382896,45856,11985,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3659431120,2762624,11997,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
3662195920,1920,12006,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
3662200016,8099264,12035,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
3670300976,5523968,12049,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
3675826448,3822336,12069,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
3679650064,2528,12084,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3679654128,2080,12102,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3679658224,46624,12142,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3679707344,3806144,12150,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
3683514768,1920,12153,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
3683518704,2444096,12156,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
3685965040,1920,12167,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
3685969136,34983616,12196,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
3720954160,139744,12237,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3721095408,12296512,12245,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
3733394640,9696,12248,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
3733406960,10774656,12251,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
3744184560,7232,12262,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
3744194928,3328,12277,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3744200976,15784672,12298,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
3759986992,5478976,12311,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
3765468400,1952,12319,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
3765472496,1440,12330,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
3765476592,1280,12341,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
3765480656,3456,12355,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
3765486800,1312,12367,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
3765490800,4448,12379,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
3765497040,1312,12392,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
3765499728,1632,12404,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
3765503216,1472,12417,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
3765507312,1280,12428,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3765511408,24768,12449,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
3765538032,3497504,12472,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
3769037040,2464,12487,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3769041136,2016,12505,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
3769045232,48544,12545,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
3769095376,3424864,12553,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
3772522768,3552,12556,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
3772528912,2510688,12559,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
3775041776,1888,12570,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
3775045872,24845056,12599,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
3799893232,11830912,12616,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
3811725200,416,12630,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
3811726480,2388928,12633,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
3814117616,3942304,12666,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3818062032,3950304,12668,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3822014736,5234368,12684,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3827250416,5660672,12707,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
3832913136,240325056,12710,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
4073240848,2157632,12717,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
4075400400,1952,12720,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
4075404496,2560,12734,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4075408624,1504,12749,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
4075412784,50400,12772,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4075465936,2760608,12784,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
4078227856,1952,12793,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4078231792,7826784,12822,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4086061328,5463200,12836,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
4091526352,3570432,12856,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
4095099184,2336,12871,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4095103184,2208,12889,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4095107312,48384,12929,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4095157488,4305856,12937,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
4099465456,1920,12940,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
4099469584,2451072,12943,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
4101923088,1856,12954,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4101927120,33911744,12983,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4135841008,206816,13024,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4136050928,12091680,13032,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
4148144336,2208,13035,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
4148148464,10751008,13038,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
4158902512,1920,13049,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4158906608,3328,13064,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4158912752,15259232,13085,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4174173648,5466400,13098,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
4179642672,1920,13106,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4179646704,1440,13117,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
4179650768,1312,13128,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4179654896,3072,13142,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
4179661040,1312,13154,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4179665040,4800,13166,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
4179671248,1120,13179,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
4179675312,1632,13191,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
4179679472,1568,13204,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
4179683536,1248,13215,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4179687696,28672,13236,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
4179718384,3498816,13259,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
4183218480,2432,13274,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4183222512,2112,13292,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4183226576,48096,13332,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4183276752,3518400,13340,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
4186796464,3456,13343,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
4186802448,2435904,13346,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
4189240560,1888,13357,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4189244656,26090176,13386,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4215337200,12096896,13403,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
4227435376,416,13417,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
4227437008,2347104,13420,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
4229786864,3730976,13453,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
4233519312,4067328,13455,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
4237588688,5428096,13471,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
4243019024,5745408,13494,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
4248766704,242399936,13497,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
4491168048,2168800,13504,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
4493338896,1920,13507,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
4493342960,2880,13521,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4493347184,1504,13536,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
4493351184,46592,13559,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4493399248,2748320,13571,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
4496149744,1952,13580,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4496153872,7929856,13609,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4504086576,5458016,13623,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
4509546736,3492480,13643,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
4513041744,2400,13658,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4513045712,1984,13676,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4513049840,47232,13716,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4513100016,4172256,13724,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
4517273808,1984,13727,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
4517277936,2537024,13730,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
4519816432,1888,13741,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4519820560,33533088,13770,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4553355536,157600,13811,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4553515248,12645408,13819,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
4566162640,2432,13822,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
4566166768,10464160,13825,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
4576632208,2400,13836,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4576636144,2208,13851,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4576640240,15692288,13872,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4592334064,5631424,13885,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
4597968112,2560,13893,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4597972208,1440,13904,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
4597976272,1472,13915,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4597980368,3680,13929,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
4597986480,3136,13941,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4597992688,6432,13953,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
4598000880,1280,13966,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
4598004976,7072,13978,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
4598013424,11168,13991,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
4598027504,1248,14002,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4598031600,71488,14023,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
4598105296,3607424,14046,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
4601714000,2464,14061,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4601718000,2080,14079,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4601722096,48672,14119,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4601772304,3869632,14127,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
4605645008,4032,14130,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
4605651152,2434688,14133,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
4608087280,1856,14144,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4608091408,25006880,14173,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4633100528,12395104,14190,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
4645497744,416,14204,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
4645499056,2352544,14207,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
4647854320,3720192,14240,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
4651577584,3872096,14242,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
4655452496,5580512,14258,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
4661034288,5712640,14281,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
4666749168,241404896,14284,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
4908156176,2166912,14291,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
4910324976,2016,14294,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
4910329072,2912,14308,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
4910333264,1536,14323,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
4910337264,48064,14346,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4910388432,2752704,14358,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
4913142992,1920,14367,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4913147120,7930720,14396,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4921079152,5452960,14410,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
4926534896,3469120,14430,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
4930006352,2464,14445,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4930010320,2016,14463,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
4930014448,46528,14503,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4930063600,3409568,14511,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
4933475536,1952,14514,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
4933479664,2424224,14517,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
4935905520,1888,14528,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4935909648,33606624,14557,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
4969519376,139712,14598,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4969661680,12773024,14606,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
4982436080,2176,14609,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
4982440176,10351520,14612,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
4992792976,1920,14623,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
4992796912,832,14638,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
4992799056,15773088,14659,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5008574736,5499680,14672,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
5014076656,2496,14680,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5014080752,1440,14691,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
5014084848,3712,14702,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5014090992,3584,14716,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
5014097008,1344,14728,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5014101392,4544,14740,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
5014107408,1088,14753,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
5014111472,1600,14765,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
5014115536,1920,14778,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
5014119664,1248,14789,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5014123728,26144,14810,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
5014152496,3804640,14833,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
5017958640,22080,14848,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5017983184,32192,14866,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5018018032,47840,14906,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5018068176,3836736,14914,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
5021906256,3744,14917,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
5021912336,2440928,14920,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
5024354576,1856,14931,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5024358640,24941184,14960,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5049301200,12450592,14977,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
5061753712,448,14991,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
5061755344,2347648,14994,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
5064105200,3717760,15027,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5067824464,3725984,15029,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5071551696,5502592,15045,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5077055728,5676000,15068,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5082734864,240707424,15071,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
5323444496,2169536,15078,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
5325615312,1952,15081,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
5325619440,2880,15095,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5325623600,1536,15110,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
5325627632,47104,15133,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5325676752,2767680,15145,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
5328446704,1920,15154,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5328450800,8481312,15183,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5336933648,5526464,15197,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
5342462192,3485760,15217,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
5345949936,2400,15232,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5345954032,1984,15250,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5345958128,46688,15290,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5346006224,3422112,15298,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
5349430640,1952,15301,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
5349434576,2419200,15304,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
5351855344,1888,15315,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5351859408,34284960,15344,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5386147088,139712,15385,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5386289392,12727072,15393,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
5399018704,1984,15396,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
5399022864,10358368,15399,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
5409383696,1920,15410,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5409387760,832,15425,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5409389872,15717632,15446,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5425110256,5460384,15459,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
5430572240,1952,15467,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5430576368,1440,15478,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
5430580464,1312,15489,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5430584560,3392,15503,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
5430590672,1376,15515,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5430594800,4256,15527,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
5430600976,1088,15540,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
5430605040,1600,15552,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
5430609104,1504,15565,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
5430613328,1248,15576,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5430617296,25312,15597,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
5430643952,3500224,15620,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
5434146064,2496,15635,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5434150192,2112,15653,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5434154224,49952,15693,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5434206448,4299072,15701,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
5438508272,3648,15704,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
5438514416,2441600,15707,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
5440958704,1888,15718,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5440962800,24888224,15747,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5465852336,12351616,15764,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
5478208304,4000,15778,,,,,,,,,,0.000,56.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
5478213936,2420480,15781,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
5480636656,3707936,15814,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5484346576,3735616,15816,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5488084176,5065568,15832,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5493151984,5827808,15855,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5498981616,241799776,15858,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
5740782896,2166496,15865,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
5742950864,1984,15868,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
5742954704,2912,15882,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5742958928,1536,15897,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
5742962896,46112,15920,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5743012048,2754048,15932,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
5745768688,1920,15941,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5745772752,7818272,15970,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5753593104,5449504,15984,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
5759044816,3466176,16004,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
5762512304,2304,16019,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5762516208,1952,16037,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5762520304,47168,16077,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5762569456,3411264,16085,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
5765983440,1952,16088,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
5765987568,2425440,16091,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
5768415504,1888,16102,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5768419568,34226240,16131,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5802648976,140032,16172,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5802791152,11884448,16180,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
5814676880,1920,16183,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
5814680848,10341376,16186,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
5825025264,1984,16197,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5825029360,2560,16212,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5825033456,15733664,16233,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5840768400,5464736,16246,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
5846234416,1920,16254,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5846238480,1408,16265,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
5846242512,1312,16276,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5846246608,3072,16290,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
5846252752,1344,16302,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
5846256880,4864,16314,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
5846263056,1088,16327,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
5846267152,1664,16339,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
5846271184,1504,16352,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
5846275312,1280,16363,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5846279376,25952,16384,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
5846308080,3495520,16407,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
5849806032,2368,16422,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5849810128,2112,16440,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
5849814224,48096,16480,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
5849864432,3438592,16488,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
5853306064,3424,16491,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
5853312368,2821952,16494,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
5856136464,1856,16505,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
5856140528,25366272,16534,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
5881509168,11847616,16551,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
5893358672,416,16565,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
5893360240,2496864,16568,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
5895859440,3826176,16601,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5899688272,3904192,16603,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5903594704,5218816,16619,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5908815088,5683744,16642,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
5914500336,241798560,16645,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
6156301584,2171424,16652,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
6158475504,1984,16655,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
6158479600,2912,16669,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6158483792,1536,16684,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
6158487792,48224,16707,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6158538960,2755936,16719,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
6161297648,1952,16728,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6161301712,7908928,16757,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6169213200,5461504,16771,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
6174677200,3478752,16791,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
6178157808,2304,16806,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6178161904,2016,16824,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6178166000,45856,16864,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6178214128,3426912,16872,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
6181643472,1984,16875,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
6181647600,2425184,16878,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
6184075504,1888,16889,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6184079600,34891392,16918,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6218972432,141280,16959,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6219116784,11920992,16967,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
6231040240,1952,16970,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
6231044336,10766528,16973,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
6241813744,2144,16984,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6241817840,832,16999,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6241819952,16326752,17020,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6258148592,5467264,17033,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
6263617776,1984,17041,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6263621904,1408,17052,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
6263625968,1312,17063,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6263630064,3264,17077,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
6263636208,1440,17089,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6263640304,4704,17101,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
6263646480,1088,17114,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
6263650544,1600,17126,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
6263654640,1472,17139,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
6263658736,1280,17150,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6263662832,26688,17171,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
6263691504,3502624,17194,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
6267195632,2400,17209,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6267199696,1984,17227,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6267203568,47584,17267,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6267254000,3522112,17275,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
6270778608,3968,17278,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
6270784752,2439168,17281,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
6273226992,1888,17292,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6273231088,25829856,17321,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6299062512,11811904,17338,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
6310876048,416,17352,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
6310877328,2378688,17355,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
6313257552,3908832,17388,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
6317167952,3972352,17390,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
6321143472,5239040,17406,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
6326385040,5665728,17429,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
6332052720,240203808,17432,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
6572259600,2177760,17439,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
6574439664,1952,17442,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
6574443728,2976,17456,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6574447984,1504,17471,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
6574451920,48864,17494,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6574502192,2758368,17506,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
6577262832,1920,17515,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6577266928,7814496,17544,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6585084176,5459264,17558,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
6590546160,3478464,17578,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
6594025904,2400,17593,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6594029808,2112,17611,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6594033904,47328,17651,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6594083024,3426688,17659,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
6597512432,1952,17662,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
6597516528,2426080,17665,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
6599944464,1856,17676,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6599948528,33846624,17705,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6633797872,172064,17746,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6633972944,12456320,17754,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
6646431952,2208,17757,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
6646436080,10598688,17760,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
6657036656,1952,17771,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6657040656,864,17786,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6657042832,15399488,17807,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6672444656,5615488,17820,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
6678061456,2016,17828,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6678065456,1664,17839,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
6678069744,1312,17850,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6678073584,3648,17864,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
6678079696,1344,17876,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6678083824,11104,17888,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
6678096208,1856,17901,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
6678100208,2080,17913,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
6678104304,4032,17926,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
6678110448,1504,17937,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6678114544,62208,17958,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
6678178992,3617312,17981,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
6681798928,2560,17996,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6681803024,2080,18014,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
6681807056,48160,18054,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6681857264,3838528,18062,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
6685698288,4000,18065,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
6685704432,2440192,18068,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
6688146672,1920,18079,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6688150768,26015712,18108,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6714169584,12030144,18125,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
6726202224,544,18139,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
6726203984,2354880,18142,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
6728560880,3711456,18175,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
6732274928,3963872,18177,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
6736241872,5437760,18193,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
6741681360,5725024,18216,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
6747408624,239633120,18219,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
6987044112,2163136,18226,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
6989208784,1952,18229,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
6989212944,2688,18243,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
6989217040,1504,18258,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
6989221104,46848,18281,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
6989270352,2756896,18293,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
6992028880,1952,18302,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
6992033040,7841664,18331,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
6999876048,5460608,18345,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
7005338864,3489408,18365,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
7008830704,2496,18380,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7008834768,1984,18398,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7008838864,46144,18438,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7008886992,3410336,18446,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
7012299984,1952,18449,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7012304112,2667616,18452,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
7014974448,3712,18463,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7014980848,34079488,18492,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7049062704,139456,18533,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7049203952,12768992,18541,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
7061975248,1984,18544,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7061979376,10367616,18547,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
7072348496,1920,18558,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7072352496,864,18573,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7072354640,15705376,18594,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7088061680,5511744,18607,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
7093574960,2272,18615,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7093578992,1408,18626,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7093583088,3680,18637,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7093589200,3616,18651,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7093595248,1408,18663,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7093599472,4448,18675,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
7093605616,1088,18688,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7093609744,1664,18700,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
7093613808,1472,18713,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7093617936,1280,18724,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7093622000,26848,18745,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
7093650640,3741248,18768,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
7097394416,13184,18783,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7097409136,13472,18801,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7097425136,97696,18841,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7097525488,3970560,18849,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
7101498576,3552,18852,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7101504752,2437920,18855,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
7103944944,1888,18866,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7103949040,24860704,18895,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7128811728,12400832,18912,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
7141214064,416,18926,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
7141215376,2338976,18929,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
7143557328,3707360,18962,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7147267280,3739648,18964,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7151008976,5512160,18980,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7156522416,5684192,19003,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7162209552,242227488,19006,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
7404438896,2180384,19013,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
7406620880,1952,19016,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
7406624976,2560,19030,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7406629072,1536,19045,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7406633200,45024,19068,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7406680304,2751680,19080,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
7409433840,1856,19089,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7409437936,7852832,19118,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7417292080,5460768,19132,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
7422755056,3467584,19152,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
7426225392,2368,19167,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7426229488,2112,19185,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7426233616,45408,19225,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7426281712,3408320,19233,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
7429691600,1952,19236,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7429695728,2422752,19239,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
7432121584,1856,19250,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7432125648,34878368,19279,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7467005072,142560,19320,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7467150576,11880896,19328,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
7479033072,1952,19331,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7479037168,10359680,19334,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
7489399088,6624,19345,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7489407216,1568,19360,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7489411312,15720800,19381,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7505134864,5521792,19394,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
7510659312,3904,19402,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7510665424,1440,19413,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7510669552,3008,19424,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7510675696,3520,19438,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7510681776,2048,19450,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7510685936,6688,19462,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
7510694128,2848,19475,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7510698288,1600,19487,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
7510702320,2304,19500,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7510706416,2240,19511,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7510710512,29504,19532,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
7510741296,3520864,19555,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
7514265104,5600,19570,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7514271984,1952,19588,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7514275952,52032,19628,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7514330416,4260224,19636,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
7518592208,1984,19639,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7518596336,2447712,19642,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
7521046800,1856,19653,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7521050608,25226208,19682,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7546278096,12453792,19699,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
7558733712,416,19713,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
7558735472,2351712,19716,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
7561090288,3714304,19749,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7564807376,3786336,19751,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7568596176,5154144,19767,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7573753072,5813440,19790,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7579568368,240922976,19793,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
7820493072,2156896,19800,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
7822651664,1952,19803,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
7822655728,3360,19817,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7822661840,1568,19832,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7822665872,51264,19855,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7822719184,2771712,19867,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
7825492144,1952,19876,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7825496304,7849312,19905,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7833347344,5451136,19919,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
7838800240,3479136,19939,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
7842280656,2368,19954,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7842284752,2144,19972,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7842288880,47424,20012,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7842339056,3417600,20020,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
7845758160,2016,20023,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7845762288,2423840,20026,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
7848189168,1888,20037,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7848193008,34270688,20066,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7882465552,140768,20107,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7882607856,11865984,20115,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
7894475120,2464,20118,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7894479088,10348864,20121,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
7904830704,1952,20132,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7904834800,2656,20147,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7904838896,15733856,20168,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7920575728,5463552,20181,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
7926041840,1920,20189,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7926045904,1440,20200,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7926050000,1312,20211,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7926054096,3520,20225,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7926060272,1344,20237,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
7926064368,4992,20249,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
7926070640,1120,20262,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
7926074640,1632,20274,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
7926078704,1504,20287,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
7926082800,1216,20298,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7926086896,25632,20319,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
7926115568,3505824,20342,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
7929623792,2432,20357,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7929627888,1920,20375,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
7929631984,47648,20415,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
7929681136,3448928,20423,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
7933133040,4032,20426,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
7933139248,2809952,20429,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
7935952112,1856,20440,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
7935956272,25293632,20469,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
7961252080,11798880,20486,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
7973053296,448,20500,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
7973054672,2382432,20503,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
7975439600,3919904,20536,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7979362544,3922944,20538,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7983287504,5215584,20554,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7988505808,5690784,20577,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
7994200656,241148000,20580,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
8235351312,2184512,20587,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
8237538512,1984,20590,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
8237542640,3040,20604,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
8237548752,1568,20619,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
8237552784,52064,20642,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8237606128,2756640,20654,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
8240365808,1888,20663,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
8240369904,7952640,20692,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
8248324368,5449216,20706,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
8253775088,3472800,20726,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
8257249520,2400,20741,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8257253616,1984,20759,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8257257712,45696,20799,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8257305840,3431232,20807,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
8260739280,1952,20810,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
8260743472,2425536,20813,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
8263171312,1888,20824,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
8263175408,34385728,20853,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
8297562448,140160,20894,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8297705712,11916032,20902,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
8309624048,2432,20905,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
8309628144,10800032,20908,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
8320430320,1920,20919,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
8320434416,832,20934,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8320436624,16112576,20955,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
8336551152,5465952,20968,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
8342019312,1952,20976,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
8342023440,1440,20987,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
8342027536,1312,20998,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
8342031600,3680,21012,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
8342037776,1312,21024,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
8342041808,4896,21036,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
8342047984,1088,21049,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
8342052080,1632,21061,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
8342056176,1472,21074,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
8342060272,1248,21085,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8342064368,26112,21106,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
8342093040,3493632,21129,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
8345587952,2560,21144,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8345592016,1984,21162,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8345596144,47840,21202,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8345646320,3444576,21210,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
8349092208,4096,21213,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
8349098224,2448192,21216,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
8351548656,1888,21227,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
8351552784,25738272,21256,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
8377295632,11891808,21273,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
8389188464,544,21287,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
8389190192,2346944,21290,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
8391538928,3761664,21323,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
8395303152,4026528,21325,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
8399331536,5417472,21341,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
8404751600,5663296,21364,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
8410416368,241593728,21367,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
8652011792,2164960,21374,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
8654178544,1920,21377,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
8654182640,2976,21391,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
8654186896,1536,21406,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
8654190832,48416,21429,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8654242032,2752160,21441,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
8656995568,1888,21450,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
8656999664,7942144,21479,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
8664943920,5458208,21493,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
8670404848,3475616,21513,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
8673883376,4320,21528,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8673889680,19776,21546,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8673912016,52288,21586,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8673967312,4322208,21594,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
8678291696,1952,21597,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
8678295760,2444640,21600,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
8680743152,1920,21611,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
8680747248,34577664,21640,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
8715326704,141088,21681,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8715469072,12382976,21689,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
8727853392,1984,21692,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
8727857360,10443584,21695,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
8738303216,1952,21706,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
8738307312,2592,21721,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8738311408,15308832,21742,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
8753622256,5479104,21755,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
8759102704,1952,21763,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
8759106800,1408,21774,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
8759110896,3648,21785,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
8759117040,3072,21799,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
8759123184,1376,21811,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
8759127184,4640,21823,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
8759133456,1152,21836,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
8759137264,1600,21848,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
8759141584,1472,21861,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
8759145712,1248,21872,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8759149776,25216,21893,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
8759176432,3694624,21916,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
8762874096,2400,21931,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8762878256,2176,21949,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
8762882288,50528,21989,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
8762934480,3638240,21997,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
8766574800,1952,22000,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
8766578928,2442912,22003,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
8769024368,1920,22014,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
8769028368,25552160,22043,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
8794582288,12077248,22060,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
8806660976,448,22074,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
8806662096,2352160,22077,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
8809015568,3721376,22110,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
8812738768,4098432,22112,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
8816839888,5346304,22128,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
8822189264,5725184,22151,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
8827916528,241606016,22154,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
9069525264,2174688,22161,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
9071702320,1952,22164,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
9071706384,2720,22178,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9071710416,1536,22193,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
9071714512,47232,22216,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9071763664,2758720,22228,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
9074524400,1952,22237,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9074528496,7921504,22266,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9082451344,5461920,22280,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
9087915248,3486688,22300,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
9091405040,2400,22315,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9091409136,2080,22333,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9091413200,46048,22373,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9091462352,3874528,22381,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
9095339248,1952,22384,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
9095343344,2624832,22387,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
9097970928,1856,22398,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9097975024,33697120,22427,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9131674896,139840,22468,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9131817200,11919712,22476,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
9143739600,1952,22479,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
9143743760,10381664,22482,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
9154128944,1952,22493,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9154133104,1824,22508,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9154136304,15711680,22529,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9169850608,5507136,22542,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
9175359728,1920,22550,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9175363824,1568,22561,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
9175367920,3168,22572,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9175374096,3392,22586,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
9175380144,1344,22598,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9175384336,4928,22610,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
9175390736,1120,22623,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
9175394448,1632,22635,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
9175398640,1696,22648,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
9175402768,7616,22659,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9175412976,26336,22680,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
9175441520,3844544,22703,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
9179287792,2336,22718,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9179291888,2048,22736,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9179295984,48768,22776,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9179347184,3828800,22784,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
9183177968,4032,22787,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
9183184112,2423808,22790,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
9185611024,1888,22801,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9185615056,24620416,22830,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9210238192,12466528,22847,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
9222706064,480,22861,,,,,,,,,,0.000,466.667,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
9222707760,2350880,22864,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
9225061616,3725056,22897,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
9228787952,3735968,22899,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
9232525520,5525984,22915,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
9238053104,5802432,22938,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
9243858128,242206816,22941,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
9486066960,2168960,22948,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
9488237840,2016,22951,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
9488241904,2944,22965,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9488246128,1536,22980,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
9488250096,45504,23003,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9488297168,2763104,23015,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
9491063024,1920,23024,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9491067120,7944544,23053,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9499014416,5450496,23067,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
9504466192,3475360,23087,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
9507942800,2368,23102,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9507946736,2048,23120,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9507950832,47552,23160,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9507999984,3421152,23168,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
9511424272,1952,23171,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
9511428336,2483008,23174,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
9513912656,2368,23185,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9513916848,34376960,23214,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9548296464,139616,23255,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9548438768,12796832,23263,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
9561237712,2240,23266,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
9561241840,10357696,23269,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
9571601648,1920,23280,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9571605744,832,23295,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9571607856,15753632,23316,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9587363056,5458112,23329,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
9592823024,1984,23337,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9592827216,1408,23348,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
9592831216,1312,23359,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9592835312,3296,23373,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
9592841456,1344,23385,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9592845456,4448,23397,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
9592851664,1088,23410,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
9592855696,1632,23422,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
9592859888,1472,23435,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
9592863984,1280,23446,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9592868080,25184,23467,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
9592894704,3628416,23490,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
9596533904,17728,23505,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9596554480,11680,23523,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9596568848,69088,23563,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9596639472,4118272,23571,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
9600760016,3840,23574,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
9600766224,2419840,23577,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
9603187952,1888,23588,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9603192048,24930496,23617,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9628124400,12466368,23634,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
9640592240,544,23648,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
9640593680,2350944,23651,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
9642946832,3729024,23684,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
9646678224,3748576,23686,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
9650428112,5354496,23702,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
9655784720,5783808,23725,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
9661571312,241193152,23728,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
9902766384,2168640,23735,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
9904937200,1952,23738,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
9904941296,2944,23752,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
9904945520,1568,23767,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
9904949456,47232,23790,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9904998640,2753088,23802,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
9907754224,1888,23811,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9907758320,7844192,23840,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9915605264,5462816,23854,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
9921070352,3466848,23874,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
9924539632,2304,23889,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9924543728,2048,23907,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
9924547792,47456,23947,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9924596944,3416192,23955,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
9928016112,1952,23958,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
9928020208,2424832,23961,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
9930448112,1856,23972,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9930452208,34503808,24001,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
9964957968,140736,24042,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9965101264,12747232,24050,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
9977850096,2208,24053,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
9977854192,10374624,24056,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
9988231408,1920,24067,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
9988235504,2528,24082,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
9988239600,15701312,24103,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10003943664,5469120,24116,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
10009414896,1920,24124,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10009418992,1408,24135,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
10009423088,1376,24146,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10009427184,3136,24160,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
10009433328,1344,24172,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10009437424,6720,24184,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
10009445616,1088,24197,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
10009449712,1600,24209,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
10009453808,1472,24222,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
10009457904,1312,24233,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10009462000,27584,24254,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
10009492688,3491904,24277,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
10012987664,2528,24292,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10012991824,2016,24310,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10012995824,48352,24350,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10013046000,4232896,24358,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
10017281232,4096,24361,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
10017287376,2498976,24364,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
10019789040,1952,24375,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10019793104,24911776,24404,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10044707056,12388768,24421,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
10057100048,416,24435,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
10057102288,2363744,24438,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
10059469072,3784768,24471,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10063255792,3747296,24473,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10067004368,5073216,24489,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10072080656,5698400,24512,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10077780336,242541280,24515,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
10320322960,2168832,24522,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
10322493680,1920,24525,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
10322497776,2944,24539,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10322502032,1536,24554,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
10322505968,49696,24577,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10322557232,2751232,24589,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
10325310704,1952,24598,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10325314800,7992640,24627,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10333310288,5463872,24641,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
10338775408,3477760,24661,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
10342254832,2368,24676,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10342258928,2048,24694,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10342263024,46048,24734,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10342312176,3447488,24742,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
10345762032,1952,24745,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
10345766128,2423296,24748,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
10348191984,1920,24759,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10348196080,34563424,24788,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10382762288,140576,24829,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10382905648,12654624,24837,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
10395562192,2688,24840,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
10395566320,10480992,24843,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
10406050032,1920,24854,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10406054096,1088,24869,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10406056656,15701088,24890,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10421760240,5465056,24903,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
10427228432,1920,24911,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10427232496,1408,24922,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
10427236592,1280,24933,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10427240688,3232,24947,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
10427246864,1344,24959,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10427250928,4832,24971,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
10427257040,1120,24984,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
10427261232,1632,24996,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
10427265264,1472,25009,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
10427269360,1280,25020,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10427273456,25920,25041,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
10427302128,3508896,25064,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
10430812144,2464,25079,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10430816496,2016,25097,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10430820592,48512,25137,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10430871792,3516544,25145,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
10434391280,4384,25148,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
10434397424,2736736,25151,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
10437136624,1888,25162,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10437140720,25314144,25191,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10462456176,12035712,25208,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
10474496016,448,25222,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
10474497360,2454048,25225,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
10476953840,3817408,25258,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10480773360,3902560,25260,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10484678928,5091520,25276,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10489772272,5706880,25299,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10495482096,242566784,25302,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
10738051344,2171840,25309,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
10740226256,1984,25312,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
10740230384,3104,25326,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10740236496,1536,25341,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
10740240528,47072,25364,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10740289776,2751712,25376,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
10743043312,1888,25385,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10743047408,7940800,25414,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10750989584,5517184,25428,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
10756508912,3821792,25448,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
10760332496,2496,25463,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10760336720,2048,25481,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10760340720,46816,25521,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10760388816,3812448,25529,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
10764204272,1920,25532,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
10764208336,2426976,25535,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
10766637296,1888,25546,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10766641392,34306016,25575,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10800949488,140000,25616,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10801091824,12269024,25624,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
10813362416,5312,25627,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
10813370608,10774496,25630,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
10824147184,4960,25641,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10824153424,7264,25656,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10824163568,15771104,25677,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10839935984,5466784,25690,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
10845405424,1920,25698,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10845409616,1440,25709,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
10845413648,3168,25720,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10845419760,3168,25734,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
10845425808,1312,25746,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
10845430000,4544,25758,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
10845436144,1088,25771,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
10845440272,1632,25783,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
10845444336,1472,25796,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
10845448400,1312,25807,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10845452528,26560,25828,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
10845481200,3503456,25851,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
10848987408,2368,25866,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10848991472,2016,25884,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
10848995568,48320,25924,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
10849045712,3440576,25932,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
10852487568,4032,25935,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
10852493584,2674176,25938,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
10855170288,1920,25949,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
10855174416,25492928,25978,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
10880668912,11806688,25995,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
10892477296,448,26009,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
10892478928,2395648,26012,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
10894875888,3943904,26045,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10898821328,3916544,26047,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10902740176,5236320,26063,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10907977936,5662944,26086,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
10913642736,242114368,26089,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
11155759408,2167616,26096,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
11157928336,1920,26099,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
11157932400,2912,26113,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11157938448,1504,26128,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
11157942384,47264,26151,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11157991632,2756704,26163,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
11160751312,1984,26172,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11160755472,7890016,26201,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
11168647472,5505440,26215,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
11174154512,3673952,26235,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
11177830640,2592,26250,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11177834704,2048,26268,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11177841872,55936,26308,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11177899216,4036064,26316,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
11181937872,1984,26319,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
11181942000,2423648,26322,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
11184367856,1888,26333,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11184371952,34814112,26362,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
11219188016,140544,26403,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11219331312,11914592,26411,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
11231248592,1952,26414,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
11231252720,10789568,26417,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
11242043632,2112,26428,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11242047696,960,26443,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11242049968,16312800,26464,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
11258365168,5469024,26477,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
11263835472,2144,26485,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11263840496,1408,26496,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
11263844592,1280,26507,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11263848720,3296,26521,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
11263854864,1440,26533,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11263858928,4320,26545,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
11263865072,1120,26558,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
11263869168,1600,26570,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
11263873264,1504,26583,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
11263877360,1248,26594,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11263881424,25120,26615,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
11263908080,3509536,26638,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
11267419344,2432,26653,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11267423472,1984,26671,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11267427568,48000,26711,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11267476944,3441024,26719,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
11270920432,3776,26722,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
11270926576,2436416,26725,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
11273364720,1856,26736,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11273368816,25923936,26765,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
11299295472,11797920,26782,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
11311094640,448,26796,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
11311095984,2385056,26799,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
11313484016,3934176,26832,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
11317421296,3955424,26834,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
11321379536,5190528,26850,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
11326571728,5673920,26873,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
11332247792,242693856,26876,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
11574944016,2174560,26883,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
11577120976,1984,26886,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
11577125104,2752,26900,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11577129168,1536,26915,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
11577133264,45664,26938,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11577180368,2752608,26950,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
11579935952,1920,26959,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11579940048,7985536,26988,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
11587928368,5481344,27002,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
11593411216,3706112,27022,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
11597119728,2560,27037,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11597123856,2336,27055,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11597127920,47072,27095,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11597178096,4002432,27103,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
11601182960,1952,27106,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
11601186800,2423616,27109,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
11603611888,1888,27120,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11603615984,34343840,27149,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
11637962000,140544,27190,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11638105328,11934080,27198,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
11650041072,2144,27201,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
11650045168,10771616,27204,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
11660819696,2016,27215,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11660823792,1056,27230,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11660827888,16033728,27251,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
11676863728,5465600,27264,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
11682330896,1952,27272,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11682334992,1440,27283,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
11682337936,1312,27294,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11682341072,3296,27308,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
11682347248,1312,27320,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11682351248,4576,27332,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
11682357488,1088,27345,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
11682361584,1632,27357,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
11682365680,1472,27370,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
11682369776,1280,27381,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11682373872,26848,27402,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
11682402544,3503616,27425,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
11685908720,2496,27440,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11685912784,1984,27458,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
11685916912,47392,27498,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11685966064,3441888,27506,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
11689409744,3264,27509,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
11689415920,2435552,27512,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
11691853072,1856,27523,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11691857136,26544096,27552,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
11718404336,11803168,27569,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
11730208624,416,27583,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
11730210192,2344192,27586,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
11732556080,3879392,27619,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
11736438000,3889536,27621,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
11740330192,5382944,27637,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
11745714416,5676192,27660,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
11751392528,241273408,27663,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
11992667440,2164000,27670,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
11994834128,1952,27673,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
11994838256,2848,27687,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
11994843376,1536,27702,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
11994847344,47232,27725,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
11994896592,2758016,27737,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
11997657328,2080,27746,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
11997661392,7871520,27775,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12005536048,5454944,27789,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
12010993904,3469600,27809,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
12014465264,2304,27824,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12014469360,2080,27842,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12014473456,46176,27882,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12014522608,3422656,27890,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
12017946864,1952,27893,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12017950960,2425152,27896,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
12020378864,1888,27907,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12020382992,34162048,27936,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12054547696,139200,27977,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12054688272,12332832,27985,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
12067024112,1952,27988,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12067028208,10806528,27991,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
12077837552,1920,28002,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12077841616,3072,28017,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12077847792,15536672,28038,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12093387024,5629920,28051,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
12099018992,1920,28059,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12099023152,1408,28070,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
12099027184,1472,28081,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12099031280,3360,28095,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
12099036400,1312,28107,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12099040496,4448,28119,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
12099046640,1184,28132,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
12099050704,1824,28144,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
12099054832,1632,28157,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
12099058928,1280,28168,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12099063024,25696,28189,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
12099091696,3543520,28212,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
12102636784,2464,28227,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12102640880,2304,28245,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12102644944,63616,28285,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12102710480,3828128,28293,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
12106541296,3296,28296,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12106547440,2443872,28299,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
12108993776,1888,28310,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12108997872,24936000,28339,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12133935344,11814080,28356,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
12145751952,416,28370,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
12145753296,2352896,28373,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
12148107472,3725216,28406,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12151834832,3740704,28408,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12155578576,5090944,28424,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12160671984,5663328,28447,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12166337776,241822912,28450,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
12408162608,2157568,28457,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
12410323184,1952,28460,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
12410327280,2976,28474,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12410331568,1504,28489,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
12410335216,50112,28512,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12410386704,2750304,28524,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
12413139184,1920,28533,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12413143280,7919840,28562,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12421064976,5455904,28576,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
12426523888,3468640,28596,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
12429995248,2368,28611,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12429999344,1984,28629,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12430003408,48128,28669,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12430054640,3424576,28677,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
12433481968,1952,28680,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12433486096,2426240,28683,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
12435913968,1856,28694,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12435918064,33735648,28723,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12469656848,140768,28764,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12469799152,12803680,28772,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
12482604336,2176,28775,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12482608368,10379136,28778,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
12492989680,1952,28789,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12492993776,864,28804,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12492995920,15888288,28825,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12508887280,5504480,28838,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
12514393328,2112,28846,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12514397424,1408,28857,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
12514401520,1280,28868,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12514405616,3744,28882,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
12514411728,1344,28894,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12514414544,4896,28906,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
12514421904,1088,28919,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
12514426096,1632,28931,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
12514430192,1760,28944,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
12514434288,1248,28955,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12514438352,34208,28976,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
12514475248,3824608,28999,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
12518302960,2528,29014,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12518307056,2016,29032,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12518311152,47136,29072,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12518360272,3841696,29080,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
12522203344,3968,29083,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12522209552,2437024,29086,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
12524648688,1888,29097,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12524652816,24963488,29126,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12549617904,12417312,29143,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
12562036624,416,29157,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
12562038256,2346336,29160,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
12564387056,3717536,29193,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12568106224,3741536,29195,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12571849040,5525152,29211,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12577376496,5689024,29234,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12583066864,242627456,29237,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
12825695632,2167232,29244,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
12827865328,1984,29247,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
12827869456,2976,29261,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12827873744,1504,29276,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
12827877616,47264,29299,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12827926736,2752640,29311,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
12830681328,1952,29320,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12830685424,7922240,29349,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12838610192,5462688,29363,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
12844074192,3470944,29383,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
12847546608,2432,29398,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12847550704,2112,29416,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12847554800,47168,29456,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12847604976,3418752,29464,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
12851025136,1952,29467,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12851029232,2467008,29470,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
12853498096,2080,29481,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12853502192,34443328,29510,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12887948560,140224,29551,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12888091888,11968896,29559,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
12900063440,2208,29562,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12900067536,10367680,29565,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
12910436592,1952,29576,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12910440688,832,29591,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12910442800,15742528,29612,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12926187760,5472992,29625,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
12931662032,1984,29633,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12931666160,1440,29644,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
12931670256,3712,29655,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12931676400,3456,29669,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
12931682448,1344,29681,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
12931686640,4352,29693,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
12931692784,1120,29706,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
12931696880,1664,29718,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
12931700976,1504,29731,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
12931705136,1280,29742,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12931709168,25920,29763,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
12931737936,3556960,29786,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
12935297264,2464,29801,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12935301360,2208,29819,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
12935305456,48480,29859,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
12935355760,4213088,29867,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
12939571472,3584,29870,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
12939577584,2431296,29873,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
12942010640,1888,29884,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
12942014704,24939456,29913,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
12966956272,12481600,29930,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
12979440560,448,29944,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
12979441872,2345504,29947,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
12981789936,3715584,29980,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12985507024,3743616,29982,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12989252848,5140192,29998,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
12994395376,5886496,30021,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13000284464,243240192,30024,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
13243527472,2167936,30031,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
13245698288,1952,30034,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
13245702352,2784,30048,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
13245706480,1536,30063,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
13245710576,47040,30086,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13245760720,2864192,30098,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
13248626928,1888,30107,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
13248631024,7919136,30136,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
13256551696,5460800,30150,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
13262013808,3467264,30170,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
13265482992,2336,30185,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13265487056,2080,30203,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13265491184,46144,30243,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13265540336,3419168,30251,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
13268962512,1984,30254,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
13268966640,2430464,30257,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
13271399664,1888,30268,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
13271403760,34480480,30297,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
13305886992,140800,30338,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13306030064,11942176,30346,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
13317974224,2464,30349,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
13317978352,10435040,30352,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
13328414960,1920,30363,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
13328419184,896,30378,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13328423184,15770560,30399,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
13344196880,5478400,30412,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
13349678320,1920,30420,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
13349682416,1440,30431,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
13349686480,1344,30442,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
13349690576,3424,30456,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
13349696752,1344,30468,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
13349700848,4288,30480,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
13349706992,1088,30493,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
13349711088,1632,30505,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
13349715184,1472,30518,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
13349719312,1440,30529,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13349723344,25152,30550,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
13349750000,3505472,30573,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
13353257200,2432,30588,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13353261296,1984,30606,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13353265360,47424,30646,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13353314512,4290496,30654,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
13357609776,5088,30657,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
13357617392,2470944,30660,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
13360091376,1920,30671,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
13360095472,24951584,30700,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
13385048336,12361888,30717,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
13397412144,416,30731,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
13397413936,2352704,30734,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
13399769328,3794400,30767,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13403566320,3743264,30769,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13407311056,5097024,30785,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13412409680,5794272,30808,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13418206448,242996160,30811,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
13661205776,2172704,30818,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
13663380688,1984,30821,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
13663384816,2752,30835,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
13663388912,1536,30850,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
13663393008,45984,30873,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13663440240,2770784,30885,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
13666213072,2016,30894,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
13666217264,7897696,30923,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
13674116368,5469888,30937,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
13679587536,3478528,30957,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
13683068144,2336,30972,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13683072336,2176,30990,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13683076528,47104,31030,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13683125584,3417632,31038,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
13686544592,1952,31041,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
13686548720,2435296,31044,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
13688985872,1920,31055,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
13688989904,35071936,31084,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
13724064048,140832,31125,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13724206320,11934080,31133,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
13736143184,2208,31136,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
13736147184,10473696,31139,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
13746623728,1920,31150,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
13746627792,864,31165,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13746629936,15780192,31186,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
13762411760,5487200,31199,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
13767901520,2016,31207,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
13767905616,1440,31218,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
13767909616,1312,31229,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
13767913808,3488,31243,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
13767919856,1312,31255,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
13767923952,4704,31267,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
13767930192,1120,31280,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
13767934224,1632,31292,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
13767938384,1536,31305,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
13767942480,1312,31316,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13767946480,26112,31337,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
13767975152,3522240,31360,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
13771499728,2336,31375,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13771503856,1984,31393,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
13771507952,47360,31433,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
13771557488,3504160,31441,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
13775063312,3200,31444,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
13775069456,2750464,31447,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
13777822128,1888,31458,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
13777826032,25831808,31487,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
13803659536,12180320,31504,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
13815841840,448,31518,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
13815843056,2370304,31521,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
13818215664,3851904,31554,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13822070000,3824896,31556,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13825897680,5097152,31572,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13830996208,5708416,31595,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
13836707152,243821344,31598,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
14080530704,2179040,31605,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
14082711760,1920,31608,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
14082715984,2752,31622,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14082720112,1632,31637,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
14082724080,49440,31660,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14082775280,2768608,31672,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
14085545200,1888,31681,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14085549296,8041856,31710,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14093592880,5523488,31724,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
14099119312,3508768,31744,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
14102629712,2336,31759,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14102633712,2016,31777,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14102637936,45984,31817,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14102685936,3415424,31825,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
14106104016,1952,31828,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
14106108144,2435424,31831,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
14108546288,1888,31842,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14108550384,34749824,31871,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14143302928,139584,31912,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14143444176,12637024,31920,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
14156083440,2208,31923,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
14156087536,10553792,31926,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
14166643952,1920,31937,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14166648048,2944,31952,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14166652272,15802976,31973,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14182456528,5484832,31986,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
14187943152,1952,31994,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14187947248,1440,32005,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
14187951440,1632,32016,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14187955440,3232,32030,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
14187961744,1376,32042,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14187965776,4896,32054,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
14187971984,1344,32067,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
14187976016,1632,32079,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
14187979984,1504,32092,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
14187984112,1280,32103,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14187988432,28864,32124,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
14188018928,3517408,32147,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
14191538512,2496,32162,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14191542512,1952,32180,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14191546704,49504,32220,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14191598928,3880416,32228,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
14195480816,3968,32231,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
14195486960,2623360,32234,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
14198112592,1888,32245,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14198116592,25732096,32274,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14223850736,12178304,32291,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
14236030896,384,32305,,,,,,,,,,0.000,583.333,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
14236032496,2502656,32308,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
14238537744,3782816,32341,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
14242322640,3773120,32343,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
14246097104,5226624,32359,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
14251325680,5718816,32382,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
14257045808,245704832,32385,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
14502752656,2174368,32392,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
14504929488,1984,32395,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
14504933616,2944,32409,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14504937872,1536,32424,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
14504941808,47296,32447,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14504990928,2773440,32459,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
14507766096,1952,32468,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14507770224,8355648,32497,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14516129136,5595744,32511,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
14521727248,3582016,32531,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
14525311184,2336,32546,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14525315312,2016,32564,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14525319504,45600,32604,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14525366544,3429056,32612,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
14528798032,1984,32615,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
14528802032,2428384,32618,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
14531231984,1984,32629,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14531236112,34769312,32658,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14566007120,140896,32699,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14566150480,12833760,32707,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
14578987216,1952,32710,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
14578991344,10459904,32713,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
14589452592,1952,32724,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14589456624,864,32739,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14589459120,15787264,32760,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14605248752,5485600,32773,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
14610736496,1952,32781,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14610740464,1472,32792,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
14610744656,1312,32803,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14610748752,3552,32817,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
14610754800,1344,32829,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14610758992,4800,32841,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
14610765072,1120,32854,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
14610769232,1632,32866,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
14610773328,1472,32879,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
14610777328,1376,32890,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14610781520,27456,32911,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
14610810256,3513824,32934,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
14614326608,2656,32949,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14614330608,1952,32967,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14614334800,54208,33007,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14614392048,4344608,33015,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
14618737968,4000,33018,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
14618744144,2452704,33021,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
14621198608,1856,33032,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14621202672,25020704,33061,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14646225104,12401824,33078,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
14658628496,416,33092,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
14658629808,2358784,33095,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
14660991216,3720512,33128,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
14664713552,3752288,33130,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
14668467408,5166208,33146,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
14673636688,5912992,33169,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
14679552240,243265760,33172,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
14922819888,2181408,33179,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
14925003024,1984,33182,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
14925007088,2944,33196,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
14925011344,1504,33211,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
14925015280,45504,33234,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14925063376,2772544,33246,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
14927838448,1888,33255,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14927842512,7908448,33284,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14935753104,5470688,33298,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
14941226224,3486752,33318,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
14944716112,2336,33333,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14944720112,1952,33351,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
14944724304,46944,33391,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14944774576,3429888,33399,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
14948205840,1952,33402,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
14948209872,2433824,33405,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
14950645008,1856,33416,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
14950649168,34601024,33445,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
14985252144,140448,33486,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
14985394416,12787328,33494,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
14998183120,1952,33497,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
14998187248,10507648,33500,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
15008697584,1920,33511,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15008701808,832,33526,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15008704208,15794528,33547,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15024500976,5490432,33560,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
15029992720,2080,33568,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15029996816,1440,33579,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
15030000912,3392,33590,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15030007024,3488,33604,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
15030013168,1312,33616,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15030017264,4608,33628,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
15030023376,1120,33641,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
15030027504,1664,33653,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
15030031600,1472,33666,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
15030035696,1280,33677,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15030039792,26944,33698,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
15030068464,3510464,33721,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
15033580752,2432,33736,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15033584848,2016,33754,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15033588944,47936,33794,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15033639152,4327456,33802,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
15037968656,3584,33805,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
15037974800,2438464,33808,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
15040415984,1856,33819,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15040420080,24980000,33848,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15065401712,12355584,33865,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
15077759920,416,33879,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
15077761520,2391040,33882,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
15080155472,3754496,33915,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15083912400,3751616,33917,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15087665392,5103328,33933,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15092770000,5872896,33956,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15098644816,243853824,33959,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
15342500112,2173248,33966,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
15344675024,1920,33969,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
15344679152,2944,33983,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15344683376,1504,33998,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
15344687344,45984,34021,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15344734640,2776768,34033,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
15347512720,1888,34042,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15347516656,7939712,34071,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15355458864,5468864,34085,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
15360929008,3482720,34105,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
15364413712,2432,34120,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15364417776,2144,34138,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15364421872,46048,34178,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15364470000,3435008,34186,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
15367906544,1920,34189,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
15367910640,2439200,34192,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
15370351856,1920,34203,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15370355952,34622048,34232,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15404980528,142464,34273,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15405124944,12814048,34281,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
15417941232,1984,34284,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
15417945328,10463552,34287,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
15428410704,1952,34298,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15428414800,864,34313,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15428418832,15803776,34334,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15444225264,5480064,34347,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
15449707856,1920,34355,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15449711952,1472,34366,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
15449715952,1312,34377,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15449720144,3552,34391,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
15449726192,1344,34403,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15449730288,4480,34415,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
15449736592,1088,34428,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
15449740496,1632,34440,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
15449744848,1504,34453,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
15449748816,1280,34464,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15449752816,25952,34485,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
15449781488,3512480,34508,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
15453295952,2400,34523,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15453299920,2176,34541,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15453304144,47104,34581,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15453354256,4206304,34589,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
15457562896,3264,34592,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
15457569040,2482784,34595,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
15460054352,1888,34606,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15460058448,24982208,34635,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15485041936,12379232,34652,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
15497423760,416,34666,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
15497425328,2372352,34669,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
15499799792,3764256,34702,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15503566032,3745824,34704,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15507314896,5102048,34720,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15512418640,5740512,34743,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15518161904,243390656,34746,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
15761553872,2179552,34753,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
15763735760,1984,34756,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
15763739888,2848,34770,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15763744048,1504,34785,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
15763748048,46464,34808,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15763796304,2774592,34820,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
15766573296,1920,34829,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15766577392,7937408,34858,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15774517552,5466464,34872,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
15779985648,3486240,34892,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
15783474512,2400,34907,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15783478608,2016,34925,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15783482608,45856,34965,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15783530864,3417632,34973,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
15786949840,1952,34976,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
15786953968,2433792,34979,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
15789390064,1856,34990,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15789394160,34660288,35019,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15824057616,140576,35060,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15824200944,12779424,35068,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
15836982480,2496,35071,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
15836986608,10441280,35074,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
15847429456,2016,35085,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15847433552,2912,35100,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15847439600,15823232,35121,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15863264496,5483136,35134,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
15868750192,1920,35142,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15868754160,1440,35153,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
15868758352,3424,35164,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15868764368,3296,35178,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
15868770640,1344,35190,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
15868774768,4416,35202,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
15868780848,1088,35215,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
15868785008,1632,35227,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
15868788976,1504,35240,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
15868793072,1280,35251,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15868797296,26336,35272,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
15868825840,3516480,35295,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
15872345328,2336,35310,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15872349424,2016,35328,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
15872353520,47872,35368,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
15872402736,3900736,35376,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
15876305232,1952,35379,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
15876309232,2704032,35382,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
15879014608,2208,35393,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
15879018736,25014784,35422,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
15904036048,12223424,35439,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
15916260400,416,35453,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
15916262000,2486112,35456,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
15918750960,3770784,35489,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15922524400,3784096,35491,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15926311152,5109984,35507,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15931423984,5723136,35530,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
15937149168,243361312,35533,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
16180513040,2178944,35540,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
16182694224,1952,35543,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
16182698320,3200,35557,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
16182704368,1536,35572,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
16182708560,49120,35595,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16182760752,2770976,35607,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
16185534736,1920,35616,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
16185538896,8428352,35645,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
16193968624,5499328,35659,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
16199469424,3676128,35679,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
16203147504,2496,35694,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16203151600,2080,35712,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16203155664,47648,35752,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16203204880,4130240,35760,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
16207336816,1952,35763,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
16207340784,2429920,35766,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
16209773776,1888,35777,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
16209778000,34611488,35806,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
16244391184,139936,35847,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16244532464,12928448,35855,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
16257463504,2208,35858,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
16257467632,10450016,35861,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
16267919600,1920,35872,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
16267923792,832,35887,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16267926288,15834848,35908,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
16283762992,5478816,35921,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
16289244400,1952,35929,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
16289248496,1408,35940,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
16289251600,1472,35951,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
16289254608,3456,35965,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
16289260784,1376,35977,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
16289264880,4800,35989,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
16289270992,1088,36002,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
16289275120,1600,36014,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
16289279216,1504,36027,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
16289283312,1280,36038,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16289287408,26624,36059,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
16289316080,3510240,36082,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
16292829520,2432,36097,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16292833488,2048,36115,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16292837616,48736,36155,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16292887760,4145728,36163,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
16297036048,3808,36166,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
16297042192,2516704,36169,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
16299560208,1888,36180,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
16299564272,25347424,36209,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
16324914416,12396480,36226,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
16337312656,768,36240,,,,,,,,,,0.000,291.667,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
16337314512,2354336,36243,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
16339671280,3785536,36276,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
16343459056,3750080,36278,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
16347211952,5090848,36294,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
16352304368,5793280,36317,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
16358099248,242750816,36320,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
16600852144,2176192,36327,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
16603030864,1984,36330,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
16603034864,3008,36344,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
16603041008,1536,36359,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
16603045072,48992,36382,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16603096272,2770880,36394,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
16605869296,2016,36403,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
16605873392,8377760,36432,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
16614253808,5504512,36446,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
16619759856,3685984,36466,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
16623447376,2432,36481,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16623451344,2176,36499,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16623455568,47456,36539,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16623504624,3658144,36547,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
16627165424,1952,36550,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
16627169520,2434048,36553,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
16629605616,1920,36564,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
16629609712,34581888,36593,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
16664193296,141376,36634,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16664336592,12743040,36642,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
16677082320,2208,36645,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
16677086416,10444640,36648,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
16687532336,1952,36659,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
16687536368,928,36674,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16687538576,15807712,36695,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
16703349008,5580608,36708,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
16708931824,1920,36716,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
16708936080,1440,36727,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
16708940080,1312,36738,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
16708946000,22080,36752,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
16709056560,1344,36764,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
16709059824,5952,36776,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
16709068016,1120,36789,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
16709072240,6144,36801,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
16709210704,1952,36814,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
16709215472,1248,36825,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16709219536,38432,36846,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
16709259504,3551520,36869,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
16712812752,5216,36884,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16712820944,2432,36902,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
16712825072,49920,36942,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
16712877296,4215904,36950,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
16717096144,4128,36953,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
16717102320,2539872,36956,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
16719643984,1856,36967,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
16719647952,24050080,36996,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
16743699728,12171808,37013,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
16755873680,416,37027,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
16755875344,2411200,37030,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
16758304432,3849120,37063,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
16762155248,3791648,37065,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
16765949136,5113664,37081,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
16771065168,5721440,37104,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
16776789232,243282592,37107,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
17020074352,2181056,37114,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
17022257456,1952,37117,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
17022261488,2752,37131,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17022265680,1536,37146,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17022268848,47200,37169,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17022317776,2768704,37181,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
17025088048,1888,37190,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17025091824,7918304,37219,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17033012496,5477536,37233,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
17038493008,3490304,37253,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
17041984752,2592,37268,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17041988848,1952,37286,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17041992944,47552,37326,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17042043088,3416192,37334,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
17045461200,2048,37337,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17045465328,2434656,37340,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
17047902576,1920,37351,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17047906512,34610912,37380,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17082518832,140352,37421,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17082661136,11964480,37429,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
17094628656,2208,37432,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17094632656,10482784,37435,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
17105117776,1920,37446,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17105121552,864,37461,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17105123696,15782656,37482,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17120908528,5489472,37495,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
17126400240,1984,37503,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17126404336,1440,37514,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17126408848,3520,37525,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17126414544,3456,37539,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17126420816,1344,37551,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17126424816,4448,37563,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
17126431056,1120,37576,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17126435184,1632,37588,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
17126439152,1472,37601,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17126443376,1248,37612,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17126447440,26944,37633,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
17126476080,3512928,37656,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
17129991504,2400,37671,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17129995504,2080,37689,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17129999824,47872,37729,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17130050128,3447936,37737,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
17133499728,3616,37740,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17133505872,2833664,37743,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
17136341232,1888,37754,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17136345424,25380896,37783,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17161729232,11852064,37800,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
17173582736,448,37814,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
17173584624,2479616,37817,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
17176067312,3863872,37850,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
17179932912,3950816,37852,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
17183885520,5209696,37868,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
17189096784,5707104,37891,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
17194805488,243530720,37894,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
17438338320,2179584,37901,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
17440520400,2048,37904,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
17440524528,2752,37918,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17440528592,1536,37933,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17440532688,45664,37956,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17440580848,2772320,37968,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
17443354864,1888,37977,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17443358960,7933664,38006,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17451293936,5481440,38020,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
17456777424,3490848,38040,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
17460270320,2304,38055,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17460274416,2176,38073,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17460278544,46048,38113,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17460327760,3426208,38121,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
17463757008,1952,38124,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17463761136,2438752,38127,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
17466202352,1888,38138,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17466206448,34688000,38167,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17500896528,140192,38208,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17501039856,12220320,38216,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
17513261552,2016,38219,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17513265424,10986432,38222,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
17524253936,1952,38233,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17524258032,896,38248,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17524260208,15775616,38269,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17540037872,5476288,38282,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
17545516272,1952,38290,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17545520368,1440,38301,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17545524464,1280,38312,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17545528560,3040,38326,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17545534704,1440,38338,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17545537392,4672,38350,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
17545544016,1088,38363,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17545548016,1600,38375,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
17545552112,1504,38388,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17545556336,1248,38399,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17545560304,25984,38420,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
17545588976,3520064,38443,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
17549110512,2688,38458,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17549114608,2016,38476,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17549118672,49440,38516,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17549169904,3445856,38524,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
17552617808,3680,38527,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17552623824,2677920,38530,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
17555303760,1888,38541,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17555307856,25633504,38570,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17580943632,11815744,38587,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
17592761232,416,38601,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
17592762832,2383392,38604,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
17595147504,3935872,38637,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
17599085808,3919136,38639,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
17603006704,5242336,38655,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
17608250704,5685216,38678,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
17613938928,243162464,38681,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
17857103120,2174272,38688,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
17859280144,1920,38691,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
17859284208,2624,38705,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17859288272,1536,38720,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17859292400,45184,38743,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17859339472,2773504,38755,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
17862114256,1920,38764,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17862118672,7919360,38793,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17870040464,5467552,38807,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
17875510480,3489920,38827,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
17879003440,2400,38842,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17879007536,1952,38860,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17879011568,47104,38900,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17879061744,3419008,38908,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
17882482064,1984,38911,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17882486096,2434304,38914,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
17884923120,1920,38925,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17884927216,34663328,38954,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17919591888,140288,38995,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17919735024,11913600,39003,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
17931650256,2176,39006,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17931654384,10436992,39009,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
17942094064,1920,39020,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17942098192,2720,39035,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17942102288,15837408,39056,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17957941488,5479968,39069,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
17963422960,1920,39077,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17963427152,1408,39088,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17963431248,1312,39099,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17963435216,3104,39113,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17963441488,1376,39125,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
17963445488,4480,39137,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
17963451472,1088,39150,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
17963454352,1632,39162,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
17963458800,1472,39175,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
17963462896,1280,39186,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17963466992,25376,39207,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
17963493648,3513056,39230,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
17967008016,2592,39245,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17967012080,1984,39263,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
17967016176,48640,39303,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
17967067376,3444384,39311,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
17970514288,3840,39314,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
17970520336,2452608,39317,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
17972974864,1856,39328,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
17972978928,25941344,39357,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
17998922992,11825184,39374,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
18010749936,448,39388,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
18010751280,2374368,39391,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
18013127920,3870592,39424,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18017000720,3897536,39426,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18020900080,5456384,39442,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18026358000,5682784,39465,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18032043216,243159328,39468,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
18275205520,2185664,39475,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
18277393616,1984,39478,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
18277397744,2976,39492,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
18277402000,1536,39507,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
18277405936,50944,39530,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18277458160,2771296,39542,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
18280232144,1856,39551,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
18280236272,7895424,39580,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
18288133392,5527488,39594,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
18293662928,3679488,39614,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
18297345264,2752,39629,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18297349328,2240,39647,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18297353424,48032,39687,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18297403984,4093952,39695,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
18301500688,1920,39698,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
18301504784,2438080,39701,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
18303944944,1888,39712,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
18303949040,34779168,39741,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
18338731280,141376,39782,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18338875696,11911488,39790,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
18350788816,2272,39793,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
18350792944,10935616,39796,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
18361731312,7584,39807,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
18361754960,8672,39822,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18361766224,16098720,39843,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
18377867504,5481952,39856,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
18383352112,1952,39864,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
18383356144,1440,39875,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
18383360304,1312,39886,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
18383364336,3648,39900,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
18383370480,1312,39912,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
18383374544,4896,39924,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
18383380720,1120,39937,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
18383383440,1632,39949,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
18383387888,1472,39962,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
18383392080,1472,39973,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18383396176,26816,39994,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
18383424752,3515712,40017,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
18386943312,2400,40032,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18386947376,2048,40050,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18386951408,49056,40090,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18387002608,3425184,40098,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
18390430032,3808,40101,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
18390436048,2433728,40104,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
18392871280,1856,40115,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
18392875248,25886688,40144,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
18418765040,11822752,40161,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
18430588816,448,40175,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
18430590416,2358144,40178,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
18432951536,3870208,40211,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18436824304,3901312,40213,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18440727888,5449536,40229,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18446178800,5678048,40252,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18451859696,244356320,40255,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
18696218928,2190848,40262,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
18698411216,1952,40265,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
18698415440,2944,40279,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
18698421488,1536,40294,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
18698425584,49888,40317,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18698477904,2770112,40329,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
18701249872,1888,40338,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
18701253616,8399808,40367,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
18709654800,5519104,40381,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
18715176304,3799840,40401,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
18718978288,4928,40416,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18718984592,2112,40434,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18718988528,46272,40474,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18719036624,3862208,40482,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
18722901232,1952,40485,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
18722905328,2437024,40488,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
18725343664,1888,40499,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
18725347600,34519392,40528,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
18759869744,140416,40569,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18760013040,11940640,40577,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
18771954928,2400,40580,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
18771959120,11079968,40583,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
18783040752,2496,40594,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
18783044848,2144,40609,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18783048944,15949184,40630,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
18798999792,5503712,40643,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
18804504816,1920,40651,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
18804509040,1440,40662,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
18804513104,1312,40673,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
18804517104,3456,40687,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
18804523344,1440,40699,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
18804527344,4800,40711,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
18804533712,1088,40724,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
18804537680,1632,40736,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
18804541648,1504,40749,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
18804544560,1280,40760,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18804548176,25920,40781,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
18804576560,3549088,40804,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
18808128752,2400,40819,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18808132912,2176,40837,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
18808136944,48352,40877,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
18808187120,3443040,40885,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
18811631824,3840,40888,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
18811638000,2447008,40891,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
18814087408,1920,40902,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
18814091600,25021376,40931,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
18839116144,11827520,40948,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
18850945936,448,40962,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
18850947568,2414464,40965,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
18853365072,3933056,40998,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18857301200,3838272,41000,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18861141200,5452544,41016,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18866595056,5752064,41039,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
18872348944,242752640,41042,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
19115104560,2175072,41049,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
19117281520,1952,41052,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
19117285712,2976,41066,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19117291728,1632,41081,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
19117295824,49280,41104,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19117347024,2774432,41116,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
19120123120,1888,41125,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19120127216,7918048,41154,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19128047888,5545440,41168,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
19133595888,3704416,41188,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
19137301744,6656,41203,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19137309936,5088,41221,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19137318096,75392,41261,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19137394928,3973760,41269,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
19141371120,1952,41272,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
19141375216,2437984,41275,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
19143814576,1888,41286,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19143818480,34589248,41315,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19178409232,140128,41356,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19178551536,11918368,41364,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
19190471888,2176,41367,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
19190475984,10877600,41370,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
19201355088,2048,41381,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19201359088,864,41396,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19201361232,16805760,41417,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19218169072,5480000,41430,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
19223650544,1952,41438,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19223654640,1408,41449,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
19223658736,3744,41460,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19223664848,3456,41474,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
19223671056,1344,41486,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19223675120,4448,41498,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
19223681264,1088,41511,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
19223685328,1664,41523,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
19223689456,1472,41536,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
19223693552,1280,41547,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19223696560,27104,41568,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
19223726320,3516032,41591,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
19227244784,2400,41606,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19227248880,2016,41624,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19227253008,48160,41664,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19227303248,3527936,41672,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
19230833904,3616,41675,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
19230840048,2491008,41678,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
19233332464,1856,41689,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19233336560,26017728,41718,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19259356400,11800000,41735,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
19271158704,416,41749,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
19271160272,2401824,41752,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
19273563376,3883328,41785,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
19277448432,3884192,41787,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
19281334928,5423936,41803,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
19286760688,5687360,41826,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
19292451056,241628288,41829,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
19534082320,2169824,41836,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
19536254192,1952,41839,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
19536258288,2944,41853,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19536262512,1536,41868,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
19536266480,46432,41891,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19536315600,2777056,41903,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
19539095792,1920,41912,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19539099888,7852032,41941,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19546954992,5480384,41955,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
19552438512,3482688,41975,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
19555924304,2336,41990,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19555928496,2048,42008,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19555932496,47680,42048,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19555981552,3432384,42056,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
19559416048,1952,42059,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
19559420208,2433024,42062,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
19561855312,1888,42073,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19561859440,34484864,42102,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19596345616,140064,42143,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19596487920,12103776,42151,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
19608593712,2048,42154,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
19608597744,10852384,42157,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
19619452144,2048,42168,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19619456240,832,42183,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19619458480,15941600,42204,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19635402992,5533120,42217,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
19640937712,2560,42225,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19640941808,1440,42236,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
19640946160,1312,42247,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19640950032,9408,42261,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
19640962320,1408,42273,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19640966384,4544,42285,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
19640972624,1088,42298,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
19640976624,11488,42310,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
19640990928,10656,42323,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
19641003248,1760,42334,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19641007472,34688,42355,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
19641045328,3517632,42378,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
19644564720,2368,42393,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19644568912,2080,42411,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19644572912,50432,42451,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19644625136,3448832,42459,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
19648076016,3360,42462,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
19648082160,2457216,42465,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
19650541936,1888,42476,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19650545904,25767264,42505,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19676315888,11886144,42522,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
19688203184,416,42536,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
19688204592,2350784,42539,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
19690556688,3769696,42572,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
19694329168,4067488,42574,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
19698399568,5439008,42590,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
19703841104,5761760,42613,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
19709604144,241841472,42616,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
19951447312,2185280,42623,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
19953635664,1952,42626,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
19953639664,2976,42640,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
19953644016,1536,42655,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
19953647952,47104,42678,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19953698000,2773376,42690,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
19956474096,1856,42699,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19956478192,7878016,42728,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
19964358928,5479584,42742,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
19969840464,3491872,42762,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
19973335248,2624,42777,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19973339344,2048,42795,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
19973343440,46688,42835,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
19973391600,4083584,42843,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
19977477392,2336,42846,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
19977481552,2549760,42849,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
19980033264,1888,42860,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
19980037360,34247328,42889,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20014287120,146144,42930,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20014434512,12350112,42938,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
20026787056,2208,42941,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
20026791152,10440128,42944,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
20037232912,1920,42955,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20037236976,3008,42970,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20037243120,14879520,42991,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20052124912,5635680,43004,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
20057762000,1984,43012,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20057766128,1440,43023,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20057770256,1440,43034,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20057774320,3072,43048,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20057780464,1792,43060,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20057784688,4544,43072,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
20057790704,7584,43085,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20057800944,1952,43097,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
20057805008,2016,43110,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20057810320,1248,43121,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20057813232,25632,43142,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
20057841904,3676192,43165,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
20061521136,2400,43180,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20061525328,2240,43198,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20061529488,49472,43238,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20061581552,3885408,43246,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
20065468624,3648,43249,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
20065474896,2445824,43252,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
20067923184,1888,43263,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20067927248,25171104,43292,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20093101264,12445568,43309,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
20105547888,416,43323,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
20105549488,2355424,43326,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
20107906704,3719296,43359,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
20111627600,3800608,43361,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
20115430608,5700544,43377,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
20121133392,5727296,43400,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
20126863664,242556320,43403,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
20369421616,2181824,43410,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
20371604816,1952,43413,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
20371608816,2976,43427,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20371613072,1504,43442,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20371616976,53440,43465,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20371673328,2776864,43477,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
20374451472,1888,43486,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20374455536,7782656,43515,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20382240016,5478016,43529,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
20387720432,3498272,43549,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
20391220464,2336,43564,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20391224656,2208,43582,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20391228496,46560,43622,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20391277808,3830208,43630,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
20395110608,2208,43633,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
20395114736,2680512,43636,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
20397796560,1888,43647,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20397801008,33924352,43676,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20431726896,141056,43717,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20431869200,11924128,43725,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
20443795792,1952,43728,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
20443799824,10514592,43731,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
20454316304,1920,43742,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20454320368,864,43757,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20454322608,15693888,43778,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20470018288,5522272,43791,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
20475542768,2080,43799,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20475546864,1408,43810,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20475550960,1312,43821,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20475555056,10752,43835,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20475567344,1376,43847,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20475571440,4672,43859,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
20475577648,1312,43872,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20475581712,1632,43884,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
20475585776,1600,43897,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20475590064,1312,43908,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20475593968,31712,43929,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
20475628752,3834528,43952,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
20479465808,2688,43967,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20479470800,2144,43985,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20479474896,49376,44025,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20479526128,3812864,44033,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
20483340528,3968,44036,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
20483346672,2433376,44039,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
20485782768,1888,44050,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20485786832,25033120,44079,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20510821616,12471040,44096,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
20523294608,416,44110,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
20523295920,2368672,44113,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
20525667568,3739904,44146,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
20529409264,3745888,44148,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
20533157104,5564416,44164,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
20538723632,5812416,44187,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
20544537840,240774464,44190,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
20785314288,2176096,44197,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
20787493328,1920,44200,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
20787497296,3328,44214,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20787503312,1664,44229,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20787507536,47328,44252,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20787557584,2772384,44264,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
20790331760,1856,44273,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20790335760,7745440,44302,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20798084368,5472320,44316,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
20803558640,3491552,44336,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
20807052528,2336,44351,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20807056816,2112,44369,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20807060720,47616,44409,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20807109840,3424352,44417,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
20810536176,1920,44420,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
20810540272,2456224,44423,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
20812997872,2112,44434,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20813001968,34352224,44463,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20847357200,142208,44504,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20847501648,12776704,44512,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
20860280016,2208,44515,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
20860284240,10460288,44518,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
20870746352,1920,44529,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20870750448,832,44544,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
20870752592,15672768,44565,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
20886427888,5476768,44578,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
20891906288,1952,44586,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20891910384,1440,44597,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20891914480,3648,44608,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20891920592,3456,44622,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20891926768,1344,44634,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20891930864,4512,44646,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
20891937008,1120,44659,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20891941136,1632,44671,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
20891945232,1760,44684,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20891949296,1280,44695,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20891953360,4640,44716,336,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
20891959536,3591232,44742,37296,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
20895552752,1952,44756,6,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20895556848,5457248,44767,391608,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20901015792,6822208,44781,783216,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20907840752,45056,44807,414,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
20907887856,6688,44821,6,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20907896176,51296,44832,4347,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20907950416,61632,44846,8694,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20908014832,4410976,44867,2,583,1,8,16,1,80,0.009,0.000,,,,,NVIDIA GB10 (0),1,,7,"void magma_sgemmEx_kernel<float, float, float, (bool)1, (bool)0, (int)6, (int)4, (int)6, (int)3, (int)4>(int, int, int, Tensor, int, Tensor, int, Tensor, int, Tensor, int, int, int, const T1 *, const T1 *, T1, T1, int, cublasLtEpilogue_t, int, const void *, long)"
20912428368,99968,44891,13,1,3,128,1,1,80,0.000,0.026,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2<cutlass_80_simt_sgemm_128x32_8x5_tn_align1>(T1::Params)
20912530672,4000,44894,1,26,1,32,16,1,46,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, float, float, float, float, (bool)0, float, float, float, (bool)1, (bool)1, (bool)0, (bool)0>(cublasLt::cublasSplitKParams<T6>, const T4 *, const T10 *, T9 *, T5 *, const T6 *, const T6 *, const T11 *, const T4 *, T11 *, void *, long, T6 *, int *, T6 *, T6 *, const T6 *, const T6 *, const T6 *, const T6 *, const T6 *)"
20912536816,59776,44908,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912599248,62272,44923,6993,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20912662832,1280,44938,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912666864,100448,44953,13986,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20912769392,86400,44965,3497,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912857424,1856,44979,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912861424,1472,44994,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912865616,5600,45006,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20912873712,1312,45017,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912877904,1536,45028,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912882064,1216,45039,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912886000,1376,45053,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912890320,1664,45065,13,1,1,128,1,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 9)]::operator ()() const::[lambda(c10::BFloat16) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
20912894288,3360,45076,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20912900336,4128,45087,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<c10::BFloat16>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20912906576,2272,45101,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
20912910576,72864,45113,13986,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20912985552,162432,45124,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20913149232,2240,45135,52,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20913153264,2016,45146,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20913157072,2848,45157,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::AUnaryFunctor<float, float, bool, at::native::<unnamed>::CompareEqFunctor<float>>, std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)"
20913166480,5664,45163,,,,,,,,,,0.000,0.177,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host]
20913239056,159936,45174,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20913400752,83872,45185,13986,1,1,128,1,1,20,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20913486800,1184,45196,1,1,1,128,1,1,30,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)"
20913489008,88064,45207,13986,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20913578192,172960,45218,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20913753040,1824,45229,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::AUnaryFunctor<float, float, bool, at::native::<unnamed>::CompareEqFunctor<float>>, std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)"
20913758992,1792,45235,,,,,,,,,,0.000,0.558,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host]
20913773072,2336,45246,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20913782096,3904,45257,52,1,1,128,1,1,20,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20913795024,4928,45268,1,1,1,128,1,1,30,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)"
20913801104,27520,45279,52,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
20913830896,1600,45290,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
20913858544,1120,45301,13,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"