2805 lines
833 KiB
CSV
2805 lines
833 KiB
CSV
|
|
Start (ns),Duration (ns),CorrId,GrdX,GrdY,GrdZ,BlkX,BlkY,BlkZ,Reg/Trd,StcSMem (MB),DymSMem (MB),Bytes (MB),Throughput (MB/s),SrcMemKd,DstMemKd,Device,Ctx,GreenCtx,Strm,Name
|
||
|
|
6687024,129408,4655,,,,,,,,,,14.322,110663.498,Device,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Device]
|
||
|
|
6818800,4288,4667,,,,,,,,,,0.053,12358.158,Device,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Device]
|
||
|
|
7204912,1888,4680,,,,,,,,,,0.000,4.237,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device]
|
||
|
|
7275376,2144,4697,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7292688,2816,4708,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7305872,3392,4719,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7313232,1088,4730,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7320496,1088,4741,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7326704,1056,4752,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7332624,2080,4763,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7339504,1824,4774,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7357872,2752,4788,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7370736,5728,4800,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl<at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7381104,1120,4811,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7392976,1984,4822,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7417616,38432,4839,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7459632,48832,4854,6993,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 12)]::operator ()() const::[lambda(c10::BFloat16) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7510000,71520,4869,6993,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7647600,4915936,4896,2336,3,1,256,1,1,212,0.000,0.049,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2<cutlass_80_simt_sgemm_256x128_8x4_tn_align1>(T1::Params)
|
||
|
|
12565488,5035136,4909,195804,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17601648,3808,4927,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17607760,44096,4953,56,6,1,128,1,1,130,0.000,0.031,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2<cutlass_80_simt_sgemm_128x64_8x5_tt_align1>(T1::Params)
|
||
|
|
17653744,30368,4966,2174,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17685488,3744,4978,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17691728,1792,4989,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17696080,1344,5000,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17700080,2048,5011,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17704272,1536,5022,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17708368,1280,5033,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17712368,1376,5044,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17716560,2080,5055,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17720400,2848,5066,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17729712,6688,5072,,,,,,,,,,0.000,0.598,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host]
|
||
|
|
17749968,1024,5083,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17754544,608,5089,,,,,,,,,,0.000,6.579,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host]
|
||
|
|
17778576,960,5102,,,,,,,,,,0.000,4.167,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device]
|
||
|
|
17807664,3766816,5114,1536,3,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::CatArrayBatchedCopy_alignedK_contig<at::native::<unnamed>::OpaqueType<(unsigned int)2>, unsigned int, (int)2, (int)128, (int)1, (int)8>(T1 *, at::native::<unnamed>::CatArrInputTensorMetadata<T1, T2, T4, T5>, at::native::<unnamed>::TensorSizeStride<T2, (unsigned int)4>, int, T2)"
|
||
|
|
26302704,20288,5127,,,,,,,,,,0.907,44727.718,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device]
|
||
|
|
26372208,8192,5141,222,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
26413968,27680,5153,7090,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
26449776,52384,5165,96,3,1,512,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::CatArrayBatchedCopy<at::native::<unnamed>::OpaqueType<(unsigned int)4>, unsigned int, (int)2, (int)64, (int)64>(T1 *, at::native::<unnamed>::CatArrInputTensorMetadata<T1, T2, T4, T5>, at::native::<unnamed>::TensorSizeStride<T2, (unsigned int)4>, int, T2)"
|
||
|
|
26503152,31872,5176,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::cos_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
26536016,39456,5187,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::sin_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
26577904,39712,5198,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
26618864,873696,5210,1536,4,1,128,1,1,37,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::CatArrayBatchedCopy_alignedK_contig<at::native::<unnamed>::OpaqueType<(unsigned int)4>, unsigned int, (int)3, (int)128, (int)1, (int)16>(T1 *, at::native::<unnamed>::CatArrInputTensorMetadata<T1, T2, T4, T5>, at::native::<unnamed>::TensorSizeStride<T2, (unsigned int)4>, int, T2)"
|
||
|
|
27495408,217120,5224,7090,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
27714544,2240,5236,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
27718736,1536,5247,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
27722736,3424,5258,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
27728848,3168,5272,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
27733072,3136,5284,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
27737488,4736,5296,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
27743504,1088,5309,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
27747568,1632,5321,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
27751536,1504,5334,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
27755760,2016,5345,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
27759856,30048,5366,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
27792368,3544672,5389,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
31338480,2464,5404,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
31342576,1984,5422,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
31346672,49568,5462,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
31398896,3452992,5470,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
34853840,2912,5473,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
34857968,2466752,5476,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
37326064,1920,5487,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
37330160,24059776,5516,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
61391216,11842304,5533,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
73250640,448,5547,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
73252304,2395136,5550,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
75649264,4001984,5583,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
79653872,3999008,5585,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
83654864,5297792,5601,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
88955088,5731328,5624,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
94689520,240315488,5627,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
335006992,2176224,5634,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
337185008,1952,5637,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
337189104,1376,5651,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
337193200,1504,5666,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
337197296,40608,5689,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
337240272,2775488,5701,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
340017392,1888,5710,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
340021488,7702240,5739,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
347725040,5461056,5753,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
353188080,3717632,5773,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
356907248,5856,5788,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
356915440,3776,5806,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
356921584,60928,5846,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
356984048,4091840,5854,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
361077968,1952,5857,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
361082096,2437024,5860,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
363520432,1920,5871,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
363524336,34563520,5900,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
398089488,140224,5941,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
398231792,11923520,5949,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
410157296,2464,5952,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
410161392,10756480,5955,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
420920560,1920,5966,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
420924656,3040,5981,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
420929776,16202336,6002,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
437133584,5455232,6015,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
442590448,1984,6023,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
442594544,1440,6034,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
442598640,1376,6045,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
442602736,3168,6059,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
442608848,1344,6071,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
442613040,4608,6083,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
442618992,1120,6096,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
442623216,1632,6108,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
442627312,1472,6121,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
442631408,1248,6132,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
442635504,29472,6153,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
442666256,3479584,6176,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
446148848,2368,6191,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
446152944,2112,6209,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
446157040,48416,6249,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
446208240,3449888,6257,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
449660144,4032,6260,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
449666288,2435424,6263,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
452104432,1888,6274,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
452108528,25770432,6303,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
477881584,11819968,6320,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
489703312,416,6334,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
489704624,2346688,6337,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
492053744,3809824,6370,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
495865072,3947776,6372,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
499815664,5406624,6388,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
505224432,5664096,6411,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
510891248,240444576,6414,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
751338768,2172448,6421,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
753512656,1952,6424,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
753516784,2912,6438,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
753520976,1536,6453,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
753525008,46304,6476,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
753573072,2749568,6488,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
756324592,1920,6497,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
756328656,7743936,6526,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
764074288,5461696,6540,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
769538256,3462944,6560,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
773002480,2304,6575,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
773006576,1984,6593,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
773010672,46624,6633,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
773058800,3966080,6641,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
777027824,1952,6644,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
777031920,2729952,6647,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
779763952,1856,6658,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
779768048,33135360,6687,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
812905712,186080,6728,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
813094096,12695904,6736,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
825791728,2176,6739,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
825795920,10475328,6742,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
836272528,2176,6753,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
836276464,832,6768,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
836278352,15614944,6789,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
851894576,5572000,6802,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
857469168,2336,6810,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
857473264,3104,6821,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
857479408,3264,6832,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
857485552,11680,6846,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
857499856,6144,6858,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
857509360,5600,6870,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
857516272,2368,6883,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
857520368,3360,6895,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
857526512,2944,6908,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
857530736,8192,6919,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
857540848,56608,6940,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
857600240,3673856,6963,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
861275376,2464,6978,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
861279440,2048,6996,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
861283536,48960,7036,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
861334736,3813056,7044,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
865150160,3776,7047,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
865156304,2434944,7050,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
867593456,1888,7061,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
867597584,23937632,7090,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
891537648,12480864,7107,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
904019856,416,7121,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
904021456,2345600,7124,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
906369264,3709440,7157,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
910080208,3724512,7159,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
913807568,5534656,7175,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
919343504,5795488,7198,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
925141232,239349632,7201,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
1164493200,2170784,7208,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
1166665936,1920,7211,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
1166670064,2912,7225,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1166674288,1504,7240,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
1166678288,47104,7263,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1166728400,2757664,7275,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
1169489136,1952,7284,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
1169493232,7799744,7313,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
1177295120,5451936,7327,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
1182748880,3475680,7347,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
1186227440,2400,7362,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1186231504,2080,7380,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1186235600,46016,7420,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1186282896,3462240,7428,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
1189746928,1984,7431,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
1189750992,2452064,7434,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
1192204528,1888,7445,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
1192208624,34645696,7474,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
1226856784,139968,7515,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1226999024,12761440,7523,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
1239763184,1920,7526,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
1239767280,10456096,7529,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
1250225392,1952,7540,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
1250229456,864,7555,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1250231632,15792128,7576,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
1266026768,5824160,7589,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
1271852272,1952,7597,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1271856368,1408,7608,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1271860432,2848,7619,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1271864560,3200,7633,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
1271870704,1376,7645,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1271874800,4416,7657,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
1271880912,1120,7670,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1271885040,1600,7682,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
1271889136,1472,7695,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
1271893200,1280,7706,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1271897392,26016,7727,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
1271925968,3545984,7750,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
1275473232,2496,7765,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1275477232,2112,7783,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1275481328,47552,7823,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1275530768,4259520,7831,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
1279793424,3712,7834,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
1279799536,2444704,7837,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
1282245872,1824,7848,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
1282249936,24882560,7877,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
1307135248,12475168,7894,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
1319611344,416,7908,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
1319612656,2339296,7911,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
1321954544,3722080,7944,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
1325678832,3736032,7946,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
1329416432,5150368,7962,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
1334569200,5873152,7985,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
1340443888,238904576,7988,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
1579351312,2158912,7995,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
1581512912,1984,7998,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
1581517040,2880,8012,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1581521200,1536,8027,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
1581525136,54208,8050,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1581580624,2747456,8062,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
1584329936,1920,8071,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
1584334064,7804864,8100,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
1592141072,5612512,8114,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
1597755632,3661824,8134,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
1601419504,2400,8149,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1601423568,2112,8167,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1601427696,46240,8207,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1601475824,3859104,8215,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
1605337328,1952,8218,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
1605341456,2438848,8221,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
1607782640,1920,8232,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
1607786736,34043744,8261,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
1641831824,139840,8302,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1641972944,12366720,8310,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
1654341872,1920,8313,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
1654345968,10667840,8316,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
1665016048,1984,8327,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
1665020144,2560,8342,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1665024240,15704512,8363,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
1680730352,5454720,8376,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
1686186480,1952,8384,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1686189744,1440,8395,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1686193392,1312,8406,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1686197488,3264,8420,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
1686203600,1344,8432,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1686207728,4640,8444,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
1686213872,1088,8457,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1686217968,1632,8469,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
1686222160,1504,8482,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
1686226160,1280,8493,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1686230256,27264,8514,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
1686258928,3493056,8537,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
1689753872,2464,8552,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1689757936,2016,8570,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
1689762032,47456,8610,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1689811152,3918208,8618,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
1693731056,3392,8621,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
1693737200,2845728,8624,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
1696585968,1856,8635,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
1696590096,25301536,8664,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
1721894096,11901664,8681,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
1733796816,384,8695,,,,,,,,,,0.000,583.333,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
1733798384,2551328,8698,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
1736351984,3738560,8731,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
1740092624,3951712,8733,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
1744046320,5119744,8749,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
1749167344,5697824,8772,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
1754867920,240193088,8775,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
1995062544,2167328,8782,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
1997232368,1920,8785,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
1997236432,2944,8799,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
1997240656,1504,8814,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
1997244624,47296,8837,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
1997293808,2750016,8849,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
2000046288,1952,8858,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2000050416,7806304,8887,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2007859472,5472288,8901,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
2013334768,3708736,8921,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
2017045712,6624,8936,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2017053936,2144,8954,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2017058000,47936,8994,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2017107216,4063680,9002,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
2021173488,1952,9005,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2021177584,2438432,9008,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
2023617776,1920,9019,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2023621872,34049568,9048,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2057673968,139872,9089,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2057816272,11888096,9097,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
2069705936,1984,9100,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2069710064,10741088,9103,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
2080453872,1952,9114,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2080457968,2496,9129,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2080462064,16236448,9150,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2096700624,5467584,9163,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
2102170832,1952,9171,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2102174960,1440,9182,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2102179056,1312,9193,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2102181904,3456,9207,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
2102187248,1344,9219,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2102191280,4608,9231,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
2102197488,1088,9244,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2102201520,1632,9256,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
2102205680,1728,9269,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
2102209776,1280,9280,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2102213840,26368,9301,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
2102242544,3495232,9324,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
2105740528,2496,9339,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2105744624,1984,9357,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2105748720,47232,9397,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2105797840,3515456,9405,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
2109316304,3360,9408,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2109322480,2434016,9411,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
2111758576,1888,9422,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2111762672,25849184,9451,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2137614608,11818912,9468,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
2149435280,416,9482,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
2149436720,2354816,9485,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
2151792912,3750336,9518,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2155544784,4064384,9520,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2159612112,5416480,9536,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2165031152,5667744,9559,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2170700176,242080160,9562,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
2412782864,2159488,9569,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
2414944496,1952,9572,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
2414948592,2944,9586,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2414952560,1504,9601,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
2414955312,49536,9624,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2415006928,2752224,9636,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
2417760496,1888,9645,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2417764560,7814080,9674,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2425580816,5456192,9688,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
2431038672,3483520,9708,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
2434524624,6336,9723,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2434532592,36384,9741,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2434571504,54208,9781,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2434628848,4283872,9789,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
2438914256,1952,9792,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2438918384,2440160,9795,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
2441360624,1888,9806,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2441364720,33763776,9835,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2475131248,139488,9876,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2475272432,12292416,9884,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
2487567568,2176,9887,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2487571728,10786304,9890,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
2498360560,1920,9901,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2498364656,832,9916,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2498366768,15820672,9937,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2514190576,5490432,9950,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
2519683344,2272,9958,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2519687440,1408,9969,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2519691504,3712,9980,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2519697648,8064,9994,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
2519707856,1568,10006,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2519711920,8096,10018,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
2519722224,6880,10031,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2519730384,1632,10043,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
2519734512,14048,10056,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
2519750896,3648,10067,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2519757040,93376,10088,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
2519853264,3592384,10111,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
2523448560,2912,10126,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2523452720,2240,10144,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2523456752,47712,10184,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2523505904,3622240,10192,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
2527129808,4000,10195,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2527135984,2437120,10198,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
2529576208,1856,10209,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2529580272,25405312,10238,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2554986864,12071296,10255,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
2567060336,448,10269,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
2567062000,2347552,10272,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
2569410832,3711616,10305,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2573123824,4098912,10307,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2577223984,5356480,10323,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2582582512,5691456,10346,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2588277008,239531264,10349,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
2827810064,2158528,10356,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
2829970640,1984,10359,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
2829974768,2720,10373,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2829978864,1504,10388,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
2829982928,46080,10411,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2830032080,2749120,10423,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
2832783600,1920,10432,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2832787696,7819616,10461,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2840610032,5449728,10475,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
2846062832,3479168,10495,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
2849543408,2336,10510,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2849547504,1984,10528,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2849551600,46464,10568,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2849600720,3413024,10576,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
2853015792,1984,10579,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2853019888,2757152,10582,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
2855779536,1856,10593,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2855783632,33791744,10622,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2889576720,141536,10663,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2889721072,12811136,10671,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
2902534352,2208,10674,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2902538480,10362432,10677,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
2912903408,1952,10688,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2912907472,864,10703,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2912909616,15744352,10724,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2928656624,5465664,10737,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
2934124752,1984,10745,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2934128912,1408,10756,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2934132976,3680,10767,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2934139120,3488,10781,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
2934145136,1344,10793,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2934147728,4608,10805,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
2934154352,1120,10818,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
2934158576,1632,10830,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
2934162672,1472,10843,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
2934166768,1280,10854,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2934170864,25920,10875,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
2934199536,3495488,10898,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
2937696496,2528,10913,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2937700560,1920,10931,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
2937704656,48864,10971,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
2937754864,3453632,10979,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
2941210832,3808,10982,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
2941217008,2435392,10985,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
2943654160,1888,10996,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
2943658224,24871360,11025,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
2968532176,12473728,11042,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
2981008304,416,11056,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
2981009616,2342880,11059,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
2983354608,3711520,11092,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2987067632,3730304,11094,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2990800112,5451680,11110,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
2996253392,5726816,11133,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3001983216,240209920,11136,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
3242196240,2180576,11143,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
3244379376,1984,11146,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
3244383504,2752,11160,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3244387536,1504,11175,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
3244391664,45632,11198,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3244438736,2899360,11210,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
3247340784,2016,11219,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
3247344848,8260608,11248,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
3255608560,5606272,11262,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
3261217040,3532864,11282,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
3264751824,2528,11297,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3264755920,1984,11315,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3264760048,46304,11355,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3264808144,3517760,11363,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
3268327664,2144,11366,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
3268331760,2417696,11369,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
3270751472,1888,11380,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
3270755568,34255616,11409,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
3305013552,141536,11450,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3305157872,12819968,11458,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
3317979344,2464,11461,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
3317983472,10346848,11464,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
3328333040,2144,11475,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
3328337136,864,11490,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3328339280,15735488,11511,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
3344077040,5454016,11524,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
3349532912,2016,11532,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3349537008,1408,11543,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3349541104,1312,11554,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3349545200,3456,11568,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
3349551344,1312,11580,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3349555568,4224,11592,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
3349561520,1088,11605,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3349565680,1632,11617,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
3349569744,1504,11630,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
3349573872,1280,11641,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3349577936,26720,11662,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
3349606640,3496640,11685,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
3353104752,2400,11700,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3353108720,2112,11718,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3353112816,47680,11758,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3353162960,3506560,11766,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
3356672208,3680,11769,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
3356678352,2438752,11772,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
3359119600,1888,11783,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
3359123664,24909184,11812,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
3384035568,12198048,11829,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
3396235152,416,11843,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
3396236784,2509216,11846,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
3398747472,3777248,11879,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3402526000,3770368,11881,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3406299344,5082752,11897,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3411384560,5710464,11920,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3417097904,240101408,11923,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
3657200912,2166688,11930,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
3659370736,1920,11933,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
3659374832,2528,11947,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3659378896,1536,11962,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
3659382896,45856,11985,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3659431120,2762624,11997,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
3662195920,1920,12006,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
3662200016,8099264,12035,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
3670300976,5523968,12049,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
3675826448,3822336,12069,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
3679650064,2528,12084,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3679654128,2080,12102,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3679658224,46624,12142,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3679707344,3806144,12150,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
3683514768,1920,12153,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
3683518704,2444096,12156,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
3685965040,1920,12167,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
3685969136,34983616,12196,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
3720954160,139744,12237,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3721095408,12296512,12245,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
3733394640,9696,12248,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
3733406960,10774656,12251,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
3744184560,7232,12262,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
3744194928,3328,12277,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3744200976,15784672,12298,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
3759986992,5478976,12311,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
3765468400,1952,12319,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3765472496,1440,12330,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3765476592,1280,12341,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3765480656,3456,12355,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
3765486800,1312,12367,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3765490800,4448,12379,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
3765497040,1312,12392,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
3765499728,1632,12404,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
3765503216,1472,12417,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
3765507312,1280,12428,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3765511408,24768,12449,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
3765538032,3497504,12472,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
3769037040,2464,12487,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3769041136,2016,12505,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
3769045232,48544,12545,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
3769095376,3424864,12553,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
3772522768,3552,12556,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
3772528912,2510688,12559,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
3775041776,1888,12570,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
3775045872,24845056,12599,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
3799893232,11830912,12616,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
3811725200,416,12630,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
3811726480,2388928,12633,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
3814117616,3942304,12666,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3818062032,3950304,12668,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3822014736,5234368,12684,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3827250416,5660672,12707,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
3832913136,240325056,12710,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
4073240848,2157632,12717,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
4075400400,1952,12720,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
4075404496,2560,12734,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4075408624,1504,12749,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
4075412784,50400,12772,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4075465936,2760608,12784,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
4078227856,1952,12793,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4078231792,7826784,12822,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4086061328,5463200,12836,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
4091526352,3570432,12856,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
4095099184,2336,12871,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4095103184,2208,12889,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4095107312,48384,12929,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4095157488,4305856,12937,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
4099465456,1920,12940,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
4099469584,2451072,12943,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
4101923088,1856,12954,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4101927120,33911744,12983,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4135841008,206816,13024,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4136050928,12091680,13032,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
4148144336,2208,13035,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
4148148464,10751008,13038,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
4158902512,1920,13049,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4158906608,3328,13064,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4158912752,15259232,13085,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4174173648,5466400,13098,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
4179642672,1920,13106,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4179646704,1440,13117,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4179650768,1312,13128,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4179654896,3072,13142,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
4179661040,1312,13154,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4179665040,4800,13166,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
4179671248,1120,13179,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4179675312,1632,13191,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
4179679472,1568,13204,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
4179683536,1248,13215,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4179687696,28672,13236,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
4179718384,3498816,13259,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
4183218480,2432,13274,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4183222512,2112,13292,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4183226576,48096,13332,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4183276752,3518400,13340,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
4186796464,3456,13343,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
4186802448,2435904,13346,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
4189240560,1888,13357,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4189244656,26090176,13386,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4215337200,12096896,13403,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
4227435376,416,13417,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
4227437008,2347104,13420,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
4229786864,3730976,13453,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
4233519312,4067328,13455,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
4237588688,5428096,13471,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
4243019024,5745408,13494,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
4248766704,242399936,13497,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
4491168048,2168800,13504,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
4493338896,1920,13507,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
4493342960,2880,13521,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4493347184,1504,13536,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
4493351184,46592,13559,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4493399248,2748320,13571,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
4496149744,1952,13580,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4496153872,7929856,13609,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4504086576,5458016,13623,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
4509546736,3492480,13643,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
4513041744,2400,13658,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4513045712,1984,13676,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4513049840,47232,13716,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4513100016,4172256,13724,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
4517273808,1984,13727,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
4517277936,2537024,13730,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
4519816432,1888,13741,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4519820560,33533088,13770,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4553355536,157600,13811,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4553515248,12645408,13819,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
4566162640,2432,13822,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
4566166768,10464160,13825,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
4576632208,2400,13836,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4576636144,2208,13851,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4576640240,15692288,13872,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4592334064,5631424,13885,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
4597968112,2560,13893,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4597972208,1440,13904,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4597976272,1472,13915,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4597980368,3680,13929,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
4597986480,3136,13941,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4597992688,6432,13953,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
4598000880,1280,13966,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4598004976,7072,13978,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
4598013424,11168,13991,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
4598027504,1248,14002,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4598031600,71488,14023,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
4598105296,3607424,14046,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
4601714000,2464,14061,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4601718000,2080,14079,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4601722096,48672,14119,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4601772304,3869632,14127,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
4605645008,4032,14130,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
4605651152,2434688,14133,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
4608087280,1856,14144,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4608091408,25006880,14173,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4633100528,12395104,14190,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
4645497744,416,14204,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
4645499056,2352544,14207,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
4647854320,3720192,14240,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
4651577584,3872096,14242,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
4655452496,5580512,14258,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
4661034288,5712640,14281,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
4666749168,241404896,14284,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
4908156176,2166912,14291,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
4910324976,2016,14294,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
4910329072,2912,14308,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
4910333264,1536,14323,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
4910337264,48064,14346,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4910388432,2752704,14358,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
4913142992,1920,14367,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4913147120,7930720,14396,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4921079152,5452960,14410,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
4926534896,3469120,14430,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
4930006352,2464,14445,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4930010320,2016,14463,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
4930014448,46528,14503,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4930063600,3409568,14511,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
4933475536,1952,14514,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
4933479664,2424224,14517,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
4935905520,1888,14528,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4935909648,33606624,14557,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
4969519376,139712,14598,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4969661680,12773024,14606,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
4982436080,2176,14609,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
4982440176,10351520,14612,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
4992792976,1920,14623,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
4992796912,832,14638,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
4992799056,15773088,14659,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5008574736,5499680,14672,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
5014076656,2496,14680,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5014080752,1440,14691,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5014084848,3712,14702,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5014090992,3584,14716,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
5014097008,1344,14728,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5014101392,4544,14740,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
5014107408,1088,14753,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5014111472,1600,14765,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
5014115536,1920,14778,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
5014119664,1248,14789,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5014123728,26144,14810,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
5014152496,3804640,14833,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
5017958640,22080,14848,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5017983184,32192,14866,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5018018032,47840,14906,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5018068176,3836736,14914,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
5021906256,3744,14917,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
5021912336,2440928,14920,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
5024354576,1856,14931,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5024358640,24941184,14960,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5049301200,12450592,14977,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
5061753712,448,14991,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
5061755344,2347648,14994,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
5064105200,3717760,15027,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5067824464,3725984,15029,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5071551696,5502592,15045,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5077055728,5676000,15068,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5082734864,240707424,15071,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
5323444496,2169536,15078,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
5325615312,1952,15081,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
5325619440,2880,15095,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5325623600,1536,15110,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
5325627632,47104,15133,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5325676752,2767680,15145,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
5328446704,1920,15154,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5328450800,8481312,15183,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5336933648,5526464,15197,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
5342462192,3485760,15217,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
5345949936,2400,15232,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5345954032,1984,15250,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5345958128,46688,15290,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5346006224,3422112,15298,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
5349430640,1952,15301,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
5349434576,2419200,15304,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
5351855344,1888,15315,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5351859408,34284960,15344,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5386147088,139712,15385,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5386289392,12727072,15393,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
5399018704,1984,15396,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
5399022864,10358368,15399,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
5409383696,1920,15410,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5409387760,832,15425,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5409389872,15717632,15446,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5425110256,5460384,15459,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
5430572240,1952,15467,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5430576368,1440,15478,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5430580464,1312,15489,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5430584560,3392,15503,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
5430590672,1376,15515,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5430594800,4256,15527,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
5430600976,1088,15540,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5430605040,1600,15552,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
5430609104,1504,15565,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
5430613328,1248,15576,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5430617296,25312,15597,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
5430643952,3500224,15620,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
5434146064,2496,15635,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5434150192,2112,15653,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5434154224,49952,15693,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5434206448,4299072,15701,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
5438508272,3648,15704,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
5438514416,2441600,15707,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
5440958704,1888,15718,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5440962800,24888224,15747,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5465852336,12351616,15764,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
5478208304,4000,15778,,,,,,,,,,0.000,56.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
5478213936,2420480,15781,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
5480636656,3707936,15814,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5484346576,3735616,15816,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5488084176,5065568,15832,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5493151984,5827808,15855,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5498981616,241799776,15858,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
5740782896,2166496,15865,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
5742950864,1984,15868,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
5742954704,2912,15882,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5742958928,1536,15897,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
5742962896,46112,15920,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5743012048,2754048,15932,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
5745768688,1920,15941,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5745772752,7818272,15970,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5753593104,5449504,15984,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
5759044816,3466176,16004,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
5762512304,2304,16019,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5762516208,1952,16037,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5762520304,47168,16077,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5762569456,3411264,16085,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
5765983440,1952,16088,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
5765987568,2425440,16091,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
5768415504,1888,16102,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5768419568,34226240,16131,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5802648976,140032,16172,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5802791152,11884448,16180,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
5814676880,1920,16183,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
5814680848,10341376,16186,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
5825025264,1984,16197,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5825029360,2560,16212,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5825033456,15733664,16233,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5840768400,5464736,16246,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
5846234416,1920,16254,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5846238480,1408,16265,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5846242512,1312,16276,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5846246608,3072,16290,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
5846252752,1344,16302,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5846256880,4864,16314,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
5846263056,1088,16327,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
5846267152,1664,16339,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
5846271184,1504,16352,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
5846275312,1280,16363,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5846279376,25952,16384,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
5846308080,3495520,16407,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
5849806032,2368,16422,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5849810128,2112,16440,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
5849814224,48096,16480,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
5849864432,3438592,16488,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
5853306064,3424,16491,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
5853312368,2821952,16494,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
5856136464,1856,16505,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
5856140528,25366272,16534,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
5881509168,11847616,16551,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
5893358672,416,16565,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
5893360240,2496864,16568,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
5895859440,3826176,16601,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5899688272,3904192,16603,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5903594704,5218816,16619,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5908815088,5683744,16642,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
5914500336,241798560,16645,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
6156301584,2171424,16652,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
6158475504,1984,16655,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
6158479600,2912,16669,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6158483792,1536,16684,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
6158487792,48224,16707,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6158538960,2755936,16719,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
6161297648,1952,16728,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6161301712,7908928,16757,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6169213200,5461504,16771,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
6174677200,3478752,16791,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
6178157808,2304,16806,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6178161904,2016,16824,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6178166000,45856,16864,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6178214128,3426912,16872,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
6181643472,1984,16875,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
6181647600,2425184,16878,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
6184075504,1888,16889,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6184079600,34891392,16918,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6218972432,141280,16959,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6219116784,11920992,16967,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
6231040240,1952,16970,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
6231044336,10766528,16973,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
6241813744,2144,16984,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6241817840,832,16999,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6241819952,16326752,17020,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6258148592,5467264,17033,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
6263617776,1984,17041,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6263621904,1408,17052,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6263625968,1312,17063,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6263630064,3264,17077,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
6263636208,1440,17089,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6263640304,4704,17101,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
6263646480,1088,17114,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6263650544,1600,17126,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
6263654640,1472,17139,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
6263658736,1280,17150,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6263662832,26688,17171,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
6263691504,3502624,17194,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
6267195632,2400,17209,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6267199696,1984,17227,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6267203568,47584,17267,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6267254000,3522112,17275,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
6270778608,3968,17278,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
6270784752,2439168,17281,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
6273226992,1888,17292,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6273231088,25829856,17321,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6299062512,11811904,17338,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
6310876048,416,17352,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
6310877328,2378688,17355,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
6313257552,3908832,17388,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
6317167952,3972352,17390,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
6321143472,5239040,17406,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
6326385040,5665728,17429,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
6332052720,240203808,17432,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
6572259600,2177760,17439,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
6574439664,1952,17442,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
6574443728,2976,17456,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6574447984,1504,17471,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
6574451920,48864,17494,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6574502192,2758368,17506,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
6577262832,1920,17515,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6577266928,7814496,17544,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6585084176,5459264,17558,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
6590546160,3478464,17578,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
6594025904,2400,17593,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6594029808,2112,17611,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6594033904,47328,17651,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6594083024,3426688,17659,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
6597512432,1952,17662,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
6597516528,2426080,17665,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
6599944464,1856,17676,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6599948528,33846624,17705,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6633797872,172064,17746,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6633972944,12456320,17754,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
6646431952,2208,17757,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
6646436080,10598688,17760,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
6657036656,1952,17771,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6657040656,864,17786,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6657042832,15399488,17807,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6672444656,5615488,17820,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
6678061456,2016,17828,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6678065456,1664,17839,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6678069744,1312,17850,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6678073584,3648,17864,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
6678079696,1344,17876,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6678083824,11104,17888,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
6678096208,1856,17901,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6678100208,2080,17913,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
6678104304,4032,17926,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
6678110448,1504,17937,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6678114544,62208,17958,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
6678178992,3617312,17981,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
6681798928,2560,17996,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6681803024,2080,18014,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
6681807056,48160,18054,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6681857264,3838528,18062,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
6685698288,4000,18065,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
6685704432,2440192,18068,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
6688146672,1920,18079,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6688150768,26015712,18108,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6714169584,12030144,18125,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
6726202224,544,18139,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
6726203984,2354880,18142,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
6728560880,3711456,18175,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
6732274928,3963872,18177,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
6736241872,5437760,18193,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
6741681360,5725024,18216,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
6747408624,239633120,18219,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
6987044112,2163136,18226,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
6989208784,1952,18229,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
6989212944,2688,18243,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
6989217040,1504,18258,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
6989221104,46848,18281,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
6989270352,2756896,18293,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
6992028880,1952,18302,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
6992033040,7841664,18331,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
6999876048,5460608,18345,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
7005338864,3489408,18365,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
7008830704,2496,18380,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7008834768,1984,18398,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7008838864,46144,18438,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7008886992,3410336,18446,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
7012299984,1952,18449,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7012304112,2667616,18452,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
7014974448,3712,18463,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7014980848,34079488,18492,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7049062704,139456,18533,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7049203952,12768992,18541,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
7061975248,1984,18544,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7061979376,10367616,18547,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
7072348496,1920,18558,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7072352496,864,18573,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7072354640,15705376,18594,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7088061680,5511744,18607,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
7093574960,2272,18615,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7093578992,1408,18626,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7093583088,3680,18637,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7093589200,3616,18651,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7093595248,1408,18663,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7093599472,4448,18675,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
7093605616,1088,18688,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7093609744,1664,18700,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
7093613808,1472,18713,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7093617936,1280,18724,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7093622000,26848,18745,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
7093650640,3741248,18768,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
7097394416,13184,18783,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7097409136,13472,18801,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7097425136,97696,18841,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7097525488,3970560,18849,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
7101498576,3552,18852,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7101504752,2437920,18855,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
7103944944,1888,18866,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7103949040,24860704,18895,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7128811728,12400832,18912,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
7141214064,416,18926,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
7141215376,2338976,18929,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
7143557328,3707360,18962,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7147267280,3739648,18964,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7151008976,5512160,18980,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7156522416,5684192,19003,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7162209552,242227488,19006,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
7404438896,2180384,19013,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
7406620880,1952,19016,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
7406624976,2560,19030,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7406629072,1536,19045,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7406633200,45024,19068,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7406680304,2751680,19080,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
7409433840,1856,19089,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7409437936,7852832,19118,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7417292080,5460768,19132,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
7422755056,3467584,19152,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
7426225392,2368,19167,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7426229488,2112,19185,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7426233616,45408,19225,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7426281712,3408320,19233,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
7429691600,1952,19236,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7429695728,2422752,19239,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
7432121584,1856,19250,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7432125648,34878368,19279,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7467005072,142560,19320,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7467150576,11880896,19328,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
7479033072,1952,19331,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7479037168,10359680,19334,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
7489399088,6624,19345,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7489407216,1568,19360,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7489411312,15720800,19381,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7505134864,5521792,19394,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
7510659312,3904,19402,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7510665424,1440,19413,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7510669552,3008,19424,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7510675696,3520,19438,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7510681776,2048,19450,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7510685936,6688,19462,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
7510694128,2848,19475,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7510698288,1600,19487,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
7510702320,2304,19500,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7510706416,2240,19511,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7510710512,29504,19532,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
7510741296,3520864,19555,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
7514265104,5600,19570,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7514271984,1952,19588,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7514275952,52032,19628,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7514330416,4260224,19636,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
7518592208,1984,19639,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7518596336,2447712,19642,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
7521046800,1856,19653,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7521050608,25226208,19682,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7546278096,12453792,19699,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
7558733712,416,19713,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
7558735472,2351712,19716,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
7561090288,3714304,19749,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7564807376,3786336,19751,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7568596176,5154144,19767,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7573753072,5813440,19790,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7579568368,240922976,19793,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
7820493072,2156896,19800,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
7822651664,1952,19803,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
7822655728,3360,19817,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7822661840,1568,19832,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7822665872,51264,19855,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7822719184,2771712,19867,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
7825492144,1952,19876,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7825496304,7849312,19905,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7833347344,5451136,19919,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
7838800240,3479136,19939,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
7842280656,2368,19954,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7842284752,2144,19972,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7842288880,47424,20012,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7842339056,3417600,20020,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
7845758160,2016,20023,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7845762288,2423840,20026,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
7848189168,1888,20037,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7848193008,34270688,20066,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7882465552,140768,20107,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7882607856,11865984,20115,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
7894475120,2464,20118,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7894479088,10348864,20121,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
7904830704,1952,20132,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7904834800,2656,20147,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7904838896,15733856,20168,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7920575728,5463552,20181,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
7926041840,1920,20189,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7926045904,1440,20200,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7926050000,1312,20211,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7926054096,3520,20225,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7926060272,1344,20237,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7926064368,4992,20249,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
7926070640,1120,20262,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
7926074640,1632,20274,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
7926078704,1504,20287,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
7926082800,1216,20298,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7926086896,25632,20319,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
7926115568,3505824,20342,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
7929623792,2432,20357,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7929627888,1920,20375,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
7929631984,47648,20415,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
7929681136,3448928,20423,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
7933133040,4032,20426,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
7933139248,2809952,20429,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
7935952112,1856,20440,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
7935956272,25293632,20469,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
7961252080,11798880,20486,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
7973053296,448,20500,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
7973054672,2382432,20503,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
7975439600,3919904,20536,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7979362544,3922944,20538,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7983287504,5215584,20554,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7988505808,5690784,20577,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
7994200656,241148000,20580,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
8235351312,2184512,20587,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
8237538512,1984,20590,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
8237542640,3040,20604,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8237548752,1568,20619,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
8237552784,52064,20642,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8237606128,2756640,20654,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
8240365808,1888,20663,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
8240369904,7952640,20692,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
8248324368,5449216,20706,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
8253775088,3472800,20726,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
8257249520,2400,20741,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8257253616,1984,20759,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8257257712,45696,20799,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8257305840,3431232,20807,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
8260739280,1952,20810,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
8260743472,2425536,20813,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
8263171312,1888,20824,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
8263175408,34385728,20853,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
8297562448,140160,20894,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8297705712,11916032,20902,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
8309624048,2432,20905,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
8309628144,10800032,20908,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
8320430320,1920,20919,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
8320434416,832,20934,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8320436624,16112576,20955,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
8336551152,5465952,20968,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
8342019312,1952,20976,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8342023440,1440,20987,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8342027536,1312,20998,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8342031600,3680,21012,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
8342037776,1312,21024,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8342041808,4896,21036,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
8342047984,1088,21049,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8342052080,1632,21061,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
8342056176,1472,21074,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
8342060272,1248,21085,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8342064368,26112,21106,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
8342093040,3493632,21129,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
8345587952,2560,21144,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8345592016,1984,21162,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8345596144,47840,21202,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8345646320,3444576,21210,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
8349092208,4096,21213,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
8349098224,2448192,21216,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
8351548656,1888,21227,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
8351552784,25738272,21256,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
8377295632,11891808,21273,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
8389188464,544,21287,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
8389190192,2346944,21290,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
8391538928,3761664,21323,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
8395303152,4026528,21325,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
8399331536,5417472,21341,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
8404751600,5663296,21364,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
8410416368,241593728,21367,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
8652011792,2164960,21374,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
8654178544,1920,21377,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
8654182640,2976,21391,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8654186896,1536,21406,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
8654190832,48416,21429,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8654242032,2752160,21441,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
8656995568,1888,21450,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
8656999664,7942144,21479,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
8664943920,5458208,21493,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
8670404848,3475616,21513,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
8673883376,4320,21528,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8673889680,19776,21546,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8673912016,52288,21586,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8673967312,4322208,21594,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
8678291696,1952,21597,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
8678295760,2444640,21600,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
8680743152,1920,21611,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
8680747248,34577664,21640,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
8715326704,141088,21681,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8715469072,12382976,21689,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
8727853392,1984,21692,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
8727857360,10443584,21695,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
8738303216,1952,21706,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
8738307312,2592,21721,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8738311408,15308832,21742,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
8753622256,5479104,21755,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
8759102704,1952,21763,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8759106800,1408,21774,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8759110896,3648,21785,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8759117040,3072,21799,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
8759123184,1376,21811,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8759127184,4640,21823,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
8759133456,1152,21836,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
8759137264,1600,21848,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
8759141584,1472,21861,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
8759145712,1248,21872,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8759149776,25216,21893,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
8759176432,3694624,21916,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
8762874096,2400,21931,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8762878256,2176,21949,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
8762882288,50528,21989,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
8762934480,3638240,21997,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
8766574800,1952,22000,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
8766578928,2442912,22003,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
8769024368,1920,22014,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
8769028368,25552160,22043,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
8794582288,12077248,22060,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
8806660976,448,22074,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
8806662096,2352160,22077,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
8809015568,3721376,22110,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
8812738768,4098432,22112,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
8816839888,5346304,22128,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
8822189264,5725184,22151,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
8827916528,241606016,22154,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
9069525264,2174688,22161,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
9071702320,1952,22164,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
9071706384,2720,22178,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9071710416,1536,22193,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
9071714512,47232,22216,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9071763664,2758720,22228,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
9074524400,1952,22237,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9074528496,7921504,22266,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9082451344,5461920,22280,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
9087915248,3486688,22300,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
9091405040,2400,22315,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9091409136,2080,22333,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9091413200,46048,22373,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9091462352,3874528,22381,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
9095339248,1952,22384,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
9095343344,2624832,22387,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
9097970928,1856,22398,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9097975024,33697120,22427,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9131674896,139840,22468,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9131817200,11919712,22476,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
9143739600,1952,22479,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
9143743760,10381664,22482,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
9154128944,1952,22493,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9154133104,1824,22508,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9154136304,15711680,22529,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9169850608,5507136,22542,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
9175359728,1920,22550,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9175363824,1568,22561,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9175367920,3168,22572,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9175374096,3392,22586,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
9175380144,1344,22598,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9175384336,4928,22610,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
9175390736,1120,22623,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9175394448,1632,22635,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
9175398640,1696,22648,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
9175402768,7616,22659,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9175412976,26336,22680,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
9175441520,3844544,22703,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
9179287792,2336,22718,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9179291888,2048,22736,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9179295984,48768,22776,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9179347184,3828800,22784,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
9183177968,4032,22787,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
9183184112,2423808,22790,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
9185611024,1888,22801,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9185615056,24620416,22830,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9210238192,12466528,22847,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
9222706064,480,22861,,,,,,,,,,0.000,466.667,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
9222707760,2350880,22864,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
9225061616,3725056,22897,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
9228787952,3735968,22899,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
9232525520,5525984,22915,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
9238053104,5802432,22938,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
9243858128,242206816,22941,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
9486066960,2168960,22948,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
9488237840,2016,22951,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
9488241904,2944,22965,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9488246128,1536,22980,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
9488250096,45504,23003,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9488297168,2763104,23015,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
9491063024,1920,23024,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9491067120,7944544,23053,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9499014416,5450496,23067,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
9504466192,3475360,23087,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
9507942800,2368,23102,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9507946736,2048,23120,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9507950832,47552,23160,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9507999984,3421152,23168,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
9511424272,1952,23171,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
9511428336,2483008,23174,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
9513912656,2368,23185,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9513916848,34376960,23214,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9548296464,139616,23255,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9548438768,12796832,23263,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
9561237712,2240,23266,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
9561241840,10357696,23269,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
9571601648,1920,23280,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9571605744,832,23295,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9571607856,15753632,23316,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9587363056,5458112,23329,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
9592823024,1984,23337,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9592827216,1408,23348,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9592831216,1312,23359,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9592835312,3296,23373,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
9592841456,1344,23385,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9592845456,4448,23397,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
9592851664,1088,23410,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9592855696,1632,23422,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
9592859888,1472,23435,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
9592863984,1280,23446,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9592868080,25184,23467,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
9592894704,3628416,23490,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
9596533904,17728,23505,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9596554480,11680,23523,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9596568848,69088,23563,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9596639472,4118272,23571,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
9600760016,3840,23574,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
9600766224,2419840,23577,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
9603187952,1888,23588,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9603192048,24930496,23617,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9628124400,12466368,23634,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
9640592240,544,23648,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
9640593680,2350944,23651,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
9642946832,3729024,23684,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
9646678224,3748576,23686,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
9650428112,5354496,23702,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
9655784720,5783808,23725,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
9661571312,241193152,23728,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
9902766384,2168640,23735,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
9904937200,1952,23738,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
9904941296,2944,23752,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
9904945520,1568,23767,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
9904949456,47232,23790,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9904998640,2753088,23802,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
9907754224,1888,23811,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9907758320,7844192,23840,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9915605264,5462816,23854,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
9921070352,3466848,23874,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
9924539632,2304,23889,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9924543728,2048,23907,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
9924547792,47456,23947,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9924596944,3416192,23955,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
9928016112,1952,23958,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
9928020208,2424832,23961,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
9930448112,1856,23972,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9930452208,34503808,24001,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
9964957968,140736,24042,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9965101264,12747232,24050,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
9977850096,2208,24053,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
9977854192,10374624,24056,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
9988231408,1920,24067,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
9988235504,2528,24082,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
9988239600,15701312,24103,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10003943664,5469120,24116,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
10009414896,1920,24124,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10009418992,1408,24135,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10009423088,1376,24146,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10009427184,3136,24160,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
10009433328,1344,24172,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10009437424,6720,24184,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
10009445616,1088,24197,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10009449712,1600,24209,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
10009453808,1472,24222,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
10009457904,1312,24233,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10009462000,27584,24254,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
10009492688,3491904,24277,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
10012987664,2528,24292,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10012991824,2016,24310,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10012995824,48352,24350,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10013046000,4232896,24358,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
10017281232,4096,24361,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
10017287376,2498976,24364,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
10019789040,1952,24375,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10019793104,24911776,24404,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10044707056,12388768,24421,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
10057100048,416,24435,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
10057102288,2363744,24438,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
10059469072,3784768,24471,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10063255792,3747296,24473,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10067004368,5073216,24489,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10072080656,5698400,24512,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10077780336,242541280,24515,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
10320322960,2168832,24522,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
10322493680,1920,24525,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
10322497776,2944,24539,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10322502032,1536,24554,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
10322505968,49696,24577,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10322557232,2751232,24589,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
10325310704,1952,24598,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10325314800,7992640,24627,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10333310288,5463872,24641,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
10338775408,3477760,24661,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
10342254832,2368,24676,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10342258928,2048,24694,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10342263024,46048,24734,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10342312176,3447488,24742,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
10345762032,1952,24745,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
10345766128,2423296,24748,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
10348191984,1920,24759,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10348196080,34563424,24788,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10382762288,140576,24829,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10382905648,12654624,24837,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
10395562192,2688,24840,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
10395566320,10480992,24843,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
10406050032,1920,24854,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10406054096,1088,24869,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10406056656,15701088,24890,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10421760240,5465056,24903,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
10427228432,1920,24911,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10427232496,1408,24922,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10427236592,1280,24933,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10427240688,3232,24947,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
10427246864,1344,24959,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10427250928,4832,24971,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
10427257040,1120,24984,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10427261232,1632,24996,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
10427265264,1472,25009,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
10427269360,1280,25020,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10427273456,25920,25041,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
10427302128,3508896,25064,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
10430812144,2464,25079,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10430816496,2016,25097,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10430820592,48512,25137,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10430871792,3516544,25145,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
10434391280,4384,25148,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
10434397424,2736736,25151,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
10437136624,1888,25162,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10437140720,25314144,25191,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10462456176,12035712,25208,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
10474496016,448,25222,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
10474497360,2454048,25225,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
10476953840,3817408,25258,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10480773360,3902560,25260,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10484678928,5091520,25276,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10489772272,5706880,25299,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10495482096,242566784,25302,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
10738051344,2171840,25309,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
10740226256,1984,25312,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
10740230384,3104,25326,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10740236496,1536,25341,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
10740240528,47072,25364,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10740289776,2751712,25376,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
10743043312,1888,25385,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10743047408,7940800,25414,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10750989584,5517184,25428,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
10756508912,3821792,25448,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
10760332496,2496,25463,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10760336720,2048,25481,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10760340720,46816,25521,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10760388816,3812448,25529,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
10764204272,1920,25532,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
10764208336,2426976,25535,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
10766637296,1888,25546,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10766641392,34306016,25575,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10800949488,140000,25616,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10801091824,12269024,25624,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
10813362416,5312,25627,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
10813370608,10774496,25630,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
10824147184,4960,25641,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10824153424,7264,25656,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10824163568,15771104,25677,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10839935984,5466784,25690,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
10845405424,1920,25698,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10845409616,1440,25709,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10845413648,3168,25720,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10845419760,3168,25734,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
10845425808,1312,25746,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10845430000,4544,25758,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
10845436144,1088,25771,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
10845440272,1632,25783,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
10845444336,1472,25796,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
10845448400,1312,25807,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10845452528,26560,25828,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
10845481200,3503456,25851,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
10848987408,2368,25866,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10848991472,2016,25884,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
10848995568,48320,25924,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
10849045712,3440576,25932,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
10852487568,4032,25935,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
10852493584,2674176,25938,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
10855170288,1920,25949,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
10855174416,25492928,25978,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
10880668912,11806688,25995,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
10892477296,448,26009,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
10892478928,2395648,26012,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
10894875888,3943904,26045,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10898821328,3916544,26047,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10902740176,5236320,26063,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10907977936,5662944,26086,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
10913642736,242114368,26089,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
11155759408,2167616,26096,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
11157928336,1920,26099,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
11157932400,2912,26113,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11157938448,1504,26128,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
11157942384,47264,26151,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11157991632,2756704,26163,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
11160751312,1984,26172,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11160755472,7890016,26201,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
11168647472,5505440,26215,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
11174154512,3673952,26235,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
11177830640,2592,26250,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11177834704,2048,26268,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11177841872,55936,26308,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11177899216,4036064,26316,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
11181937872,1984,26319,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
11181942000,2423648,26322,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
11184367856,1888,26333,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11184371952,34814112,26362,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
11219188016,140544,26403,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11219331312,11914592,26411,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
11231248592,1952,26414,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
11231252720,10789568,26417,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
11242043632,2112,26428,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11242047696,960,26443,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11242049968,16312800,26464,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
11258365168,5469024,26477,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
11263835472,2144,26485,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11263840496,1408,26496,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11263844592,1280,26507,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11263848720,3296,26521,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
11263854864,1440,26533,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11263858928,4320,26545,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
11263865072,1120,26558,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11263869168,1600,26570,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
11263873264,1504,26583,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
11263877360,1248,26594,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11263881424,25120,26615,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
11263908080,3509536,26638,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
11267419344,2432,26653,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11267423472,1984,26671,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11267427568,48000,26711,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11267476944,3441024,26719,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
11270920432,3776,26722,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
11270926576,2436416,26725,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
11273364720,1856,26736,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11273368816,25923936,26765,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
11299295472,11797920,26782,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
11311094640,448,26796,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
11311095984,2385056,26799,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
11313484016,3934176,26832,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
11317421296,3955424,26834,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
11321379536,5190528,26850,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
11326571728,5673920,26873,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
11332247792,242693856,26876,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
11574944016,2174560,26883,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
11577120976,1984,26886,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
11577125104,2752,26900,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11577129168,1536,26915,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
11577133264,45664,26938,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11577180368,2752608,26950,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
11579935952,1920,26959,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11579940048,7985536,26988,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
11587928368,5481344,27002,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
11593411216,3706112,27022,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
11597119728,2560,27037,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11597123856,2336,27055,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11597127920,47072,27095,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11597178096,4002432,27103,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
11601182960,1952,27106,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
11601186800,2423616,27109,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
11603611888,1888,27120,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11603615984,34343840,27149,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
11637962000,140544,27190,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11638105328,11934080,27198,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
11650041072,2144,27201,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
11650045168,10771616,27204,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
11660819696,2016,27215,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11660823792,1056,27230,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11660827888,16033728,27251,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
11676863728,5465600,27264,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
11682330896,1952,27272,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11682334992,1440,27283,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11682337936,1312,27294,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11682341072,3296,27308,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
11682347248,1312,27320,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11682351248,4576,27332,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
11682357488,1088,27345,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11682361584,1632,27357,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
11682365680,1472,27370,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
11682369776,1280,27381,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11682373872,26848,27402,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
11682402544,3503616,27425,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
11685908720,2496,27440,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11685912784,1984,27458,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
11685916912,47392,27498,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11685966064,3441888,27506,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
11689409744,3264,27509,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
11689415920,2435552,27512,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
11691853072,1856,27523,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11691857136,26544096,27552,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
11718404336,11803168,27569,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
11730208624,416,27583,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
11730210192,2344192,27586,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
11732556080,3879392,27619,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
11736438000,3889536,27621,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
11740330192,5382944,27637,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
11745714416,5676192,27660,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
11751392528,241273408,27663,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
11992667440,2164000,27670,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
11994834128,1952,27673,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
11994838256,2848,27687,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
11994843376,1536,27702,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
11994847344,47232,27725,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
11994896592,2758016,27737,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
11997657328,2080,27746,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
11997661392,7871520,27775,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12005536048,5454944,27789,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
12010993904,3469600,27809,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
12014465264,2304,27824,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12014469360,2080,27842,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12014473456,46176,27882,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12014522608,3422656,27890,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
12017946864,1952,27893,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12017950960,2425152,27896,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
12020378864,1888,27907,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12020382992,34162048,27936,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12054547696,139200,27977,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12054688272,12332832,27985,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
12067024112,1952,27988,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12067028208,10806528,27991,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
12077837552,1920,28002,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12077841616,3072,28017,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12077847792,15536672,28038,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12093387024,5629920,28051,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
12099018992,1920,28059,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12099023152,1408,28070,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12099027184,1472,28081,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12099031280,3360,28095,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
12099036400,1312,28107,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12099040496,4448,28119,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
12099046640,1184,28132,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12099050704,1824,28144,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
12099054832,1632,28157,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
12099058928,1280,28168,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12099063024,25696,28189,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
12099091696,3543520,28212,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
12102636784,2464,28227,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12102640880,2304,28245,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12102644944,63616,28285,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12102710480,3828128,28293,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
12106541296,3296,28296,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12106547440,2443872,28299,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
12108993776,1888,28310,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12108997872,24936000,28339,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12133935344,11814080,28356,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
12145751952,416,28370,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
12145753296,2352896,28373,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
12148107472,3725216,28406,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12151834832,3740704,28408,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12155578576,5090944,28424,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12160671984,5663328,28447,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12166337776,241822912,28450,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
12408162608,2157568,28457,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
12410323184,1952,28460,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
12410327280,2976,28474,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12410331568,1504,28489,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
12410335216,50112,28512,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12410386704,2750304,28524,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
12413139184,1920,28533,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12413143280,7919840,28562,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12421064976,5455904,28576,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
12426523888,3468640,28596,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
12429995248,2368,28611,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12429999344,1984,28629,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12430003408,48128,28669,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12430054640,3424576,28677,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
12433481968,1952,28680,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12433486096,2426240,28683,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
12435913968,1856,28694,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12435918064,33735648,28723,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12469656848,140768,28764,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12469799152,12803680,28772,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
12482604336,2176,28775,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12482608368,10379136,28778,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
12492989680,1952,28789,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12492993776,864,28804,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12492995920,15888288,28825,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12508887280,5504480,28838,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
12514393328,2112,28846,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12514397424,1408,28857,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12514401520,1280,28868,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12514405616,3744,28882,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
12514411728,1344,28894,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12514414544,4896,28906,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
12514421904,1088,28919,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12514426096,1632,28931,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
12514430192,1760,28944,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
12514434288,1248,28955,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12514438352,34208,28976,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
12514475248,3824608,28999,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
12518302960,2528,29014,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12518307056,2016,29032,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12518311152,47136,29072,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12518360272,3841696,29080,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
12522203344,3968,29083,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12522209552,2437024,29086,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
12524648688,1888,29097,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12524652816,24963488,29126,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12549617904,12417312,29143,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
12562036624,416,29157,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
12562038256,2346336,29160,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
12564387056,3717536,29193,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12568106224,3741536,29195,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12571849040,5525152,29211,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12577376496,5689024,29234,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12583066864,242627456,29237,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
12825695632,2167232,29244,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
12827865328,1984,29247,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
12827869456,2976,29261,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12827873744,1504,29276,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
12827877616,47264,29299,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12827926736,2752640,29311,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
12830681328,1952,29320,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12830685424,7922240,29349,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12838610192,5462688,29363,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
12844074192,3470944,29383,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
12847546608,2432,29398,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12847550704,2112,29416,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12847554800,47168,29456,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12847604976,3418752,29464,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
12851025136,1952,29467,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12851029232,2467008,29470,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
12853498096,2080,29481,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12853502192,34443328,29510,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12887948560,140224,29551,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12888091888,11968896,29559,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
12900063440,2208,29562,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12900067536,10367680,29565,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
12910436592,1952,29576,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12910440688,832,29591,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12910442800,15742528,29612,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12926187760,5472992,29625,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
12931662032,1984,29633,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12931666160,1440,29644,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12931670256,3712,29655,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12931676400,3456,29669,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
12931682448,1344,29681,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12931686640,4352,29693,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
12931692784,1120,29706,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
12931696880,1664,29718,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
12931700976,1504,29731,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
12931705136,1280,29742,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12931709168,25920,29763,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
12931737936,3556960,29786,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
12935297264,2464,29801,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12935301360,2208,29819,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
12935305456,48480,29859,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
12935355760,4213088,29867,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
12939571472,3584,29870,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
12939577584,2431296,29873,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
12942010640,1888,29884,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
12942014704,24939456,29913,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
12966956272,12481600,29930,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
12979440560,448,29944,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
12979441872,2345504,29947,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
12981789936,3715584,29980,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12985507024,3743616,29982,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12989252848,5140192,29998,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
12994395376,5886496,30021,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13000284464,243240192,30024,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
13243527472,2167936,30031,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
13245698288,1952,30034,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
13245702352,2784,30048,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13245706480,1536,30063,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
13245710576,47040,30086,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13245760720,2864192,30098,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
13248626928,1888,30107,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
13248631024,7919136,30136,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
13256551696,5460800,30150,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
13262013808,3467264,30170,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
13265482992,2336,30185,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13265487056,2080,30203,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13265491184,46144,30243,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13265540336,3419168,30251,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
13268962512,1984,30254,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
13268966640,2430464,30257,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
13271399664,1888,30268,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
13271403760,34480480,30297,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
13305886992,140800,30338,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13306030064,11942176,30346,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
13317974224,2464,30349,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
13317978352,10435040,30352,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
13328414960,1920,30363,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
13328419184,896,30378,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13328423184,15770560,30399,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
13344196880,5478400,30412,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
13349678320,1920,30420,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13349682416,1440,30431,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13349686480,1344,30442,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13349690576,3424,30456,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
13349696752,1344,30468,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13349700848,4288,30480,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
13349706992,1088,30493,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13349711088,1632,30505,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
13349715184,1472,30518,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
13349719312,1440,30529,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13349723344,25152,30550,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
13349750000,3505472,30573,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
13353257200,2432,30588,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13353261296,1984,30606,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13353265360,47424,30646,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13353314512,4290496,30654,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
13357609776,5088,30657,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
13357617392,2470944,30660,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
13360091376,1920,30671,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
13360095472,24951584,30700,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
13385048336,12361888,30717,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
13397412144,416,30731,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
13397413936,2352704,30734,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
13399769328,3794400,30767,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13403566320,3743264,30769,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13407311056,5097024,30785,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13412409680,5794272,30808,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13418206448,242996160,30811,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
13661205776,2172704,30818,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
13663380688,1984,30821,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
13663384816,2752,30835,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13663388912,1536,30850,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
13663393008,45984,30873,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13663440240,2770784,30885,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
13666213072,2016,30894,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
13666217264,7897696,30923,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
13674116368,5469888,30937,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
13679587536,3478528,30957,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
13683068144,2336,30972,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13683072336,2176,30990,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13683076528,47104,31030,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13683125584,3417632,31038,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
13686544592,1952,31041,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
13686548720,2435296,31044,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
13688985872,1920,31055,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
13688989904,35071936,31084,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
13724064048,140832,31125,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13724206320,11934080,31133,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
13736143184,2208,31136,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
13736147184,10473696,31139,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
13746623728,1920,31150,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
13746627792,864,31165,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13746629936,15780192,31186,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
13762411760,5487200,31199,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
13767901520,2016,31207,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13767905616,1440,31218,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13767909616,1312,31229,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13767913808,3488,31243,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
13767919856,1312,31255,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13767923952,4704,31267,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
13767930192,1120,31280,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
13767934224,1632,31292,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
13767938384,1536,31305,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
13767942480,1312,31316,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13767946480,26112,31337,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
13767975152,3522240,31360,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
13771499728,2336,31375,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13771503856,1984,31393,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
13771507952,47360,31433,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
13771557488,3504160,31441,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
13775063312,3200,31444,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
13775069456,2750464,31447,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
13777822128,1888,31458,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
13777826032,25831808,31487,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
13803659536,12180320,31504,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
13815841840,448,31518,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
13815843056,2370304,31521,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
13818215664,3851904,31554,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13822070000,3824896,31556,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13825897680,5097152,31572,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13830996208,5708416,31595,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
13836707152,243821344,31598,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
14080530704,2179040,31605,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
14082711760,1920,31608,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
14082715984,2752,31622,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14082720112,1632,31637,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
14082724080,49440,31660,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14082775280,2768608,31672,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
14085545200,1888,31681,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14085549296,8041856,31710,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14093592880,5523488,31724,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
14099119312,3508768,31744,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
14102629712,2336,31759,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14102633712,2016,31777,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14102637936,45984,31817,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14102685936,3415424,31825,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
14106104016,1952,31828,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
14106108144,2435424,31831,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
14108546288,1888,31842,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14108550384,34749824,31871,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14143302928,139584,31912,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14143444176,12637024,31920,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
14156083440,2208,31923,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
14156087536,10553792,31926,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
14166643952,1920,31937,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14166648048,2944,31952,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14166652272,15802976,31973,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14182456528,5484832,31986,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
14187943152,1952,31994,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14187947248,1440,32005,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14187951440,1632,32016,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14187955440,3232,32030,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
14187961744,1376,32042,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14187965776,4896,32054,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
14187971984,1344,32067,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14187976016,1632,32079,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
14187979984,1504,32092,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
14187984112,1280,32103,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14187988432,28864,32124,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
14188018928,3517408,32147,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
14191538512,2496,32162,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14191542512,1952,32180,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14191546704,49504,32220,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14191598928,3880416,32228,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
14195480816,3968,32231,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
14195486960,2623360,32234,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
14198112592,1888,32245,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14198116592,25732096,32274,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14223850736,12178304,32291,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
14236030896,384,32305,,,,,,,,,,0.000,583.333,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
14236032496,2502656,32308,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
14238537744,3782816,32341,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
14242322640,3773120,32343,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
14246097104,5226624,32359,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
14251325680,5718816,32382,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
14257045808,245704832,32385,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
14502752656,2174368,32392,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
14504929488,1984,32395,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
14504933616,2944,32409,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14504937872,1536,32424,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
14504941808,47296,32447,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14504990928,2773440,32459,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
14507766096,1952,32468,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14507770224,8355648,32497,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14516129136,5595744,32511,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
14521727248,3582016,32531,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
14525311184,2336,32546,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14525315312,2016,32564,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14525319504,45600,32604,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14525366544,3429056,32612,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
14528798032,1984,32615,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
14528802032,2428384,32618,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
14531231984,1984,32629,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14531236112,34769312,32658,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14566007120,140896,32699,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14566150480,12833760,32707,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
14578987216,1952,32710,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
14578991344,10459904,32713,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
14589452592,1952,32724,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14589456624,864,32739,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14589459120,15787264,32760,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14605248752,5485600,32773,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
14610736496,1952,32781,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14610740464,1472,32792,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14610744656,1312,32803,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14610748752,3552,32817,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
14610754800,1344,32829,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14610758992,4800,32841,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
14610765072,1120,32854,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14610769232,1632,32866,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
14610773328,1472,32879,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
14610777328,1376,32890,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14610781520,27456,32911,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
14610810256,3513824,32934,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
14614326608,2656,32949,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14614330608,1952,32967,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14614334800,54208,33007,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14614392048,4344608,33015,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
14618737968,4000,33018,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
14618744144,2452704,33021,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
14621198608,1856,33032,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14621202672,25020704,33061,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14646225104,12401824,33078,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
14658628496,416,33092,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
14658629808,2358784,33095,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
14660991216,3720512,33128,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
14664713552,3752288,33130,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
14668467408,5166208,33146,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
14673636688,5912992,33169,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
14679552240,243265760,33172,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
14922819888,2181408,33179,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
14925003024,1984,33182,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
14925007088,2944,33196,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
14925011344,1504,33211,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
14925015280,45504,33234,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14925063376,2772544,33246,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
14927838448,1888,33255,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14927842512,7908448,33284,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14935753104,5470688,33298,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
14941226224,3486752,33318,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
14944716112,2336,33333,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14944720112,1952,33351,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
14944724304,46944,33391,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14944774576,3429888,33399,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
14948205840,1952,33402,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
14948209872,2433824,33405,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
14950645008,1856,33416,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
14950649168,34601024,33445,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
14985252144,140448,33486,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
14985394416,12787328,33494,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
14998183120,1952,33497,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
14998187248,10507648,33500,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
15008697584,1920,33511,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15008701808,832,33526,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15008704208,15794528,33547,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15024500976,5490432,33560,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
15029992720,2080,33568,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15029996816,1440,33579,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15030000912,3392,33590,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15030007024,3488,33604,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
15030013168,1312,33616,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15030017264,4608,33628,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
15030023376,1120,33641,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15030027504,1664,33653,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
15030031600,1472,33666,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
15030035696,1280,33677,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15030039792,26944,33698,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
15030068464,3510464,33721,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
15033580752,2432,33736,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15033584848,2016,33754,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15033588944,47936,33794,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15033639152,4327456,33802,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
15037968656,3584,33805,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
15037974800,2438464,33808,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
15040415984,1856,33819,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15040420080,24980000,33848,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15065401712,12355584,33865,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
15077759920,416,33879,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
15077761520,2391040,33882,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
15080155472,3754496,33915,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15083912400,3751616,33917,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15087665392,5103328,33933,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15092770000,5872896,33956,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15098644816,243853824,33959,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
15342500112,2173248,33966,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
15344675024,1920,33969,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
15344679152,2944,33983,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15344683376,1504,33998,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
15344687344,45984,34021,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15344734640,2776768,34033,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
15347512720,1888,34042,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15347516656,7939712,34071,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15355458864,5468864,34085,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
15360929008,3482720,34105,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
15364413712,2432,34120,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15364417776,2144,34138,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15364421872,46048,34178,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15364470000,3435008,34186,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
15367906544,1920,34189,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
15367910640,2439200,34192,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
15370351856,1920,34203,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15370355952,34622048,34232,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15404980528,142464,34273,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15405124944,12814048,34281,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
15417941232,1984,34284,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
15417945328,10463552,34287,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
15428410704,1952,34298,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15428414800,864,34313,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15428418832,15803776,34334,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15444225264,5480064,34347,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
15449707856,1920,34355,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15449711952,1472,34366,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15449715952,1312,34377,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15449720144,3552,34391,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
15449726192,1344,34403,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15449730288,4480,34415,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
15449736592,1088,34428,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15449740496,1632,34440,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
15449744848,1504,34453,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
15449748816,1280,34464,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15449752816,25952,34485,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
15449781488,3512480,34508,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
15453295952,2400,34523,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15453299920,2176,34541,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15453304144,47104,34581,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15453354256,4206304,34589,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
15457562896,3264,34592,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
15457569040,2482784,34595,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
15460054352,1888,34606,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15460058448,24982208,34635,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15485041936,12379232,34652,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
15497423760,416,34666,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
15497425328,2372352,34669,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
15499799792,3764256,34702,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15503566032,3745824,34704,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15507314896,5102048,34720,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15512418640,5740512,34743,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15518161904,243390656,34746,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
15761553872,2179552,34753,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
15763735760,1984,34756,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
15763739888,2848,34770,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15763744048,1504,34785,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
15763748048,46464,34808,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15763796304,2774592,34820,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
15766573296,1920,34829,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15766577392,7937408,34858,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15774517552,5466464,34872,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
15779985648,3486240,34892,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
15783474512,2400,34907,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15783478608,2016,34925,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15783482608,45856,34965,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15783530864,3417632,34973,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
15786949840,1952,34976,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
15786953968,2433792,34979,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
15789390064,1856,34990,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15789394160,34660288,35019,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15824057616,140576,35060,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15824200944,12779424,35068,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
15836982480,2496,35071,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
15836986608,10441280,35074,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
15847429456,2016,35085,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15847433552,2912,35100,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15847439600,15823232,35121,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15863264496,5483136,35134,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
15868750192,1920,35142,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15868754160,1440,35153,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15868758352,3424,35164,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15868764368,3296,35178,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
15868770640,1344,35190,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15868774768,4416,35202,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
15868780848,1088,35215,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
15868785008,1632,35227,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
15868788976,1504,35240,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
15868793072,1280,35251,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15868797296,26336,35272,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
15868825840,3516480,35295,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
15872345328,2336,35310,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15872349424,2016,35328,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
15872353520,47872,35368,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
15872402736,3900736,35376,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
15876305232,1952,35379,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
15876309232,2704032,35382,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
15879014608,2208,35393,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
15879018736,25014784,35422,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
15904036048,12223424,35439,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
15916260400,416,35453,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
15916262000,2486112,35456,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
15918750960,3770784,35489,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15922524400,3784096,35491,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15926311152,5109984,35507,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15931423984,5723136,35530,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
15937149168,243361312,35533,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
16180513040,2178944,35540,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
16182694224,1952,35543,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
16182698320,3200,35557,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16182704368,1536,35572,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
16182708560,49120,35595,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16182760752,2770976,35607,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
16185534736,1920,35616,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
16185538896,8428352,35645,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
16193968624,5499328,35659,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
16199469424,3676128,35679,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
16203147504,2496,35694,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16203151600,2080,35712,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16203155664,47648,35752,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16203204880,4130240,35760,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
16207336816,1952,35763,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
16207340784,2429920,35766,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
16209773776,1888,35777,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
16209778000,34611488,35806,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
16244391184,139936,35847,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16244532464,12928448,35855,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
16257463504,2208,35858,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
16257467632,10450016,35861,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
16267919600,1920,35872,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
16267923792,832,35887,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16267926288,15834848,35908,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
16283762992,5478816,35921,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
16289244400,1952,35929,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16289248496,1408,35940,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16289251600,1472,35951,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16289254608,3456,35965,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
16289260784,1376,35977,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16289264880,4800,35989,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
16289270992,1088,36002,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16289275120,1600,36014,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
16289279216,1504,36027,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
16289283312,1280,36038,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16289287408,26624,36059,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
16289316080,3510240,36082,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
16292829520,2432,36097,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16292833488,2048,36115,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16292837616,48736,36155,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16292887760,4145728,36163,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
16297036048,3808,36166,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
16297042192,2516704,36169,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
16299560208,1888,36180,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
16299564272,25347424,36209,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
16324914416,12396480,36226,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
16337312656,768,36240,,,,,,,,,,0.000,291.667,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
16337314512,2354336,36243,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
16339671280,3785536,36276,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
16343459056,3750080,36278,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
16347211952,5090848,36294,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
16352304368,5793280,36317,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
16358099248,242750816,36320,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
16600852144,2176192,36327,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
16603030864,1984,36330,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
16603034864,3008,36344,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16603041008,1536,36359,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
16603045072,48992,36382,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16603096272,2770880,36394,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
16605869296,2016,36403,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
16605873392,8377760,36432,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
16614253808,5504512,36446,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
16619759856,3685984,36466,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
16623447376,2432,36481,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16623451344,2176,36499,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16623455568,47456,36539,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16623504624,3658144,36547,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
16627165424,1952,36550,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
16627169520,2434048,36553,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
16629605616,1920,36564,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
16629609712,34581888,36593,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
16664193296,141376,36634,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16664336592,12743040,36642,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
16677082320,2208,36645,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
16677086416,10444640,36648,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
16687532336,1952,36659,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
16687536368,928,36674,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16687538576,15807712,36695,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
16703349008,5580608,36708,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
16708931824,1920,36716,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16708936080,1440,36727,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16708940080,1312,36738,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16708946000,22080,36752,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
16709056560,1344,36764,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16709059824,5952,36776,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
16709068016,1120,36789,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
16709072240,6144,36801,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
16709210704,1952,36814,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
16709215472,1248,36825,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16709219536,38432,36846,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
16709259504,3551520,36869,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
16712812752,5216,36884,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16712820944,2432,36902,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
16712825072,49920,36942,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
16712877296,4215904,36950,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
16717096144,4128,36953,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
16717102320,2539872,36956,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
16719643984,1856,36967,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
16719647952,24050080,36996,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
16743699728,12171808,37013,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
16755873680,416,37027,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
16755875344,2411200,37030,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
16758304432,3849120,37063,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
16762155248,3791648,37065,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
16765949136,5113664,37081,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
16771065168,5721440,37104,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
16776789232,243282592,37107,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
17020074352,2181056,37114,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
17022257456,1952,37117,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
17022261488,2752,37131,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17022265680,1536,37146,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17022268848,47200,37169,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17022317776,2768704,37181,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
17025088048,1888,37190,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17025091824,7918304,37219,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17033012496,5477536,37233,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
17038493008,3490304,37253,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
17041984752,2592,37268,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17041988848,1952,37286,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17041992944,47552,37326,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17042043088,3416192,37334,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
17045461200,2048,37337,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17045465328,2434656,37340,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
17047902576,1920,37351,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17047906512,34610912,37380,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17082518832,140352,37421,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17082661136,11964480,37429,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
17094628656,2208,37432,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17094632656,10482784,37435,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
17105117776,1920,37446,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17105121552,864,37461,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17105123696,15782656,37482,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17120908528,5489472,37495,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
17126400240,1984,37503,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17126404336,1440,37514,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17126408848,3520,37525,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17126414544,3456,37539,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17126420816,1344,37551,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17126424816,4448,37563,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
17126431056,1120,37576,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17126435184,1632,37588,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
17126439152,1472,37601,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17126443376,1248,37612,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17126447440,26944,37633,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
17126476080,3512928,37656,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
17129991504,2400,37671,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17129995504,2080,37689,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17129999824,47872,37729,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17130050128,3447936,37737,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
17133499728,3616,37740,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17133505872,2833664,37743,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
17136341232,1888,37754,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17136345424,25380896,37783,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17161729232,11852064,37800,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
17173582736,448,37814,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
17173584624,2479616,37817,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
17176067312,3863872,37850,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
17179932912,3950816,37852,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
17183885520,5209696,37868,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
17189096784,5707104,37891,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
17194805488,243530720,37894,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
17438338320,2179584,37901,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
17440520400,2048,37904,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
17440524528,2752,37918,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17440528592,1536,37933,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17440532688,45664,37956,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17440580848,2772320,37968,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
17443354864,1888,37977,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17443358960,7933664,38006,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17451293936,5481440,38020,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
17456777424,3490848,38040,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
17460270320,2304,38055,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17460274416,2176,38073,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17460278544,46048,38113,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17460327760,3426208,38121,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
17463757008,1952,38124,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17463761136,2438752,38127,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
17466202352,1888,38138,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17466206448,34688000,38167,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17500896528,140192,38208,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17501039856,12220320,38216,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
17513261552,2016,38219,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17513265424,10986432,38222,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
17524253936,1952,38233,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17524258032,896,38248,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17524260208,15775616,38269,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17540037872,5476288,38282,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
17545516272,1952,38290,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17545520368,1440,38301,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17545524464,1280,38312,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17545528560,3040,38326,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17545534704,1440,38338,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17545537392,4672,38350,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
17545544016,1088,38363,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17545548016,1600,38375,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
17545552112,1504,38388,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17545556336,1248,38399,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17545560304,25984,38420,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
17545588976,3520064,38443,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
17549110512,2688,38458,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17549114608,2016,38476,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17549118672,49440,38516,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17549169904,3445856,38524,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
17552617808,3680,38527,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17552623824,2677920,38530,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
17555303760,1888,38541,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17555307856,25633504,38570,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17580943632,11815744,38587,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
17592761232,416,38601,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
17592762832,2383392,38604,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
17595147504,3935872,38637,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
17599085808,3919136,38639,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
17603006704,5242336,38655,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
17608250704,5685216,38678,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
17613938928,243162464,38681,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
17857103120,2174272,38688,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
17859280144,1920,38691,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
17859284208,2624,38705,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17859288272,1536,38720,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17859292400,45184,38743,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17859339472,2773504,38755,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
17862114256,1920,38764,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17862118672,7919360,38793,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17870040464,5467552,38807,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
17875510480,3489920,38827,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
17879003440,2400,38842,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17879007536,1952,38860,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17879011568,47104,38900,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17879061744,3419008,38908,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
17882482064,1984,38911,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17882486096,2434304,38914,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
17884923120,1920,38925,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17884927216,34663328,38954,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17919591888,140288,38995,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17919735024,11913600,39003,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
17931650256,2176,39006,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17931654384,10436992,39009,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
17942094064,1920,39020,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17942098192,2720,39035,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17942102288,15837408,39056,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17957941488,5479968,39069,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
17963422960,1920,39077,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17963427152,1408,39088,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17963431248,1312,39099,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17963435216,3104,39113,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17963441488,1376,39125,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17963445488,4480,39137,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
17963451472,1088,39150,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
17963454352,1632,39162,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
17963458800,1472,39175,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
17963462896,1280,39186,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17963466992,25376,39207,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
17963493648,3513056,39230,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
17967008016,2592,39245,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17967012080,1984,39263,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
17967016176,48640,39303,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
17967067376,3444384,39311,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
17970514288,3840,39314,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
17970520336,2452608,39317,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
17972974864,1856,39328,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
17972978928,25941344,39357,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
17998922992,11825184,39374,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
18010749936,448,39388,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
18010751280,2374368,39391,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
18013127920,3870592,39424,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18017000720,3897536,39426,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18020900080,5456384,39442,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18026358000,5682784,39465,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18032043216,243159328,39468,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
18275205520,2185664,39475,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
18277393616,1984,39478,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
18277397744,2976,39492,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18277402000,1536,39507,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
18277405936,50944,39530,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18277458160,2771296,39542,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
18280232144,1856,39551,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
18280236272,7895424,39580,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
18288133392,5527488,39594,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
18293662928,3679488,39614,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
18297345264,2752,39629,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18297349328,2240,39647,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18297353424,48032,39687,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18297403984,4093952,39695,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
18301500688,1920,39698,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
18301504784,2438080,39701,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
18303944944,1888,39712,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
18303949040,34779168,39741,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
18338731280,141376,39782,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18338875696,11911488,39790,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
18350788816,2272,39793,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
18350792944,10935616,39796,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
18361731312,7584,39807,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
18361754960,8672,39822,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18361766224,16098720,39843,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
18377867504,5481952,39856,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
18383352112,1952,39864,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18383356144,1440,39875,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18383360304,1312,39886,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18383364336,3648,39900,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
18383370480,1312,39912,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18383374544,4896,39924,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
18383380720,1120,39937,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18383383440,1632,39949,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
18383387888,1472,39962,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
18383392080,1472,39973,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18383396176,26816,39994,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
18383424752,3515712,40017,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
18386943312,2400,40032,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18386947376,2048,40050,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18386951408,49056,40090,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18387002608,3425184,40098,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
18390430032,3808,40101,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
18390436048,2433728,40104,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
18392871280,1856,40115,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
18392875248,25886688,40144,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
18418765040,11822752,40161,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
18430588816,448,40175,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
18430590416,2358144,40178,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
18432951536,3870208,40211,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18436824304,3901312,40213,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18440727888,5449536,40229,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18446178800,5678048,40252,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18451859696,244356320,40255,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
18696218928,2190848,40262,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
18698411216,1952,40265,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
18698415440,2944,40279,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18698421488,1536,40294,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
18698425584,49888,40317,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18698477904,2770112,40329,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
18701249872,1888,40338,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
18701253616,8399808,40367,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
18709654800,5519104,40381,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
18715176304,3799840,40401,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
18718978288,4928,40416,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18718984592,2112,40434,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18718988528,46272,40474,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18719036624,3862208,40482,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
18722901232,1952,40485,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
18722905328,2437024,40488,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
18725343664,1888,40499,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
18725347600,34519392,40528,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
18759869744,140416,40569,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18760013040,11940640,40577,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
18771954928,2400,40580,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
18771959120,11079968,40583,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
18783040752,2496,40594,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
18783044848,2144,40609,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18783048944,15949184,40630,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
18798999792,5503712,40643,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
18804504816,1920,40651,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18804509040,1440,40662,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18804513104,1312,40673,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18804517104,3456,40687,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
18804523344,1440,40699,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18804527344,4800,40711,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
18804533712,1088,40724,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
18804537680,1632,40736,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
18804541648,1504,40749,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
18804544560,1280,40760,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18804548176,25920,40781,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
18804576560,3549088,40804,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
18808128752,2400,40819,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18808132912,2176,40837,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
18808136944,48352,40877,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
18808187120,3443040,40885,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
18811631824,3840,40888,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
18811638000,2447008,40891,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
18814087408,1920,40902,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
18814091600,25021376,40931,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
18839116144,11827520,40948,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
18850945936,448,40962,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
18850947568,2414464,40965,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
18853365072,3933056,40998,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18857301200,3838272,41000,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18861141200,5452544,41016,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18866595056,5752064,41039,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
18872348944,242752640,41042,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
19115104560,2175072,41049,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
19117281520,1952,41052,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
19117285712,2976,41066,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19117291728,1632,41081,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
19117295824,49280,41104,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19117347024,2774432,41116,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
19120123120,1888,41125,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19120127216,7918048,41154,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19128047888,5545440,41168,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
19133595888,3704416,41188,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
19137301744,6656,41203,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19137309936,5088,41221,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19137318096,75392,41261,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19137394928,3973760,41269,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
19141371120,1952,41272,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
19141375216,2437984,41275,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
19143814576,1888,41286,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19143818480,34589248,41315,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19178409232,140128,41356,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19178551536,11918368,41364,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
19190471888,2176,41367,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
19190475984,10877600,41370,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
19201355088,2048,41381,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19201359088,864,41396,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19201361232,16805760,41417,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19218169072,5480000,41430,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
19223650544,1952,41438,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19223654640,1408,41449,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19223658736,3744,41460,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19223664848,3456,41474,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
19223671056,1344,41486,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19223675120,4448,41498,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
19223681264,1088,41511,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19223685328,1664,41523,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
19223689456,1472,41536,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
19223693552,1280,41547,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19223696560,27104,41568,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
19223726320,3516032,41591,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
19227244784,2400,41606,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19227248880,2016,41624,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19227253008,48160,41664,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19227303248,3527936,41672,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
19230833904,3616,41675,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
19230840048,2491008,41678,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
19233332464,1856,41689,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19233336560,26017728,41718,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19259356400,11800000,41735,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
19271158704,416,41749,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
19271160272,2401824,41752,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
19273563376,3883328,41785,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
19277448432,3884192,41787,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
19281334928,5423936,41803,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
19286760688,5687360,41826,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
19292451056,241628288,41829,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
19534082320,2169824,41836,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
19536254192,1952,41839,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
19536258288,2944,41853,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19536262512,1536,41868,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
19536266480,46432,41891,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19536315600,2777056,41903,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
19539095792,1920,41912,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19539099888,7852032,41941,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19546954992,5480384,41955,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
19552438512,3482688,41975,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
19555924304,2336,41990,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19555928496,2048,42008,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19555932496,47680,42048,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19555981552,3432384,42056,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
19559416048,1952,42059,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
19559420208,2433024,42062,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
19561855312,1888,42073,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19561859440,34484864,42102,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19596345616,140064,42143,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19596487920,12103776,42151,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
19608593712,2048,42154,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
19608597744,10852384,42157,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
19619452144,2048,42168,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19619456240,832,42183,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19619458480,15941600,42204,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19635402992,5533120,42217,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
19640937712,2560,42225,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19640941808,1440,42236,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19640946160,1312,42247,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19640950032,9408,42261,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
19640962320,1408,42273,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19640966384,4544,42285,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
19640972624,1088,42298,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19640976624,11488,42310,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
19640990928,10656,42323,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
19641003248,1760,42334,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19641007472,34688,42355,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
19641045328,3517632,42378,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
19644564720,2368,42393,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19644568912,2080,42411,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19644572912,50432,42451,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19644625136,3448832,42459,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
19648076016,3360,42462,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
19648082160,2457216,42465,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
19650541936,1888,42476,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19650545904,25767264,42505,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19676315888,11886144,42522,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
19688203184,416,42536,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
19688204592,2350784,42539,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
19690556688,3769696,42572,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
19694329168,4067488,42574,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
19698399568,5439008,42590,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
19703841104,5761760,42613,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
19709604144,241841472,42616,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
19951447312,2185280,42623,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
19953635664,1952,42626,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
19953639664,2976,42640,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
19953644016,1536,42655,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
19953647952,47104,42678,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19953698000,2773376,42690,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
19956474096,1856,42699,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19956478192,7878016,42728,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
19964358928,5479584,42742,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
19969840464,3491872,42762,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
19973335248,2624,42777,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19973339344,2048,42795,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
19973343440,46688,42835,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
19973391600,4083584,42843,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
19977477392,2336,42846,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
19977481552,2549760,42849,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
19980033264,1888,42860,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
19980037360,34247328,42889,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20014287120,146144,42930,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20014434512,12350112,42938,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
20026787056,2208,42941,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
20026791152,10440128,42944,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
20037232912,1920,42955,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20037236976,3008,42970,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20037243120,14879520,42991,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20052124912,5635680,43004,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
20057762000,1984,43012,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20057766128,1440,43023,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20057770256,1440,43034,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20057774320,3072,43048,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20057780464,1792,43060,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20057784688,4544,43072,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
20057790704,7584,43085,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20057800944,1952,43097,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
20057805008,2016,43110,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20057810320,1248,43121,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20057813232,25632,43142,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
20057841904,3676192,43165,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
20061521136,2400,43180,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20061525328,2240,43198,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20061529488,49472,43238,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20061581552,3885408,43246,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
20065468624,3648,43249,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
20065474896,2445824,43252,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
20067923184,1888,43263,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20067927248,25171104,43292,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20093101264,12445568,43309,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
20105547888,416,43323,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
20105549488,2355424,43326,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
20107906704,3719296,43359,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
20111627600,3800608,43361,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
20115430608,5700544,43377,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
20121133392,5727296,43400,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
20126863664,242556320,43403,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
20369421616,2181824,43410,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
20371604816,1952,43413,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
20371608816,2976,43427,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20371613072,1504,43442,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20371616976,53440,43465,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20371673328,2776864,43477,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
20374451472,1888,43486,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20374455536,7782656,43515,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20382240016,5478016,43529,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
20387720432,3498272,43549,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
20391220464,2336,43564,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20391224656,2208,43582,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20391228496,46560,43622,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20391277808,3830208,43630,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
20395110608,2208,43633,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
20395114736,2680512,43636,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
20397796560,1888,43647,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20397801008,33924352,43676,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20431726896,141056,43717,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20431869200,11924128,43725,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
20443795792,1952,43728,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
20443799824,10514592,43731,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
20454316304,1920,43742,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20454320368,864,43757,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20454322608,15693888,43778,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20470018288,5522272,43791,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
20475542768,2080,43799,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20475546864,1408,43810,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20475550960,1312,43821,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20475555056,10752,43835,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20475567344,1376,43847,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20475571440,4672,43859,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
20475577648,1312,43872,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20475581712,1632,43884,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
20475585776,1600,43897,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20475590064,1312,43908,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20475593968,31712,43929,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
20475628752,3834528,43952,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
20479465808,2688,43967,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20479470800,2144,43985,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20479474896,49376,44025,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20479526128,3812864,44033,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
20483340528,3968,44036,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
20483346672,2433376,44039,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
20485782768,1888,44050,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20485786832,25033120,44079,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20510821616,12471040,44096,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)"
|
||
|
|
20523294608,416,44110,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset]
|
||
|
|
20523295920,2368672,44113,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp<c10::BFloat16, at::native::MeanOps<c10::BFloat16, float, float, c10::BFloat16>, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)"
|
||
|
|
20525667568,3739904,44146,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
20529409264,3745888,44148,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
20533157104,5564416,44164,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
20538723632,5812416,44187,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)"
|
||
|
|
20544537840,240774464,44190,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)"
|
||
|
|
20785314288,2176096,44197,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)"
|
||
|
|
20787493328,1920,44200,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_warp_kernel(const float *, float *, long, float)"
|
||
|
|
20787497296,3328,44214,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20787503312,1664,44229,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20787507536,47328,44252,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20787557584,2772384,44264,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::<unnamed>::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)"
|
||
|
|
20790331760,1856,44273,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20790335760,7745440,44302,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20798084368,5472320,44316,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
20803558640,3491552,44336,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
20807052528,2336,44351,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20807056816,2112,44369,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20807060720,47616,44409,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20807109840,3424352,44417,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::partial_absmax_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)"
|
||
|
|
20810536176,1920,44420,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
20810540272,2456224,44423,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void <unnamed>::quantize_nvfp4_modulated_bf16_kernel<float>(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)"
|
||
|
|
20812997872,2112,44434,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20813001968,34352224,44463,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20847357200,142208,44504,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<unsigned char>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20847501648,12776704,44512,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)"
|
||
|
|
20860280016,2208,44515,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::final_scale_bf16_compat_kernel(const float *, float *, long, float)"
|
||
|
|
20860284240,10460288,44518,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"<unnamed>::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)"
|
||
|
|
20870746352,1920,44529,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20870750448,832,44544,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor<float>, std::array<char *, (unsigned long)1>>(int, T2, T3)"
|
||
|
|
20870752592,15672768,44565,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu
|
||
|
|
20886427888,5476768,44578,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel
|
||
|
|
20891906288,1952,44586,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20891910384,1440,44597,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20891914480,3648,44608,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20891920592,3456,44622,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20891926768,1344,44634,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::<unnamed>::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20891930864,4512,44646,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
20891937008,1120,44659,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add<long>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20891941136,1632,44671,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel<void at::native::index_kernel_impl<at::native::OpaqueType<(int)4>>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef<long>, c10::ArrayRef<long>, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)"
|
||
|
|
20891945232,1760,44684,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20891949296,1280,44695,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::<unnamed>::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20891953360,4640,44716,336,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if<!T7, void>::type internal::gemvx::kernel<int, int, float, float, float, float, (bool)0, (bool)1, (bool)1, (bool)0, (int)6, (bool)0, cublasGemvParamsEx<int, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<const float>, cublasGemvTensorStridedBatched<float>, float>>(T13)"
|
||
|
|
20891959536,3591232,44742,37296,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
20895552752,1952,44756,6,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20895556848,5457248,44767,391608,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20901015792,6822208,44781,783216,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20907840752,45056,44807,414,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::<unnamed>::vectorized_layer_norm_kernel<c10::BFloat16, float, (bool)1>(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)"
|
||
|
|
20907887856,6688,44821,6,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20907896176,51296,44832,4347,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20907950416,61632,44846,8694,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<float>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20908014832,4410976,44867,2,583,1,8,16,1,80,0.009,0.000,,,,,NVIDIA GB10 (0),1,,7,"void magma_sgemmEx_kernel<float, float, float, (bool)1, (bool)0, (int)6, (int)4, (int)6, (int)3, (int)4>(int, int, int, Tensor, int, Tensor, int, Tensor, int, Tensor, int, int, int, const T1 *, const T1 *, T1, T1, int, cublasLtEpilogue_t, int, const void *, long)"
|
||
|
|
20912428368,99968,44891,13,1,3,128,1,1,80,0.000,0.026,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2<cutlass_80_simt_sgemm_128x32_8x5_tn_align1>(T1::Params)
|
||
|
|
20912530672,4000,44894,1,26,1,32,16,1,46,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, float, float, float, float, (bool)0, float, float, float, (bool)1, (bool)1, (bool)0, (bool)0>(cublasLt::cublasSplitKParams<T6>, const T4 *, const T10 *, T9 *, T5 *, const T6 *, const T6 *, const T11 *, const T4 *, T11 *, void *, long, T6 *, int *, T6 *, T6 *, const T6 *, const T6 *, const T6 *, const T6 *, const T6 *)"
|
||
|
|
20912536816,59776,44908,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912599248,62272,44923,6993,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20912662832,1280,44938,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912666864,100448,44953,13986,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20912769392,86400,44965,3497,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912857424,1856,44979,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912861424,1472,44994,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912865616,5600,45006,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20912873712,1312,45017,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912877904,1536,45028,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912882064,1216,45039,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912886000,1376,45053,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912890320,1664,45065,13,1,1,128,1,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 9)]::operator ()() const::[lambda(c10::BFloat16) (instance 1)], std::array<char *, (unsigned long)2>>(int, T2, T3)"
|
||
|
|
20912894288,3360,45076,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20912900336,4128,45087,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast<at::native::CUDAFunctor_add<c10::BFloat16>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20912906576,2272,45101,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 3)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20912910576,72864,45113,13986,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20912985552,162432,45124,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20913149232,2240,45135,52,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20913153264,2016,45146,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20913157072,2848,45157,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::AUnaryFunctor<float, float, bool, at::native::<unnamed>::CompareEqFunctor<float>>, std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20913166480,5664,45163,,,,,,,,,,0.000,0.177,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host]
|
||
|
|
20913239056,159936,45174,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20913400752,83872,45185,13986,1,1,128,1,1,20,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20913486800,1184,45196,1,1,1,128,1,1,30,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20913489008,88064,45207,13986,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20913578192,172960,45218,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20913753040,1824,45229,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::AUnaryFunctor<float, float, bool, at::native::<unnamed>::CompareEqFunctor<float>>, std::array<char *, (unsigned long)2>, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20913758992,1792,45235,,,,,,,,,,0.000,0.558,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host]
|
||
|
|
20913773072,2336,45246,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20913782096,3904,45257,52,1,1,128,1,1,20,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::DivFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20913795024,4928,45268,1,1,1,128,1,1,30,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel<at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)"
|
||
|
|
20913801104,27520,45279,52,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)"
|
||
|
|
20913830896,1600,45290,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add<float>, std::array<char *, (unsigned long)3>>(int, T2, T3)"
|
||
|
|
20913858544,1120,45301,13,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float>>, std::array<char *, (unsigned long)2>>(int, T2, T3)"
|