Start (ns),Duration (ns),CorrId,GrdX,GrdY,GrdZ,BlkX,BlkY,BlkZ,Reg/Trd,StcSMem (MB),DymSMem (MB),Bytes (MB),Throughput (MB/s),SrcMemKd,DstMemKd,Device,Ctx,GreenCtx,Strm,Name 6687024,129408,4655,,,,,,,,,,14.322,110663.498,Device,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Device] 6818800,4288,4667,,,,,,,,,,0.053,12358.158,Device,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Device] 7204912,1888,4680,,,,,,,,,,0.000,4.237,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device] 7275376,2144,4697,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 7292688,2816,4708,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 7305872,3392,4719,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7313232,1088,4730,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 7320496,1088,4741,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 7326704,1056,4752,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 7332624,2080,4763,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7339504,1824,4774,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7357872,2752,4788,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7370736,5728,4800,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7381104,1120,4811,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 7392976,1984,4822,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BUnaryFunctor>, std::array>(int, T2, T3)" 7417616,38432,4839,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7459632,48832,4854,6993,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7510000,71520,4869,6993,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7647600,4915936,4896,2336,3,1,256,1,1,212,0.000,0.049,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2(T1::Params) 12565488,5035136,4909,195804,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17601648,3808,4927,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17607760,44096,4953,56,6,1,128,1,1,130,0.000,0.031,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2(T1::Params) 17653744,30368,4966,2174,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17685488,3744,4978,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17691728,1792,4989,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 17696080,1344,5000,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 17700080,2048,5011,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17704272,1536,5022,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 17708368,1280,5033,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 17712368,1376,5044,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 17716560,2080,5055,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17720400,2848,5066,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add, std::array>(int, T2, T3)" 17729712,6688,5072,,,,,,,,,,0.000,0.598,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host] 17749968,1024,5083,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnOther_add, std::array>(int, T2, T3)" 17754544,608,5089,,,,,,,,,,0.000,6.579,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host] 17778576,960,5102,,,,,,,,,,0.000,4.167,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device] 17807664,3766816,5114,1536,3,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::CatArrayBatchedCopy_alignedK_contig::OpaqueType<(unsigned int)2>, unsigned int, (int)2, (int)128, (int)1, (int)8>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" 26302704,20288,5127,,,,,,,,,,0.907,44727.718,Pageable,Device,NVIDIA GB10 (0),1,,7,[CUDA memcpy Host-to-Device] 26372208,8192,5141,222,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 26413968,27680,5153,7090,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 26449776,52384,5165,96,3,1,512,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::CatArrayBatchedCopy::OpaqueType<(unsigned int)4>, unsigned int, (int)2, (int)64, (int)64>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" 26503152,31872,5176,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::cos_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 26536016,39456,5187,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::sin_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 26577904,39712,5198,1773,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 26618864,873696,5210,1536,4,1,128,1,1,37,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::CatArrayBatchedCopy_alignedK_contig::OpaqueType<(unsigned int)4>, unsigned int, (int)3, (int)128, (int)1, (int)16>(T1 *, at::native::::CatArrInputTensorMetadata, at::native::::TensorSizeStride, int, T2)" 27495408,217120,5224,7090,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 27714544,2240,5236,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 27718736,1536,5247,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 27722736,3424,5258,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 27728848,3168,5272,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 27733072,3136,5284,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 27737488,4736,5296,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 27743504,1088,5309,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 27747568,1632,5321,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 27751536,1504,5334,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 27755760,2016,5345,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 27759856,30048,5366,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 27792368,3544672,5389,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 31338480,2464,5404,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 31342576,1984,5422,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 31346672,49568,5462,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 31398896,3452992,5470,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 34853840,2912,5473,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 34857968,2466752,5476,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 37326064,1920,5487,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 37330160,24059776,5516,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 61391216,11842304,5533,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 73250640,448,5547,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 73252304,2395136,5550,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 75649264,4001984,5583,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 79653872,3999008,5585,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 83654864,5297792,5601,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 88955088,5731328,5624,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 94689520,240315488,5627,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 335006992,2176224,5634,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 337185008,1952,5637,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 337189104,1376,5651,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 337193200,1504,5666,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 337197296,40608,5689,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 337240272,2775488,5701,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 340017392,1888,5710,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 340021488,7702240,5739,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 347725040,5461056,5753,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 353188080,3717632,5773,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 356907248,5856,5788,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 356915440,3776,5806,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 356921584,60928,5846,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 356984048,4091840,5854,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 361077968,1952,5857,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 361082096,2437024,5860,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 363520432,1920,5871,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 363524336,34563520,5900,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 398089488,140224,5941,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 398231792,11923520,5949,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 410157296,2464,5952,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 410161392,10756480,5955,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 420920560,1920,5966,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 420924656,3040,5981,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 420929776,16202336,6002,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 437133584,5455232,6015,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 442590448,1984,6023,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 442594544,1440,6034,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 442598640,1376,6045,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 442602736,3168,6059,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 442608848,1344,6071,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 442613040,4608,6083,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 442618992,1120,6096,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 442623216,1632,6108,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 442627312,1472,6121,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 442631408,1248,6132,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 442635504,29472,6153,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 442666256,3479584,6176,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 446148848,2368,6191,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 446152944,2112,6209,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 446157040,48416,6249,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 446208240,3449888,6257,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 449660144,4032,6260,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 449666288,2435424,6263,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 452104432,1888,6274,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 452108528,25770432,6303,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 477881584,11819968,6320,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 489703312,416,6334,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 489704624,2346688,6337,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 492053744,3809824,6370,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 495865072,3947776,6372,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 499815664,5406624,6388,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 505224432,5664096,6411,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 510891248,240444576,6414,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 751338768,2172448,6421,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 753512656,1952,6424,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 753516784,2912,6438,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 753520976,1536,6453,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 753525008,46304,6476,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 753573072,2749568,6488,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 756324592,1920,6497,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 756328656,7743936,6526,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 764074288,5461696,6540,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 769538256,3462944,6560,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 773002480,2304,6575,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 773006576,1984,6593,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 773010672,46624,6633,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 773058800,3966080,6641,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 777027824,1952,6644,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 777031920,2729952,6647,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 779763952,1856,6658,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 779768048,33135360,6687,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 812905712,186080,6728,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 813094096,12695904,6736,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 825791728,2176,6739,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 825795920,10475328,6742,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 836272528,2176,6753,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 836276464,832,6768,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 836278352,15614944,6789,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 851894576,5572000,6802,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 857469168,2336,6810,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 857473264,3104,6821,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 857479408,3264,6832,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 857485552,11680,6846,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 857499856,6144,6858,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 857509360,5600,6870,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 857516272,2368,6883,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 857520368,3360,6895,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 857526512,2944,6908,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 857530736,8192,6919,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 857540848,56608,6940,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 857600240,3673856,6963,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 861275376,2464,6978,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 861279440,2048,6996,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 861283536,48960,7036,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 861334736,3813056,7044,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 865150160,3776,7047,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 865156304,2434944,7050,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 867593456,1888,7061,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 867597584,23937632,7090,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 891537648,12480864,7107,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 904019856,416,7121,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 904021456,2345600,7124,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 906369264,3709440,7157,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 910080208,3724512,7159,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 913807568,5534656,7175,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 919343504,5795488,7198,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 925141232,239349632,7201,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 1164493200,2170784,7208,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 1166665936,1920,7211,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 1166670064,2912,7225,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 1166674288,1504,7240,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 1166678288,47104,7263,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1166728400,2757664,7275,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 1169489136,1952,7284,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 1169493232,7799744,7313,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 1177295120,5451936,7327,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 1182748880,3475680,7347,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 1186227440,2400,7362,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1186231504,2080,7380,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1186235600,46016,7420,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1186282896,3462240,7428,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 1189746928,1984,7431,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 1189750992,2452064,7434,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 1192204528,1888,7445,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 1192208624,34645696,7474,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 1226856784,139968,7515,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1226999024,12761440,7523,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 1239763184,1920,7526,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 1239767280,10456096,7529,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 1250225392,1952,7540,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 1250229456,864,7555,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1250231632,15792128,7576,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 1266026768,5824160,7589,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 1271852272,1952,7597,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 1271856368,1408,7608,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 1271860432,2848,7619,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 1271864560,3200,7633,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 1271870704,1376,7645,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 1271874800,4416,7657,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 1271880912,1120,7670,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 1271885040,1600,7682,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 1271889136,1472,7695,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 1271893200,1280,7706,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1271897392,26016,7727,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 1271925968,3545984,7750,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 1275473232,2496,7765,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1275477232,2112,7783,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1275481328,47552,7823,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1275530768,4259520,7831,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 1279793424,3712,7834,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 1279799536,2444704,7837,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 1282245872,1824,7848,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 1282249936,24882560,7877,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 1307135248,12475168,7894,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 1319611344,416,7908,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 1319612656,2339296,7911,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 1321954544,3722080,7944,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 1325678832,3736032,7946,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 1329416432,5150368,7962,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 1334569200,5873152,7985,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 1340443888,238904576,7988,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 1579351312,2158912,7995,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 1581512912,1984,7998,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 1581517040,2880,8012,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 1581521200,1536,8027,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 1581525136,54208,8050,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1581580624,2747456,8062,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 1584329936,1920,8071,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 1584334064,7804864,8100,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 1592141072,5612512,8114,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 1597755632,3661824,8134,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 1601419504,2400,8149,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1601423568,2112,8167,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1601427696,46240,8207,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1601475824,3859104,8215,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 1605337328,1952,8218,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 1605341456,2438848,8221,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 1607782640,1920,8232,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 1607786736,34043744,8261,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 1641831824,139840,8302,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1641972944,12366720,8310,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 1654341872,1920,8313,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 1654345968,10667840,8316,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 1665016048,1984,8327,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 1665020144,2560,8342,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1665024240,15704512,8363,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 1680730352,5454720,8376,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 1686186480,1952,8384,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 1686189744,1440,8395,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 1686193392,1312,8406,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 1686197488,3264,8420,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 1686203600,1344,8432,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 1686207728,4640,8444,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 1686213872,1088,8457,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 1686217968,1632,8469,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 1686222160,1504,8482,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 1686226160,1280,8493,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1686230256,27264,8514,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 1686258928,3493056,8537,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 1689753872,2464,8552,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1689757936,2016,8570,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 1689762032,47456,8610,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1689811152,3918208,8618,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 1693731056,3392,8621,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 1693737200,2845728,8624,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 1696585968,1856,8635,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 1696590096,25301536,8664,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 1721894096,11901664,8681,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 1733796816,384,8695,,,,,,,,,,0.000,583.333,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 1733798384,2551328,8698,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 1736351984,3738560,8731,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 1740092624,3951712,8733,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 1744046320,5119744,8749,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 1749167344,5697824,8772,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 1754867920,240193088,8775,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 1995062544,2167328,8782,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 1997232368,1920,8785,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 1997236432,2944,8799,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 1997240656,1504,8814,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 1997244624,47296,8837,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 1997293808,2750016,8849,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 2000046288,1952,8858,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2000050416,7806304,8887,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2007859472,5472288,8901,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 2013334768,3708736,8921,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 2017045712,6624,8936,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2017053936,2144,8954,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2017058000,47936,8994,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2017107216,4063680,9002,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 2021173488,1952,9005,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2021177584,2438432,9008,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 2023617776,1920,9019,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2023621872,34049568,9048,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2057673968,139872,9089,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2057816272,11888096,9097,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 2069705936,1984,9100,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2069710064,10741088,9103,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 2080453872,1952,9114,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2080457968,2496,9129,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2080462064,16236448,9150,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2096700624,5467584,9163,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 2102170832,1952,9171,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 2102174960,1440,9182,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 2102179056,1312,9193,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 2102181904,3456,9207,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 2102187248,1344,9219,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 2102191280,4608,9231,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 2102197488,1088,9244,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 2102201520,1632,9256,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 2102205680,1728,9269,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 2102209776,1280,9280,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2102213840,26368,9301,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 2102242544,3495232,9324,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 2105740528,2496,9339,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2105744624,1984,9357,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2105748720,47232,9397,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2105797840,3515456,9405,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 2109316304,3360,9408,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2109322480,2434016,9411,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 2111758576,1888,9422,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2111762672,25849184,9451,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2137614608,11818912,9468,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 2149435280,416,9482,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 2149436720,2354816,9485,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 2151792912,3750336,9518,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2155544784,4064384,9520,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2159612112,5416480,9536,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2165031152,5667744,9559,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2170700176,242080160,9562,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 2412782864,2159488,9569,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 2414944496,1952,9572,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 2414948592,2944,9586,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 2414952560,1504,9601,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 2414955312,49536,9624,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2415006928,2752224,9636,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 2417760496,1888,9645,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2417764560,7814080,9674,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2425580816,5456192,9688,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 2431038672,3483520,9708,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 2434524624,6336,9723,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2434532592,36384,9741,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2434571504,54208,9781,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2434628848,4283872,9789,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 2438914256,1952,9792,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2438918384,2440160,9795,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 2441360624,1888,9806,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2441364720,33763776,9835,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2475131248,139488,9876,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2475272432,12292416,9884,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 2487567568,2176,9887,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2487571728,10786304,9890,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 2498360560,1920,9901,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2498364656,832,9916,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2498366768,15820672,9937,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2514190576,5490432,9950,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 2519683344,2272,9958,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 2519687440,1408,9969,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 2519691504,3712,9980,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 2519697648,8064,9994,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 2519707856,1568,10006,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 2519711920,8096,10018,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 2519722224,6880,10031,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 2519730384,1632,10043,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 2519734512,14048,10056,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 2519750896,3648,10067,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2519757040,93376,10088,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 2519853264,3592384,10111,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 2523448560,2912,10126,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2523452720,2240,10144,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2523456752,47712,10184,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2523505904,3622240,10192,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 2527129808,4000,10195,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2527135984,2437120,10198,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 2529576208,1856,10209,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2529580272,25405312,10238,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2554986864,12071296,10255,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 2567060336,448,10269,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 2567062000,2347552,10272,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 2569410832,3711616,10305,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2573123824,4098912,10307,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2577223984,5356480,10323,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2582582512,5691456,10346,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2588277008,239531264,10349,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 2827810064,2158528,10356,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 2829970640,1984,10359,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 2829974768,2720,10373,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 2829978864,1504,10388,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 2829982928,46080,10411,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2830032080,2749120,10423,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 2832783600,1920,10432,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2832787696,7819616,10461,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2840610032,5449728,10475,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 2846062832,3479168,10495,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 2849543408,2336,10510,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2849547504,1984,10528,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2849551600,46464,10568,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2849600720,3413024,10576,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 2853015792,1984,10579,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2853019888,2757152,10582,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 2855779536,1856,10593,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2855783632,33791744,10622,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2889576720,141536,10663,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2889721072,12811136,10671,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 2902534352,2208,10674,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2902538480,10362432,10677,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 2912903408,1952,10688,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2912907472,864,10703,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2912909616,15744352,10724,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2928656624,5465664,10737,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 2934124752,1984,10745,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 2934128912,1408,10756,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 2934132976,3680,10767,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 2934139120,3488,10781,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 2934145136,1344,10793,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 2934147728,4608,10805,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 2934154352,1120,10818,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 2934158576,1632,10830,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 2934162672,1472,10843,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 2934166768,1280,10854,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2934170864,25920,10875,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 2934199536,3495488,10898,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 2937696496,2528,10913,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2937700560,1920,10931,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 2937704656,48864,10971,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 2937754864,3453632,10979,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 2941210832,3808,10982,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 2941217008,2435392,10985,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 2943654160,1888,10996,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 2943658224,24871360,11025,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 2968532176,12473728,11042,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 2981008304,416,11056,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 2981009616,2342880,11059,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 2983354608,3711520,11092,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2987067632,3730304,11094,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2990800112,5451680,11110,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 2996253392,5726816,11133,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3001983216,240209920,11136,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 3242196240,2180576,11143,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 3244379376,1984,11146,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 3244383504,2752,11160,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 3244387536,1504,11175,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 3244391664,45632,11198,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3244438736,2899360,11210,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 3247340784,2016,11219,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 3247344848,8260608,11248,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 3255608560,5606272,11262,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 3261217040,3532864,11282,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 3264751824,2528,11297,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3264755920,1984,11315,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3264760048,46304,11355,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3264808144,3517760,11363,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 3268327664,2144,11366,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 3268331760,2417696,11369,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 3270751472,1888,11380,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 3270755568,34255616,11409,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 3305013552,141536,11450,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3305157872,12819968,11458,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 3317979344,2464,11461,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 3317983472,10346848,11464,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 3328333040,2144,11475,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 3328337136,864,11490,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3328339280,15735488,11511,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 3344077040,5454016,11524,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 3349532912,2016,11532,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 3349537008,1408,11543,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 3349541104,1312,11554,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 3349545200,3456,11568,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 3349551344,1312,11580,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 3349555568,4224,11592,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 3349561520,1088,11605,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 3349565680,1632,11617,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 3349569744,1504,11630,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 3349573872,1280,11641,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3349577936,26720,11662,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 3349606640,3496640,11685,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 3353104752,2400,11700,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3353108720,2112,11718,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3353112816,47680,11758,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3353162960,3506560,11766,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 3356672208,3680,11769,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 3356678352,2438752,11772,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 3359119600,1888,11783,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 3359123664,24909184,11812,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 3384035568,12198048,11829,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 3396235152,416,11843,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 3396236784,2509216,11846,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 3398747472,3777248,11879,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3402526000,3770368,11881,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3406299344,5082752,11897,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3411384560,5710464,11920,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3417097904,240101408,11923,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 3657200912,2166688,11930,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 3659370736,1920,11933,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 3659374832,2528,11947,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 3659378896,1536,11962,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 3659382896,45856,11985,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3659431120,2762624,11997,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 3662195920,1920,12006,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 3662200016,8099264,12035,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 3670300976,5523968,12049,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 3675826448,3822336,12069,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 3679650064,2528,12084,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3679654128,2080,12102,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3679658224,46624,12142,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3679707344,3806144,12150,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 3683514768,1920,12153,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 3683518704,2444096,12156,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 3685965040,1920,12167,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 3685969136,34983616,12196,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 3720954160,139744,12237,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3721095408,12296512,12245,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 3733394640,9696,12248,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 3733406960,10774656,12251,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 3744184560,7232,12262,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 3744194928,3328,12277,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3744200976,15784672,12298,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 3759986992,5478976,12311,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 3765468400,1952,12319,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 3765472496,1440,12330,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 3765476592,1280,12341,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 3765480656,3456,12355,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 3765486800,1312,12367,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 3765490800,4448,12379,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 3765497040,1312,12392,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 3765499728,1632,12404,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 3765503216,1472,12417,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 3765507312,1280,12428,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3765511408,24768,12449,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 3765538032,3497504,12472,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 3769037040,2464,12487,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3769041136,2016,12505,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 3769045232,48544,12545,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 3769095376,3424864,12553,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 3772522768,3552,12556,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 3772528912,2510688,12559,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 3775041776,1888,12570,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 3775045872,24845056,12599,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 3799893232,11830912,12616,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 3811725200,416,12630,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 3811726480,2388928,12633,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 3814117616,3942304,12666,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3818062032,3950304,12668,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3822014736,5234368,12684,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3827250416,5660672,12707,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 3832913136,240325056,12710,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 4073240848,2157632,12717,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 4075400400,1952,12720,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 4075404496,2560,12734,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 4075408624,1504,12749,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 4075412784,50400,12772,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4075465936,2760608,12784,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 4078227856,1952,12793,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4078231792,7826784,12822,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4086061328,5463200,12836,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 4091526352,3570432,12856,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 4095099184,2336,12871,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4095103184,2208,12889,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4095107312,48384,12929,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4095157488,4305856,12937,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 4099465456,1920,12940,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 4099469584,2451072,12943,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 4101923088,1856,12954,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4101927120,33911744,12983,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4135841008,206816,13024,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4136050928,12091680,13032,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 4148144336,2208,13035,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 4148148464,10751008,13038,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 4158902512,1920,13049,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4158906608,3328,13064,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4158912752,15259232,13085,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4174173648,5466400,13098,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 4179642672,1920,13106,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 4179646704,1440,13117,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 4179650768,1312,13128,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 4179654896,3072,13142,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 4179661040,1312,13154,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 4179665040,4800,13166,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 4179671248,1120,13179,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 4179675312,1632,13191,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 4179679472,1568,13204,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 4179683536,1248,13215,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4179687696,28672,13236,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 4179718384,3498816,13259,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 4183218480,2432,13274,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4183222512,2112,13292,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4183226576,48096,13332,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4183276752,3518400,13340,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 4186796464,3456,13343,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 4186802448,2435904,13346,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 4189240560,1888,13357,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4189244656,26090176,13386,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4215337200,12096896,13403,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 4227435376,416,13417,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 4227437008,2347104,13420,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 4229786864,3730976,13453,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 4233519312,4067328,13455,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 4237588688,5428096,13471,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 4243019024,5745408,13494,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 4248766704,242399936,13497,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 4491168048,2168800,13504,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 4493338896,1920,13507,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 4493342960,2880,13521,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 4493347184,1504,13536,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 4493351184,46592,13559,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4493399248,2748320,13571,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 4496149744,1952,13580,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4496153872,7929856,13609,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4504086576,5458016,13623,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 4509546736,3492480,13643,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 4513041744,2400,13658,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4513045712,1984,13676,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4513049840,47232,13716,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4513100016,4172256,13724,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 4517273808,1984,13727,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 4517277936,2537024,13730,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 4519816432,1888,13741,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4519820560,33533088,13770,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4553355536,157600,13811,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4553515248,12645408,13819,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 4566162640,2432,13822,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 4566166768,10464160,13825,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 4576632208,2400,13836,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4576636144,2208,13851,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4576640240,15692288,13872,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4592334064,5631424,13885,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 4597968112,2560,13893,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 4597972208,1440,13904,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 4597976272,1472,13915,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 4597980368,3680,13929,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 4597986480,3136,13941,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 4597992688,6432,13953,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 4598000880,1280,13966,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 4598004976,7072,13978,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 4598013424,11168,13991,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 4598027504,1248,14002,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4598031600,71488,14023,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 4598105296,3607424,14046,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 4601714000,2464,14061,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4601718000,2080,14079,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4601722096,48672,14119,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4601772304,3869632,14127,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 4605645008,4032,14130,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 4605651152,2434688,14133,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 4608087280,1856,14144,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4608091408,25006880,14173,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4633100528,12395104,14190,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 4645497744,416,14204,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 4645499056,2352544,14207,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 4647854320,3720192,14240,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 4651577584,3872096,14242,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 4655452496,5580512,14258,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 4661034288,5712640,14281,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 4666749168,241404896,14284,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 4908156176,2166912,14291,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 4910324976,2016,14294,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 4910329072,2912,14308,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 4910333264,1536,14323,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 4910337264,48064,14346,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4910388432,2752704,14358,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 4913142992,1920,14367,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4913147120,7930720,14396,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4921079152,5452960,14410,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 4926534896,3469120,14430,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 4930006352,2464,14445,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4930010320,2016,14463,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 4930014448,46528,14503,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4930063600,3409568,14511,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 4933475536,1952,14514,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 4933479664,2424224,14517,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 4935905520,1888,14528,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4935909648,33606624,14557,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 4969519376,139712,14598,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4969661680,12773024,14606,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 4982436080,2176,14609,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 4982440176,10351520,14612,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 4992792976,1920,14623,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 4992796912,832,14638,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 4992799056,15773088,14659,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5008574736,5499680,14672,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 5014076656,2496,14680,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 5014080752,1440,14691,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 5014084848,3712,14702,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 5014090992,3584,14716,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 5014097008,1344,14728,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 5014101392,4544,14740,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 5014107408,1088,14753,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 5014111472,1600,14765,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 5014115536,1920,14778,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 5014119664,1248,14789,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5014123728,26144,14810,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 5014152496,3804640,14833,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 5017958640,22080,14848,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5017983184,32192,14866,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5018018032,47840,14906,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5018068176,3836736,14914,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 5021906256,3744,14917,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 5021912336,2440928,14920,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 5024354576,1856,14931,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5024358640,24941184,14960,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5049301200,12450592,14977,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 5061753712,448,14991,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 5061755344,2347648,14994,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 5064105200,3717760,15027,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5067824464,3725984,15029,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5071551696,5502592,15045,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5077055728,5676000,15068,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5082734864,240707424,15071,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 5323444496,2169536,15078,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 5325615312,1952,15081,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 5325619440,2880,15095,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 5325623600,1536,15110,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 5325627632,47104,15133,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5325676752,2767680,15145,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 5328446704,1920,15154,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5328450800,8481312,15183,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5336933648,5526464,15197,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 5342462192,3485760,15217,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 5345949936,2400,15232,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5345954032,1984,15250,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5345958128,46688,15290,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5346006224,3422112,15298,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 5349430640,1952,15301,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 5349434576,2419200,15304,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 5351855344,1888,15315,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5351859408,34284960,15344,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5386147088,139712,15385,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5386289392,12727072,15393,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 5399018704,1984,15396,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 5399022864,10358368,15399,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 5409383696,1920,15410,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5409387760,832,15425,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5409389872,15717632,15446,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5425110256,5460384,15459,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 5430572240,1952,15467,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 5430576368,1440,15478,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 5430580464,1312,15489,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 5430584560,3392,15503,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 5430590672,1376,15515,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 5430594800,4256,15527,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 5430600976,1088,15540,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 5430605040,1600,15552,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 5430609104,1504,15565,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 5430613328,1248,15576,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5430617296,25312,15597,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 5430643952,3500224,15620,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 5434146064,2496,15635,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5434150192,2112,15653,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5434154224,49952,15693,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5434206448,4299072,15701,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 5438508272,3648,15704,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 5438514416,2441600,15707,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 5440958704,1888,15718,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5440962800,24888224,15747,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5465852336,12351616,15764,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 5478208304,4000,15778,,,,,,,,,,0.000,56.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 5478213936,2420480,15781,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 5480636656,3707936,15814,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5484346576,3735616,15816,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5488084176,5065568,15832,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5493151984,5827808,15855,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5498981616,241799776,15858,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 5740782896,2166496,15865,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 5742950864,1984,15868,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 5742954704,2912,15882,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 5742958928,1536,15897,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 5742962896,46112,15920,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5743012048,2754048,15932,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 5745768688,1920,15941,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5745772752,7818272,15970,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5753593104,5449504,15984,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 5759044816,3466176,16004,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 5762512304,2304,16019,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5762516208,1952,16037,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5762520304,47168,16077,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5762569456,3411264,16085,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 5765983440,1952,16088,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 5765987568,2425440,16091,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 5768415504,1888,16102,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5768419568,34226240,16131,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5802648976,140032,16172,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5802791152,11884448,16180,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 5814676880,1920,16183,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 5814680848,10341376,16186,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 5825025264,1984,16197,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5825029360,2560,16212,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5825033456,15733664,16233,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5840768400,5464736,16246,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 5846234416,1920,16254,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 5846238480,1408,16265,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 5846242512,1312,16276,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 5846246608,3072,16290,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 5846252752,1344,16302,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 5846256880,4864,16314,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 5846263056,1088,16327,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 5846267152,1664,16339,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 5846271184,1504,16352,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 5846275312,1280,16363,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5846279376,25952,16384,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 5846308080,3495520,16407,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 5849806032,2368,16422,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5849810128,2112,16440,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 5849814224,48096,16480,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 5849864432,3438592,16488,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 5853306064,3424,16491,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 5853312368,2821952,16494,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 5856136464,1856,16505,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 5856140528,25366272,16534,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 5881509168,11847616,16551,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 5893358672,416,16565,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 5893360240,2496864,16568,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 5895859440,3826176,16601,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5899688272,3904192,16603,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5903594704,5218816,16619,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5908815088,5683744,16642,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 5914500336,241798560,16645,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 6156301584,2171424,16652,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 6158475504,1984,16655,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 6158479600,2912,16669,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 6158483792,1536,16684,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 6158487792,48224,16707,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6158538960,2755936,16719,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 6161297648,1952,16728,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6161301712,7908928,16757,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6169213200,5461504,16771,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 6174677200,3478752,16791,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 6178157808,2304,16806,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6178161904,2016,16824,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6178166000,45856,16864,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6178214128,3426912,16872,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 6181643472,1984,16875,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 6181647600,2425184,16878,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 6184075504,1888,16889,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6184079600,34891392,16918,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6218972432,141280,16959,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6219116784,11920992,16967,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 6231040240,1952,16970,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 6231044336,10766528,16973,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 6241813744,2144,16984,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6241817840,832,16999,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6241819952,16326752,17020,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6258148592,5467264,17033,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 6263617776,1984,17041,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 6263621904,1408,17052,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 6263625968,1312,17063,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 6263630064,3264,17077,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 6263636208,1440,17089,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 6263640304,4704,17101,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 6263646480,1088,17114,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 6263650544,1600,17126,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 6263654640,1472,17139,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 6263658736,1280,17150,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6263662832,26688,17171,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 6263691504,3502624,17194,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 6267195632,2400,17209,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6267199696,1984,17227,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6267203568,47584,17267,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6267254000,3522112,17275,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 6270778608,3968,17278,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 6270784752,2439168,17281,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 6273226992,1888,17292,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6273231088,25829856,17321,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6299062512,11811904,17338,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 6310876048,416,17352,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 6310877328,2378688,17355,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 6313257552,3908832,17388,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 6317167952,3972352,17390,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 6321143472,5239040,17406,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 6326385040,5665728,17429,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 6332052720,240203808,17432,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 6572259600,2177760,17439,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 6574439664,1952,17442,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 6574443728,2976,17456,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 6574447984,1504,17471,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 6574451920,48864,17494,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6574502192,2758368,17506,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 6577262832,1920,17515,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6577266928,7814496,17544,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6585084176,5459264,17558,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 6590546160,3478464,17578,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 6594025904,2400,17593,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6594029808,2112,17611,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6594033904,47328,17651,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6594083024,3426688,17659,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 6597512432,1952,17662,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 6597516528,2426080,17665,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 6599944464,1856,17676,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6599948528,33846624,17705,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6633797872,172064,17746,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6633972944,12456320,17754,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 6646431952,2208,17757,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 6646436080,10598688,17760,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 6657036656,1952,17771,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6657040656,864,17786,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6657042832,15399488,17807,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6672444656,5615488,17820,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 6678061456,2016,17828,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 6678065456,1664,17839,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 6678069744,1312,17850,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 6678073584,3648,17864,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 6678079696,1344,17876,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 6678083824,11104,17888,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 6678096208,1856,17901,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 6678100208,2080,17913,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 6678104304,4032,17926,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 6678110448,1504,17937,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6678114544,62208,17958,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 6678178992,3617312,17981,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 6681798928,2560,17996,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6681803024,2080,18014,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 6681807056,48160,18054,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6681857264,3838528,18062,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 6685698288,4000,18065,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 6685704432,2440192,18068,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 6688146672,1920,18079,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6688150768,26015712,18108,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6714169584,12030144,18125,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 6726202224,544,18139,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 6726203984,2354880,18142,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 6728560880,3711456,18175,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 6732274928,3963872,18177,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 6736241872,5437760,18193,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 6741681360,5725024,18216,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 6747408624,239633120,18219,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 6987044112,2163136,18226,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 6989208784,1952,18229,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 6989212944,2688,18243,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 6989217040,1504,18258,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 6989221104,46848,18281,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 6989270352,2756896,18293,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 6992028880,1952,18302,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 6992033040,7841664,18331,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 6999876048,5460608,18345,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 7005338864,3489408,18365,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 7008830704,2496,18380,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7008834768,1984,18398,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7008838864,46144,18438,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7008886992,3410336,18446,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 7012299984,1952,18449,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7012304112,2667616,18452,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 7014974448,3712,18463,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7014980848,34079488,18492,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7049062704,139456,18533,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7049203952,12768992,18541,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 7061975248,1984,18544,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7061979376,10367616,18547,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 7072348496,1920,18558,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7072352496,864,18573,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7072354640,15705376,18594,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7088061680,5511744,18607,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 7093574960,2272,18615,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7093578992,1408,18626,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 7093583088,3680,18637,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7093589200,3616,18651,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7093595248,1408,18663,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 7093599472,4448,18675,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 7093605616,1088,18688,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 7093609744,1664,18700,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 7093613808,1472,18713,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7093617936,1280,18724,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7093622000,26848,18745,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 7093650640,3741248,18768,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 7097394416,13184,18783,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7097409136,13472,18801,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7097425136,97696,18841,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7097525488,3970560,18849,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 7101498576,3552,18852,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7101504752,2437920,18855,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 7103944944,1888,18866,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7103949040,24860704,18895,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7128811728,12400832,18912,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 7141214064,416,18926,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 7141215376,2338976,18929,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 7143557328,3707360,18962,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7147267280,3739648,18964,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7151008976,5512160,18980,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7156522416,5684192,19003,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7162209552,242227488,19006,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 7404438896,2180384,19013,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 7406620880,1952,19016,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 7406624976,2560,19030,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7406629072,1536,19045,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7406633200,45024,19068,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7406680304,2751680,19080,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 7409433840,1856,19089,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7409437936,7852832,19118,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7417292080,5460768,19132,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 7422755056,3467584,19152,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 7426225392,2368,19167,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7426229488,2112,19185,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7426233616,45408,19225,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7426281712,3408320,19233,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 7429691600,1952,19236,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7429695728,2422752,19239,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 7432121584,1856,19250,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7432125648,34878368,19279,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7467005072,142560,19320,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7467150576,11880896,19328,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 7479033072,1952,19331,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7479037168,10359680,19334,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 7489399088,6624,19345,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7489407216,1568,19360,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7489411312,15720800,19381,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7505134864,5521792,19394,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 7510659312,3904,19402,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7510665424,1440,19413,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 7510669552,3008,19424,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7510675696,3520,19438,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7510681776,2048,19450,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 7510685936,6688,19462,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 7510694128,2848,19475,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 7510698288,1600,19487,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 7510702320,2304,19500,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7510706416,2240,19511,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7510710512,29504,19532,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 7510741296,3520864,19555,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 7514265104,5600,19570,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7514271984,1952,19588,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7514275952,52032,19628,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7514330416,4260224,19636,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 7518592208,1984,19639,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7518596336,2447712,19642,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 7521046800,1856,19653,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7521050608,25226208,19682,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7546278096,12453792,19699,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 7558733712,416,19713,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 7558735472,2351712,19716,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 7561090288,3714304,19749,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7564807376,3786336,19751,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7568596176,5154144,19767,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7573753072,5813440,19790,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7579568368,240922976,19793,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 7820493072,2156896,19800,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 7822651664,1952,19803,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 7822655728,3360,19817,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7822661840,1568,19832,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7822665872,51264,19855,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7822719184,2771712,19867,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 7825492144,1952,19876,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7825496304,7849312,19905,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7833347344,5451136,19919,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 7838800240,3479136,19939,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 7842280656,2368,19954,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7842284752,2144,19972,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7842288880,47424,20012,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7842339056,3417600,20020,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 7845758160,2016,20023,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7845762288,2423840,20026,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 7848189168,1888,20037,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7848193008,34270688,20066,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7882465552,140768,20107,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7882607856,11865984,20115,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 7894475120,2464,20118,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7894479088,10348864,20121,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 7904830704,1952,20132,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7904834800,2656,20147,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7904838896,15733856,20168,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7920575728,5463552,20181,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 7926041840,1920,20189,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7926045904,1440,20200,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 7926050000,1312,20211,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 7926054096,3520,20225,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7926060272,1344,20237,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 7926064368,4992,20249,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 7926070640,1120,20262,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 7926074640,1632,20274,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 7926078704,1504,20287,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 7926082800,1216,20298,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7926086896,25632,20319,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 7926115568,3505824,20342,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 7929623792,2432,20357,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7929627888,1920,20375,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 7929631984,47648,20415,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 7929681136,3448928,20423,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 7933133040,4032,20426,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 7933139248,2809952,20429,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 7935952112,1856,20440,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 7935956272,25293632,20469,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 7961252080,11798880,20486,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 7973053296,448,20500,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 7973054672,2382432,20503,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 7975439600,3919904,20536,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7979362544,3922944,20538,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7983287504,5215584,20554,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7988505808,5690784,20577,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 7994200656,241148000,20580,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 8235351312,2184512,20587,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 8237538512,1984,20590,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 8237542640,3040,20604,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 8237548752,1568,20619,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 8237552784,52064,20642,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8237606128,2756640,20654,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 8240365808,1888,20663,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 8240369904,7952640,20692,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 8248324368,5449216,20706,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 8253775088,3472800,20726,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 8257249520,2400,20741,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8257253616,1984,20759,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8257257712,45696,20799,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8257305840,3431232,20807,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 8260739280,1952,20810,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 8260743472,2425536,20813,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 8263171312,1888,20824,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 8263175408,34385728,20853,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 8297562448,140160,20894,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8297705712,11916032,20902,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 8309624048,2432,20905,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 8309628144,10800032,20908,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 8320430320,1920,20919,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 8320434416,832,20934,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8320436624,16112576,20955,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 8336551152,5465952,20968,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 8342019312,1952,20976,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 8342023440,1440,20987,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 8342027536,1312,20998,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 8342031600,3680,21012,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 8342037776,1312,21024,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 8342041808,4896,21036,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 8342047984,1088,21049,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 8342052080,1632,21061,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 8342056176,1472,21074,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 8342060272,1248,21085,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8342064368,26112,21106,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 8342093040,3493632,21129,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 8345587952,2560,21144,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8345592016,1984,21162,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8345596144,47840,21202,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8345646320,3444576,21210,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 8349092208,4096,21213,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 8349098224,2448192,21216,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 8351548656,1888,21227,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 8351552784,25738272,21256,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 8377295632,11891808,21273,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 8389188464,544,21287,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 8389190192,2346944,21290,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 8391538928,3761664,21323,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 8395303152,4026528,21325,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 8399331536,5417472,21341,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 8404751600,5663296,21364,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 8410416368,241593728,21367,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 8652011792,2164960,21374,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 8654178544,1920,21377,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 8654182640,2976,21391,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 8654186896,1536,21406,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 8654190832,48416,21429,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8654242032,2752160,21441,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 8656995568,1888,21450,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 8656999664,7942144,21479,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 8664943920,5458208,21493,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 8670404848,3475616,21513,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 8673883376,4320,21528,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8673889680,19776,21546,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8673912016,52288,21586,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8673967312,4322208,21594,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 8678291696,1952,21597,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 8678295760,2444640,21600,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 8680743152,1920,21611,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 8680747248,34577664,21640,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 8715326704,141088,21681,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8715469072,12382976,21689,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 8727853392,1984,21692,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 8727857360,10443584,21695,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 8738303216,1952,21706,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 8738307312,2592,21721,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8738311408,15308832,21742,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 8753622256,5479104,21755,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 8759102704,1952,21763,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 8759106800,1408,21774,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 8759110896,3648,21785,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 8759117040,3072,21799,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 8759123184,1376,21811,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 8759127184,4640,21823,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 8759133456,1152,21836,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 8759137264,1600,21848,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 8759141584,1472,21861,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 8759145712,1248,21872,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8759149776,25216,21893,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 8759176432,3694624,21916,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 8762874096,2400,21931,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8762878256,2176,21949,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 8762882288,50528,21989,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 8762934480,3638240,21997,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 8766574800,1952,22000,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 8766578928,2442912,22003,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 8769024368,1920,22014,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 8769028368,25552160,22043,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 8794582288,12077248,22060,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 8806660976,448,22074,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 8806662096,2352160,22077,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 8809015568,3721376,22110,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 8812738768,4098432,22112,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 8816839888,5346304,22128,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 8822189264,5725184,22151,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 8827916528,241606016,22154,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 9069525264,2174688,22161,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 9071702320,1952,22164,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 9071706384,2720,22178,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 9071710416,1536,22193,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 9071714512,47232,22216,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9071763664,2758720,22228,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 9074524400,1952,22237,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9074528496,7921504,22266,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9082451344,5461920,22280,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 9087915248,3486688,22300,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 9091405040,2400,22315,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9091409136,2080,22333,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9091413200,46048,22373,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9091462352,3874528,22381,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 9095339248,1952,22384,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 9095343344,2624832,22387,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 9097970928,1856,22398,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9097975024,33697120,22427,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9131674896,139840,22468,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9131817200,11919712,22476,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 9143739600,1952,22479,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 9143743760,10381664,22482,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 9154128944,1952,22493,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9154133104,1824,22508,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9154136304,15711680,22529,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9169850608,5507136,22542,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 9175359728,1920,22550,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 9175363824,1568,22561,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 9175367920,3168,22572,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 9175374096,3392,22586,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 9175380144,1344,22598,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 9175384336,4928,22610,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 9175390736,1120,22623,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 9175394448,1632,22635,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 9175398640,1696,22648,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 9175402768,7616,22659,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9175412976,26336,22680,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 9175441520,3844544,22703,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 9179287792,2336,22718,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9179291888,2048,22736,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9179295984,48768,22776,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9179347184,3828800,22784,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 9183177968,4032,22787,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 9183184112,2423808,22790,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 9185611024,1888,22801,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9185615056,24620416,22830,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9210238192,12466528,22847,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 9222706064,480,22861,,,,,,,,,,0.000,466.667,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 9222707760,2350880,22864,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 9225061616,3725056,22897,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 9228787952,3735968,22899,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 9232525520,5525984,22915,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 9238053104,5802432,22938,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 9243858128,242206816,22941,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 9486066960,2168960,22948,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 9488237840,2016,22951,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 9488241904,2944,22965,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 9488246128,1536,22980,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 9488250096,45504,23003,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9488297168,2763104,23015,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 9491063024,1920,23024,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9491067120,7944544,23053,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9499014416,5450496,23067,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 9504466192,3475360,23087,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 9507942800,2368,23102,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9507946736,2048,23120,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9507950832,47552,23160,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9507999984,3421152,23168,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 9511424272,1952,23171,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 9511428336,2483008,23174,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 9513912656,2368,23185,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9513916848,34376960,23214,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9548296464,139616,23255,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9548438768,12796832,23263,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 9561237712,2240,23266,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 9561241840,10357696,23269,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 9571601648,1920,23280,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9571605744,832,23295,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9571607856,15753632,23316,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9587363056,5458112,23329,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 9592823024,1984,23337,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 9592827216,1408,23348,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 9592831216,1312,23359,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 9592835312,3296,23373,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 9592841456,1344,23385,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 9592845456,4448,23397,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 9592851664,1088,23410,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 9592855696,1632,23422,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 9592859888,1472,23435,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 9592863984,1280,23446,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9592868080,25184,23467,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 9592894704,3628416,23490,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 9596533904,17728,23505,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9596554480,11680,23523,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9596568848,69088,23563,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9596639472,4118272,23571,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 9600760016,3840,23574,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 9600766224,2419840,23577,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 9603187952,1888,23588,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9603192048,24930496,23617,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9628124400,12466368,23634,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 9640592240,544,23648,,,,,,,,,,0.000,411.765,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 9640593680,2350944,23651,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 9642946832,3729024,23684,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 9646678224,3748576,23686,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 9650428112,5354496,23702,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 9655784720,5783808,23725,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 9661571312,241193152,23728,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 9902766384,2168640,23735,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 9904937200,1952,23738,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 9904941296,2944,23752,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 9904945520,1568,23767,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 9904949456,47232,23790,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9904998640,2753088,23802,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 9907754224,1888,23811,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9907758320,7844192,23840,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9915605264,5462816,23854,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 9921070352,3466848,23874,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 9924539632,2304,23889,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9924543728,2048,23907,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 9924547792,47456,23947,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9924596944,3416192,23955,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 9928016112,1952,23958,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 9928020208,2424832,23961,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 9930448112,1856,23972,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9930452208,34503808,24001,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 9964957968,140736,24042,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9965101264,12747232,24050,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 9977850096,2208,24053,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 9977854192,10374624,24056,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 9988231408,1920,24067,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 9988235504,2528,24082,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 9988239600,15701312,24103,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10003943664,5469120,24116,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 10009414896,1920,24124,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 10009418992,1408,24135,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 10009423088,1376,24146,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 10009427184,3136,24160,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 10009433328,1344,24172,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 10009437424,6720,24184,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 10009445616,1088,24197,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 10009449712,1600,24209,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 10009453808,1472,24222,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 10009457904,1312,24233,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10009462000,27584,24254,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 10009492688,3491904,24277,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 10012987664,2528,24292,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10012991824,2016,24310,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10012995824,48352,24350,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10013046000,4232896,24358,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 10017281232,4096,24361,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 10017287376,2498976,24364,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 10019789040,1952,24375,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10019793104,24911776,24404,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10044707056,12388768,24421,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 10057100048,416,24435,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 10057102288,2363744,24438,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 10059469072,3784768,24471,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10063255792,3747296,24473,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10067004368,5073216,24489,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10072080656,5698400,24512,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10077780336,242541280,24515,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 10320322960,2168832,24522,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 10322493680,1920,24525,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 10322497776,2944,24539,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 10322502032,1536,24554,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 10322505968,49696,24577,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10322557232,2751232,24589,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 10325310704,1952,24598,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10325314800,7992640,24627,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10333310288,5463872,24641,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 10338775408,3477760,24661,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 10342254832,2368,24676,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10342258928,2048,24694,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10342263024,46048,24734,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10342312176,3447488,24742,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 10345762032,1952,24745,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 10345766128,2423296,24748,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 10348191984,1920,24759,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10348196080,34563424,24788,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10382762288,140576,24829,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10382905648,12654624,24837,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 10395562192,2688,24840,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 10395566320,10480992,24843,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 10406050032,1920,24854,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10406054096,1088,24869,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10406056656,15701088,24890,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10421760240,5465056,24903,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 10427228432,1920,24911,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 10427232496,1408,24922,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 10427236592,1280,24933,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 10427240688,3232,24947,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 10427246864,1344,24959,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 10427250928,4832,24971,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 10427257040,1120,24984,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 10427261232,1632,24996,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 10427265264,1472,25009,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 10427269360,1280,25020,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10427273456,25920,25041,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 10427302128,3508896,25064,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 10430812144,2464,25079,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10430816496,2016,25097,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10430820592,48512,25137,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10430871792,3516544,25145,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 10434391280,4384,25148,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 10434397424,2736736,25151,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 10437136624,1888,25162,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10437140720,25314144,25191,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10462456176,12035712,25208,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 10474496016,448,25222,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 10474497360,2454048,25225,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 10476953840,3817408,25258,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10480773360,3902560,25260,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10484678928,5091520,25276,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10489772272,5706880,25299,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10495482096,242566784,25302,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 10738051344,2171840,25309,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 10740226256,1984,25312,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 10740230384,3104,25326,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 10740236496,1536,25341,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 10740240528,47072,25364,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10740289776,2751712,25376,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 10743043312,1888,25385,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10743047408,7940800,25414,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10750989584,5517184,25428,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 10756508912,3821792,25448,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 10760332496,2496,25463,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10760336720,2048,25481,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10760340720,46816,25521,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10760388816,3812448,25529,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 10764204272,1920,25532,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 10764208336,2426976,25535,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 10766637296,1888,25546,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10766641392,34306016,25575,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10800949488,140000,25616,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10801091824,12269024,25624,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 10813362416,5312,25627,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 10813370608,10774496,25630,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 10824147184,4960,25641,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10824153424,7264,25656,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10824163568,15771104,25677,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10839935984,5466784,25690,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 10845405424,1920,25698,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 10845409616,1440,25709,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 10845413648,3168,25720,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 10845419760,3168,25734,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 10845425808,1312,25746,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 10845430000,4544,25758,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 10845436144,1088,25771,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 10845440272,1632,25783,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 10845444336,1472,25796,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 10845448400,1312,25807,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10845452528,26560,25828,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 10845481200,3503456,25851,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 10848987408,2368,25866,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10848991472,2016,25884,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 10848995568,48320,25924,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 10849045712,3440576,25932,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 10852487568,4032,25935,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 10852493584,2674176,25938,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 10855170288,1920,25949,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 10855174416,25492928,25978,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 10880668912,11806688,25995,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 10892477296,448,26009,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 10892478928,2395648,26012,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 10894875888,3943904,26045,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10898821328,3916544,26047,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10902740176,5236320,26063,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10907977936,5662944,26086,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 10913642736,242114368,26089,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 11155759408,2167616,26096,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 11157928336,1920,26099,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 11157932400,2912,26113,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 11157938448,1504,26128,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 11157942384,47264,26151,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11157991632,2756704,26163,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 11160751312,1984,26172,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11160755472,7890016,26201,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 11168647472,5505440,26215,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 11174154512,3673952,26235,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 11177830640,2592,26250,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11177834704,2048,26268,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11177841872,55936,26308,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11177899216,4036064,26316,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 11181937872,1984,26319,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 11181942000,2423648,26322,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 11184367856,1888,26333,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11184371952,34814112,26362,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 11219188016,140544,26403,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11219331312,11914592,26411,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 11231248592,1952,26414,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 11231252720,10789568,26417,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 11242043632,2112,26428,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11242047696,960,26443,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11242049968,16312800,26464,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 11258365168,5469024,26477,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 11263835472,2144,26485,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 11263840496,1408,26496,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 11263844592,1280,26507,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 11263848720,3296,26521,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 11263854864,1440,26533,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 11263858928,4320,26545,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 11263865072,1120,26558,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 11263869168,1600,26570,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 11263873264,1504,26583,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 11263877360,1248,26594,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11263881424,25120,26615,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 11263908080,3509536,26638,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 11267419344,2432,26653,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11267423472,1984,26671,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11267427568,48000,26711,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11267476944,3441024,26719,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 11270920432,3776,26722,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 11270926576,2436416,26725,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 11273364720,1856,26736,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11273368816,25923936,26765,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 11299295472,11797920,26782,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 11311094640,448,26796,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 11311095984,2385056,26799,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 11313484016,3934176,26832,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 11317421296,3955424,26834,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 11321379536,5190528,26850,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 11326571728,5673920,26873,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 11332247792,242693856,26876,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 11574944016,2174560,26883,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 11577120976,1984,26886,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 11577125104,2752,26900,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 11577129168,1536,26915,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 11577133264,45664,26938,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11577180368,2752608,26950,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 11579935952,1920,26959,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11579940048,7985536,26988,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 11587928368,5481344,27002,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 11593411216,3706112,27022,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 11597119728,2560,27037,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11597123856,2336,27055,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11597127920,47072,27095,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11597178096,4002432,27103,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 11601182960,1952,27106,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 11601186800,2423616,27109,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 11603611888,1888,27120,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11603615984,34343840,27149,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 11637962000,140544,27190,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11638105328,11934080,27198,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 11650041072,2144,27201,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 11650045168,10771616,27204,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 11660819696,2016,27215,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11660823792,1056,27230,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11660827888,16033728,27251,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 11676863728,5465600,27264,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 11682330896,1952,27272,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 11682334992,1440,27283,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 11682337936,1312,27294,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 11682341072,3296,27308,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 11682347248,1312,27320,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 11682351248,4576,27332,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 11682357488,1088,27345,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 11682361584,1632,27357,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 11682365680,1472,27370,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 11682369776,1280,27381,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11682373872,26848,27402,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 11682402544,3503616,27425,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 11685908720,2496,27440,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11685912784,1984,27458,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 11685916912,47392,27498,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11685966064,3441888,27506,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 11689409744,3264,27509,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 11689415920,2435552,27512,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 11691853072,1856,27523,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11691857136,26544096,27552,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 11718404336,11803168,27569,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 11730208624,416,27583,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 11730210192,2344192,27586,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 11732556080,3879392,27619,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 11736438000,3889536,27621,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 11740330192,5382944,27637,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 11745714416,5676192,27660,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 11751392528,241273408,27663,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 11992667440,2164000,27670,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 11994834128,1952,27673,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 11994838256,2848,27687,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 11994843376,1536,27702,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 11994847344,47232,27725,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 11994896592,2758016,27737,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 11997657328,2080,27746,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 11997661392,7871520,27775,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12005536048,5454944,27789,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 12010993904,3469600,27809,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 12014465264,2304,27824,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12014469360,2080,27842,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12014473456,46176,27882,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12014522608,3422656,27890,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 12017946864,1952,27893,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12017950960,2425152,27896,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 12020378864,1888,27907,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12020382992,34162048,27936,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12054547696,139200,27977,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12054688272,12332832,27985,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 12067024112,1952,27988,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12067028208,10806528,27991,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 12077837552,1920,28002,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12077841616,3072,28017,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12077847792,15536672,28038,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12093387024,5629920,28051,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 12099018992,1920,28059,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 12099023152,1408,28070,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 12099027184,1472,28081,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 12099031280,3360,28095,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 12099036400,1312,28107,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 12099040496,4448,28119,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 12099046640,1184,28132,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 12099050704,1824,28144,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 12099054832,1632,28157,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 12099058928,1280,28168,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12099063024,25696,28189,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 12099091696,3543520,28212,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 12102636784,2464,28227,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12102640880,2304,28245,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12102644944,63616,28285,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12102710480,3828128,28293,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 12106541296,3296,28296,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12106547440,2443872,28299,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 12108993776,1888,28310,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12108997872,24936000,28339,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12133935344,11814080,28356,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 12145751952,416,28370,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 12145753296,2352896,28373,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 12148107472,3725216,28406,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12151834832,3740704,28408,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12155578576,5090944,28424,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12160671984,5663328,28447,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12166337776,241822912,28450,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 12408162608,2157568,28457,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 12410323184,1952,28460,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 12410327280,2976,28474,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 12410331568,1504,28489,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 12410335216,50112,28512,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12410386704,2750304,28524,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 12413139184,1920,28533,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12413143280,7919840,28562,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12421064976,5455904,28576,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 12426523888,3468640,28596,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 12429995248,2368,28611,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12429999344,1984,28629,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12430003408,48128,28669,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12430054640,3424576,28677,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 12433481968,1952,28680,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12433486096,2426240,28683,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 12435913968,1856,28694,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12435918064,33735648,28723,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12469656848,140768,28764,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12469799152,12803680,28772,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 12482604336,2176,28775,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12482608368,10379136,28778,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 12492989680,1952,28789,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12492993776,864,28804,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12492995920,15888288,28825,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12508887280,5504480,28838,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 12514393328,2112,28846,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 12514397424,1408,28857,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 12514401520,1280,28868,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 12514405616,3744,28882,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 12514411728,1344,28894,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 12514414544,4896,28906,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 12514421904,1088,28919,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 12514426096,1632,28931,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 12514430192,1760,28944,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 12514434288,1248,28955,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12514438352,34208,28976,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 12514475248,3824608,28999,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 12518302960,2528,29014,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12518307056,2016,29032,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12518311152,47136,29072,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12518360272,3841696,29080,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 12522203344,3968,29083,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12522209552,2437024,29086,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 12524648688,1888,29097,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12524652816,24963488,29126,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12549617904,12417312,29143,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 12562036624,416,29157,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 12562038256,2346336,29160,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 12564387056,3717536,29193,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12568106224,3741536,29195,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12571849040,5525152,29211,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12577376496,5689024,29234,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12583066864,242627456,29237,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 12825695632,2167232,29244,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 12827865328,1984,29247,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 12827869456,2976,29261,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 12827873744,1504,29276,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 12827877616,47264,29299,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12827926736,2752640,29311,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 12830681328,1952,29320,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12830685424,7922240,29349,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12838610192,5462688,29363,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 12844074192,3470944,29383,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 12847546608,2432,29398,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12847550704,2112,29416,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12847554800,47168,29456,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12847604976,3418752,29464,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 12851025136,1952,29467,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12851029232,2467008,29470,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 12853498096,2080,29481,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12853502192,34443328,29510,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12887948560,140224,29551,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12888091888,11968896,29559,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 12900063440,2208,29562,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12900067536,10367680,29565,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 12910436592,1952,29576,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12910440688,832,29591,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12910442800,15742528,29612,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12926187760,5472992,29625,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 12931662032,1984,29633,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 12931666160,1440,29644,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 12931670256,3712,29655,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 12931676400,3456,29669,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 12931682448,1344,29681,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 12931686640,4352,29693,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 12931692784,1120,29706,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 12931696880,1664,29718,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 12931700976,1504,29731,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 12931705136,1280,29742,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12931709168,25920,29763,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 12931737936,3556960,29786,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 12935297264,2464,29801,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12935301360,2208,29819,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 12935305456,48480,29859,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 12935355760,4213088,29867,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 12939571472,3584,29870,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 12939577584,2431296,29873,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 12942010640,1888,29884,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 12942014704,24939456,29913,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 12966956272,12481600,29930,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 12979440560,448,29944,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 12979441872,2345504,29947,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 12981789936,3715584,29980,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12985507024,3743616,29982,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12989252848,5140192,29998,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 12994395376,5886496,30021,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13000284464,243240192,30024,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 13243527472,2167936,30031,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 13245698288,1952,30034,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 13245702352,2784,30048,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 13245706480,1536,30063,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 13245710576,47040,30086,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13245760720,2864192,30098,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 13248626928,1888,30107,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 13248631024,7919136,30136,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 13256551696,5460800,30150,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 13262013808,3467264,30170,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 13265482992,2336,30185,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13265487056,2080,30203,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13265491184,46144,30243,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13265540336,3419168,30251,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 13268962512,1984,30254,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 13268966640,2430464,30257,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 13271399664,1888,30268,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 13271403760,34480480,30297,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 13305886992,140800,30338,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13306030064,11942176,30346,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 13317974224,2464,30349,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 13317978352,10435040,30352,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 13328414960,1920,30363,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 13328419184,896,30378,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13328423184,15770560,30399,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 13344196880,5478400,30412,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 13349678320,1920,30420,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 13349682416,1440,30431,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 13349686480,1344,30442,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 13349690576,3424,30456,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 13349696752,1344,30468,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 13349700848,4288,30480,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 13349706992,1088,30493,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 13349711088,1632,30505,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 13349715184,1472,30518,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 13349719312,1440,30529,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13349723344,25152,30550,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 13349750000,3505472,30573,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 13353257200,2432,30588,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13353261296,1984,30606,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13353265360,47424,30646,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13353314512,4290496,30654,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 13357609776,5088,30657,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 13357617392,2470944,30660,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 13360091376,1920,30671,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 13360095472,24951584,30700,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 13385048336,12361888,30717,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 13397412144,416,30731,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 13397413936,2352704,30734,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 13399769328,3794400,30767,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13403566320,3743264,30769,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13407311056,5097024,30785,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13412409680,5794272,30808,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13418206448,242996160,30811,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 13661205776,2172704,30818,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 13663380688,1984,30821,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 13663384816,2752,30835,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 13663388912,1536,30850,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 13663393008,45984,30873,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13663440240,2770784,30885,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 13666213072,2016,30894,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 13666217264,7897696,30923,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 13674116368,5469888,30937,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 13679587536,3478528,30957,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 13683068144,2336,30972,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13683072336,2176,30990,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13683076528,47104,31030,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13683125584,3417632,31038,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 13686544592,1952,31041,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 13686548720,2435296,31044,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 13688985872,1920,31055,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 13688989904,35071936,31084,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 13724064048,140832,31125,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13724206320,11934080,31133,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 13736143184,2208,31136,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 13736147184,10473696,31139,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 13746623728,1920,31150,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 13746627792,864,31165,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13746629936,15780192,31186,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 13762411760,5487200,31199,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 13767901520,2016,31207,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 13767905616,1440,31218,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 13767909616,1312,31229,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 13767913808,3488,31243,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 13767919856,1312,31255,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 13767923952,4704,31267,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 13767930192,1120,31280,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 13767934224,1632,31292,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 13767938384,1536,31305,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 13767942480,1312,31316,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13767946480,26112,31337,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 13767975152,3522240,31360,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 13771499728,2336,31375,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13771503856,1984,31393,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 13771507952,47360,31433,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 13771557488,3504160,31441,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 13775063312,3200,31444,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 13775069456,2750464,31447,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 13777822128,1888,31458,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 13777826032,25831808,31487,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 13803659536,12180320,31504,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 13815841840,448,31518,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 13815843056,2370304,31521,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 13818215664,3851904,31554,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13822070000,3824896,31556,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13825897680,5097152,31572,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13830996208,5708416,31595,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 13836707152,243821344,31598,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 14080530704,2179040,31605,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 14082711760,1920,31608,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 14082715984,2752,31622,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 14082720112,1632,31637,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 14082724080,49440,31660,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14082775280,2768608,31672,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 14085545200,1888,31681,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14085549296,8041856,31710,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14093592880,5523488,31724,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 14099119312,3508768,31744,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 14102629712,2336,31759,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14102633712,2016,31777,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14102637936,45984,31817,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14102685936,3415424,31825,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 14106104016,1952,31828,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 14106108144,2435424,31831,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 14108546288,1888,31842,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14108550384,34749824,31871,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14143302928,139584,31912,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14143444176,12637024,31920,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 14156083440,2208,31923,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 14156087536,10553792,31926,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 14166643952,1920,31937,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14166648048,2944,31952,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14166652272,15802976,31973,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14182456528,5484832,31986,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 14187943152,1952,31994,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 14187947248,1440,32005,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 14187951440,1632,32016,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 14187955440,3232,32030,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 14187961744,1376,32042,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 14187965776,4896,32054,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 14187971984,1344,32067,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 14187976016,1632,32079,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 14187979984,1504,32092,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 14187984112,1280,32103,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14187988432,28864,32124,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 14188018928,3517408,32147,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 14191538512,2496,32162,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14191542512,1952,32180,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14191546704,49504,32220,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14191598928,3880416,32228,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 14195480816,3968,32231,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 14195486960,2623360,32234,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 14198112592,1888,32245,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14198116592,25732096,32274,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14223850736,12178304,32291,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 14236030896,384,32305,,,,,,,,,,0.000,583.333,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 14236032496,2502656,32308,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 14238537744,3782816,32341,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 14242322640,3773120,32343,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 14246097104,5226624,32359,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 14251325680,5718816,32382,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 14257045808,245704832,32385,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 14502752656,2174368,32392,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 14504929488,1984,32395,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 14504933616,2944,32409,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 14504937872,1536,32424,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 14504941808,47296,32447,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14504990928,2773440,32459,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 14507766096,1952,32468,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14507770224,8355648,32497,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14516129136,5595744,32511,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 14521727248,3582016,32531,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 14525311184,2336,32546,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14525315312,2016,32564,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14525319504,45600,32604,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14525366544,3429056,32612,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 14528798032,1984,32615,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 14528802032,2428384,32618,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 14531231984,1984,32629,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14531236112,34769312,32658,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14566007120,140896,32699,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14566150480,12833760,32707,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 14578987216,1952,32710,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 14578991344,10459904,32713,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 14589452592,1952,32724,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14589456624,864,32739,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14589459120,15787264,32760,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14605248752,5485600,32773,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 14610736496,1952,32781,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 14610740464,1472,32792,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 14610744656,1312,32803,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 14610748752,3552,32817,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 14610754800,1344,32829,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 14610758992,4800,32841,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 14610765072,1120,32854,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 14610769232,1632,32866,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 14610773328,1472,32879,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 14610777328,1376,32890,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14610781520,27456,32911,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 14610810256,3513824,32934,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 14614326608,2656,32949,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14614330608,1952,32967,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14614334800,54208,33007,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14614392048,4344608,33015,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 14618737968,4000,33018,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 14618744144,2452704,33021,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 14621198608,1856,33032,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14621202672,25020704,33061,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14646225104,12401824,33078,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 14658628496,416,33092,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 14658629808,2358784,33095,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 14660991216,3720512,33128,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 14664713552,3752288,33130,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 14668467408,5166208,33146,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 14673636688,5912992,33169,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 14679552240,243265760,33172,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 14922819888,2181408,33179,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 14925003024,1984,33182,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 14925007088,2944,33196,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 14925011344,1504,33211,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 14925015280,45504,33234,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14925063376,2772544,33246,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 14927838448,1888,33255,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14927842512,7908448,33284,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14935753104,5470688,33298,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 14941226224,3486752,33318,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 14944716112,2336,33333,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14944720112,1952,33351,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 14944724304,46944,33391,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14944774576,3429888,33399,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 14948205840,1952,33402,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 14948209872,2433824,33405,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 14950645008,1856,33416,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 14950649168,34601024,33445,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 14985252144,140448,33486,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 14985394416,12787328,33494,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 14998183120,1952,33497,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 14998187248,10507648,33500,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 15008697584,1920,33511,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15008701808,832,33526,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15008704208,15794528,33547,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15024500976,5490432,33560,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 15029992720,2080,33568,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 15029996816,1440,33579,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 15030000912,3392,33590,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 15030007024,3488,33604,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 15030013168,1312,33616,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 15030017264,4608,33628,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 15030023376,1120,33641,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 15030027504,1664,33653,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 15030031600,1472,33666,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 15030035696,1280,33677,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15030039792,26944,33698,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 15030068464,3510464,33721,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 15033580752,2432,33736,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15033584848,2016,33754,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15033588944,47936,33794,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15033639152,4327456,33802,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 15037968656,3584,33805,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 15037974800,2438464,33808,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 15040415984,1856,33819,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15040420080,24980000,33848,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15065401712,12355584,33865,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 15077759920,416,33879,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 15077761520,2391040,33882,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 15080155472,3754496,33915,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15083912400,3751616,33917,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15087665392,5103328,33933,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15092770000,5872896,33956,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15098644816,243853824,33959,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 15342500112,2173248,33966,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 15344675024,1920,33969,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 15344679152,2944,33983,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 15344683376,1504,33998,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 15344687344,45984,34021,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15344734640,2776768,34033,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 15347512720,1888,34042,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15347516656,7939712,34071,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15355458864,5468864,34085,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 15360929008,3482720,34105,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 15364413712,2432,34120,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15364417776,2144,34138,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15364421872,46048,34178,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15364470000,3435008,34186,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 15367906544,1920,34189,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 15367910640,2439200,34192,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 15370351856,1920,34203,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15370355952,34622048,34232,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15404980528,142464,34273,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15405124944,12814048,34281,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 15417941232,1984,34284,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 15417945328,10463552,34287,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 15428410704,1952,34298,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15428414800,864,34313,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15428418832,15803776,34334,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15444225264,5480064,34347,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 15449707856,1920,34355,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 15449711952,1472,34366,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 15449715952,1312,34377,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 15449720144,3552,34391,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 15449726192,1344,34403,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 15449730288,4480,34415,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 15449736592,1088,34428,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 15449740496,1632,34440,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 15449744848,1504,34453,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 15449748816,1280,34464,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15449752816,25952,34485,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 15449781488,3512480,34508,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 15453295952,2400,34523,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15453299920,2176,34541,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15453304144,47104,34581,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15453354256,4206304,34589,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 15457562896,3264,34592,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 15457569040,2482784,34595,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 15460054352,1888,34606,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15460058448,24982208,34635,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15485041936,12379232,34652,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 15497423760,416,34666,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 15497425328,2372352,34669,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 15499799792,3764256,34702,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15503566032,3745824,34704,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15507314896,5102048,34720,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15512418640,5740512,34743,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15518161904,243390656,34746,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 15761553872,2179552,34753,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 15763735760,1984,34756,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 15763739888,2848,34770,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 15763744048,1504,34785,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 15763748048,46464,34808,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15763796304,2774592,34820,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 15766573296,1920,34829,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15766577392,7937408,34858,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15774517552,5466464,34872,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 15779985648,3486240,34892,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 15783474512,2400,34907,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15783478608,2016,34925,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15783482608,45856,34965,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15783530864,3417632,34973,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 15786949840,1952,34976,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 15786953968,2433792,34979,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 15789390064,1856,34990,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15789394160,34660288,35019,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15824057616,140576,35060,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15824200944,12779424,35068,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 15836982480,2496,35071,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 15836986608,10441280,35074,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 15847429456,2016,35085,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15847433552,2912,35100,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15847439600,15823232,35121,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15863264496,5483136,35134,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 15868750192,1920,35142,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 15868754160,1440,35153,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 15868758352,3424,35164,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 15868764368,3296,35178,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 15868770640,1344,35190,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 15868774768,4416,35202,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 15868780848,1088,35215,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 15868785008,1632,35227,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 15868788976,1504,35240,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 15868793072,1280,35251,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15868797296,26336,35272,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 15868825840,3516480,35295,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 15872345328,2336,35310,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15872349424,2016,35328,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 15872353520,47872,35368,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 15872402736,3900736,35376,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 15876305232,1952,35379,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 15876309232,2704032,35382,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 15879014608,2208,35393,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 15879018736,25014784,35422,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 15904036048,12223424,35439,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 15916260400,416,35453,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 15916262000,2486112,35456,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 15918750960,3770784,35489,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15922524400,3784096,35491,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15926311152,5109984,35507,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15931423984,5723136,35530,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 15937149168,243361312,35533,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 16180513040,2178944,35540,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 16182694224,1952,35543,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 16182698320,3200,35557,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 16182704368,1536,35572,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 16182708560,49120,35595,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16182760752,2770976,35607,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 16185534736,1920,35616,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 16185538896,8428352,35645,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 16193968624,5499328,35659,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 16199469424,3676128,35679,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 16203147504,2496,35694,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16203151600,2080,35712,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16203155664,47648,35752,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16203204880,4130240,35760,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 16207336816,1952,35763,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 16207340784,2429920,35766,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 16209773776,1888,35777,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 16209778000,34611488,35806,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 16244391184,139936,35847,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16244532464,12928448,35855,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 16257463504,2208,35858,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 16257467632,10450016,35861,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 16267919600,1920,35872,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 16267923792,832,35887,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16267926288,15834848,35908,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 16283762992,5478816,35921,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 16289244400,1952,35929,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 16289248496,1408,35940,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 16289251600,1472,35951,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 16289254608,3456,35965,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 16289260784,1376,35977,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 16289264880,4800,35989,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 16289270992,1088,36002,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 16289275120,1600,36014,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 16289279216,1504,36027,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 16289283312,1280,36038,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16289287408,26624,36059,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 16289316080,3510240,36082,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 16292829520,2432,36097,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16292833488,2048,36115,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16292837616,48736,36155,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16292887760,4145728,36163,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 16297036048,3808,36166,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 16297042192,2516704,36169,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 16299560208,1888,36180,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 16299564272,25347424,36209,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 16324914416,12396480,36226,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 16337312656,768,36240,,,,,,,,,,0.000,291.667,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 16337314512,2354336,36243,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 16339671280,3785536,36276,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 16343459056,3750080,36278,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 16347211952,5090848,36294,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 16352304368,5793280,36317,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 16358099248,242750816,36320,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 16600852144,2176192,36327,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 16603030864,1984,36330,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 16603034864,3008,36344,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 16603041008,1536,36359,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 16603045072,48992,36382,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16603096272,2770880,36394,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 16605869296,2016,36403,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 16605873392,8377760,36432,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 16614253808,5504512,36446,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 16619759856,3685984,36466,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 16623447376,2432,36481,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16623451344,2176,36499,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16623455568,47456,36539,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16623504624,3658144,36547,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 16627165424,1952,36550,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 16627169520,2434048,36553,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 16629605616,1920,36564,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 16629609712,34581888,36593,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 16664193296,141376,36634,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16664336592,12743040,36642,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 16677082320,2208,36645,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 16677086416,10444640,36648,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 16687532336,1952,36659,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 16687536368,928,36674,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16687538576,15807712,36695,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 16703349008,5580608,36708,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 16708931824,1920,36716,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 16708936080,1440,36727,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 16708940080,1312,36738,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 16708946000,22080,36752,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 16709056560,1344,36764,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 16709059824,5952,36776,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 16709068016,1120,36789,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 16709072240,6144,36801,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 16709210704,1952,36814,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 16709215472,1248,36825,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16709219536,38432,36846,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 16709259504,3551520,36869,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 16712812752,5216,36884,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16712820944,2432,36902,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 16712825072,49920,36942,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 16712877296,4215904,36950,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 16717096144,4128,36953,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 16717102320,2539872,36956,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 16719643984,1856,36967,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 16719647952,24050080,36996,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 16743699728,12171808,37013,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 16755873680,416,37027,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 16755875344,2411200,37030,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 16758304432,3849120,37063,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 16762155248,3791648,37065,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 16765949136,5113664,37081,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 16771065168,5721440,37104,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 16776789232,243282592,37107,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 17020074352,2181056,37114,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 17022257456,1952,37117,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 17022261488,2752,37131,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17022265680,1536,37146,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17022268848,47200,37169,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17022317776,2768704,37181,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 17025088048,1888,37190,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17025091824,7918304,37219,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17033012496,5477536,37233,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 17038493008,3490304,37253,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 17041984752,2592,37268,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17041988848,1952,37286,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17041992944,47552,37326,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17042043088,3416192,37334,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 17045461200,2048,37337,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17045465328,2434656,37340,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 17047902576,1920,37351,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17047906512,34610912,37380,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17082518832,140352,37421,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17082661136,11964480,37429,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 17094628656,2208,37432,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17094632656,10482784,37435,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 17105117776,1920,37446,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17105121552,864,37461,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17105123696,15782656,37482,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17120908528,5489472,37495,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 17126400240,1984,37503,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17126404336,1440,37514,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 17126408848,3520,37525,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17126414544,3456,37539,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17126420816,1344,37551,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 17126424816,4448,37563,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 17126431056,1120,37576,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 17126435184,1632,37588,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 17126439152,1472,37601,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17126443376,1248,37612,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17126447440,26944,37633,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 17126476080,3512928,37656,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 17129991504,2400,37671,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17129995504,2080,37689,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17129999824,47872,37729,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17130050128,3447936,37737,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 17133499728,3616,37740,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17133505872,2833664,37743,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 17136341232,1888,37754,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17136345424,25380896,37783,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17161729232,11852064,37800,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 17173582736,448,37814,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 17173584624,2479616,37817,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 17176067312,3863872,37850,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 17179932912,3950816,37852,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 17183885520,5209696,37868,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 17189096784,5707104,37891,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 17194805488,243530720,37894,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 17438338320,2179584,37901,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 17440520400,2048,37904,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 17440524528,2752,37918,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17440528592,1536,37933,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17440532688,45664,37956,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17440580848,2772320,37968,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 17443354864,1888,37977,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17443358960,7933664,38006,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17451293936,5481440,38020,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 17456777424,3490848,38040,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 17460270320,2304,38055,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17460274416,2176,38073,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17460278544,46048,38113,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17460327760,3426208,38121,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 17463757008,1952,38124,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17463761136,2438752,38127,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 17466202352,1888,38138,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17466206448,34688000,38167,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17500896528,140192,38208,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17501039856,12220320,38216,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 17513261552,2016,38219,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17513265424,10986432,38222,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 17524253936,1952,38233,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17524258032,896,38248,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17524260208,15775616,38269,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17540037872,5476288,38282,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 17545516272,1952,38290,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17545520368,1440,38301,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 17545524464,1280,38312,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17545528560,3040,38326,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17545534704,1440,38338,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 17545537392,4672,38350,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 17545544016,1088,38363,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 17545548016,1600,38375,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 17545552112,1504,38388,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17545556336,1248,38399,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17545560304,25984,38420,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 17545588976,3520064,38443,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 17549110512,2688,38458,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17549114608,2016,38476,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17549118672,49440,38516,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17549169904,3445856,38524,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 17552617808,3680,38527,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17552623824,2677920,38530,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 17555303760,1888,38541,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17555307856,25633504,38570,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17580943632,11815744,38587,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 17592761232,416,38601,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 17592762832,2383392,38604,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 17595147504,3935872,38637,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 17599085808,3919136,38639,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 17603006704,5242336,38655,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 17608250704,5685216,38678,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 17613938928,243162464,38681,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 17857103120,2174272,38688,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 17859280144,1920,38691,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 17859284208,2624,38705,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17859288272,1536,38720,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17859292400,45184,38743,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17859339472,2773504,38755,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 17862114256,1920,38764,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17862118672,7919360,38793,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17870040464,5467552,38807,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 17875510480,3489920,38827,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 17879003440,2400,38842,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17879007536,1952,38860,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17879011568,47104,38900,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17879061744,3419008,38908,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 17882482064,1984,38911,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17882486096,2434304,38914,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 17884923120,1920,38925,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17884927216,34663328,38954,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17919591888,140288,38995,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17919735024,11913600,39003,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 17931650256,2176,39006,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17931654384,10436992,39009,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 17942094064,1920,39020,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17942098192,2720,39035,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17942102288,15837408,39056,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17957941488,5479968,39069,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 17963422960,1920,39077,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17963427152,1408,39088,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 17963431248,1312,39099,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 17963435216,3104,39113,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17963441488,1376,39125,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 17963445488,4480,39137,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 17963451472,1088,39150,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 17963454352,1632,39162,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 17963458800,1472,39175,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 17963462896,1280,39186,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17963466992,25376,39207,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 17963493648,3513056,39230,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 17967008016,2592,39245,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17967012080,1984,39263,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 17967016176,48640,39303,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 17967067376,3444384,39311,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 17970514288,3840,39314,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 17970520336,2452608,39317,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 17972974864,1856,39328,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 17972978928,25941344,39357,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 17998922992,11825184,39374,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 18010749936,448,39388,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 18010751280,2374368,39391,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 18013127920,3870592,39424,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18017000720,3897536,39426,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18020900080,5456384,39442,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18026358000,5682784,39465,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18032043216,243159328,39468,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 18275205520,2185664,39475,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 18277393616,1984,39478,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 18277397744,2976,39492,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 18277402000,1536,39507,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 18277405936,50944,39530,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18277458160,2771296,39542,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 18280232144,1856,39551,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 18280236272,7895424,39580,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 18288133392,5527488,39594,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 18293662928,3679488,39614,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 18297345264,2752,39629,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18297349328,2240,39647,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18297353424,48032,39687,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18297403984,4093952,39695,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 18301500688,1920,39698,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 18301504784,2438080,39701,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 18303944944,1888,39712,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 18303949040,34779168,39741,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 18338731280,141376,39782,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18338875696,11911488,39790,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 18350788816,2272,39793,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 18350792944,10935616,39796,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 18361731312,7584,39807,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 18361754960,8672,39822,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18361766224,16098720,39843,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 18377867504,5481952,39856,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 18383352112,1952,39864,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 18383356144,1440,39875,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 18383360304,1312,39886,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 18383364336,3648,39900,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 18383370480,1312,39912,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 18383374544,4896,39924,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 18383380720,1120,39937,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 18383383440,1632,39949,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 18383387888,1472,39962,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 18383392080,1472,39973,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18383396176,26816,39994,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 18383424752,3515712,40017,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 18386943312,2400,40032,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18386947376,2048,40050,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18386951408,49056,40090,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18387002608,3425184,40098,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 18390430032,3808,40101,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 18390436048,2433728,40104,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 18392871280,1856,40115,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 18392875248,25886688,40144,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 18418765040,11822752,40161,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 18430588816,448,40175,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 18430590416,2358144,40178,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 18432951536,3870208,40211,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18436824304,3901312,40213,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18440727888,5449536,40229,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18446178800,5678048,40252,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18451859696,244356320,40255,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 18696218928,2190848,40262,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 18698411216,1952,40265,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 18698415440,2944,40279,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 18698421488,1536,40294,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 18698425584,49888,40317,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18698477904,2770112,40329,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 18701249872,1888,40338,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 18701253616,8399808,40367,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 18709654800,5519104,40381,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 18715176304,3799840,40401,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 18718978288,4928,40416,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18718984592,2112,40434,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18718988528,46272,40474,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18719036624,3862208,40482,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 18722901232,1952,40485,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 18722905328,2437024,40488,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 18725343664,1888,40499,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 18725347600,34519392,40528,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 18759869744,140416,40569,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18760013040,11940640,40577,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 18771954928,2400,40580,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 18771959120,11079968,40583,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 18783040752,2496,40594,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 18783044848,2144,40609,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18783048944,15949184,40630,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 18798999792,5503712,40643,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 18804504816,1920,40651,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 18804509040,1440,40662,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 18804513104,1312,40673,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 18804517104,3456,40687,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 18804523344,1440,40699,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 18804527344,4800,40711,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 18804533712,1088,40724,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 18804537680,1632,40736,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 18804541648,1504,40749,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 18804544560,1280,40760,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18804548176,25920,40781,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 18804576560,3549088,40804,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 18808128752,2400,40819,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18808132912,2176,40837,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 18808136944,48352,40877,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 18808187120,3443040,40885,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 18811631824,3840,40888,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 18811638000,2447008,40891,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 18814087408,1920,40902,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 18814091600,25021376,40931,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 18839116144,11827520,40948,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 18850945936,448,40962,,,,,,,,,,0.000,500.000,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 18850947568,2414464,40965,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 18853365072,3933056,40998,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18857301200,3838272,41000,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18861141200,5452544,41016,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18866595056,5752064,41039,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 18872348944,242752640,41042,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 19115104560,2175072,41049,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 19117281520,1952,41052,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 19117285712,2976,41066,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 19117291728,1632,41081,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 19117295824,49280,41104,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19117347024,2774432,41116,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 19120123120,1888,41125,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19120127216,7918048,41154,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19128047888,5545440,41168,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 19133595888,3704416,41188,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 19137301744,6656,41203,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19137309936,5088,41221,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19137318096,75392,41261,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19137394928,3973760,41269,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 19141371120,1952,41272,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 19141375216,2437984,41275,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 19143814576,1888,41286,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19143818480,34589248,41315,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19178409232,140128,41356,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19178551536,11918368,41364,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 19190471888,2176,41367,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 19190475984,10877600,41370,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 19201355088,2048,41381,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19201359088,864,41396,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19201361232,16805760,41417,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19218169072,5480000,41430,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 19223650544,1952,41438,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 19223654640,1408,41449,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 19223658736,3744,41460,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 19223664848,3456,41474,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 19223671056,1344,41486,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 19223675120,4448,41498,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 19223681264,1088,41511,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 19223685328,1664,41523,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 19223689456,1472,41536,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 19223693552,1280,41547,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19223696560,27104,41568,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 19223726320,3516032,41591,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 19227244784,2400,41606,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19227248880,2016,41624,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19227253008,48160,41664,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19227303248,3527936,41672,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 19230833904,3616,41675,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 19230840048,2491008,41678,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 19233332464,1856,41689,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19233336560,26017728,41718,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19259356400,11800000,41735,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 19271158704,416,41749,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 19271160272,2401824,41752,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 19273563376,3883328,41785,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 19277448432,3884192,41787,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 19281334928,5423936,41803,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 19286760688,5687360,41826,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 19292451056,241628288,41829,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 19534082320,2169824,41836,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 19536254192,1952,41839,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 19536258288,2944,41853,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 19536262512,1536,41868,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 19536266480,46432,41891,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19536315600,2777056,41903,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 19539095792,1920,41912,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19539099888,7852032,41941,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19546954992,5480384,41955,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 19552438512,3482688,41975,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 19555924304,2336,41990,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19555928496,2048,42008,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19555932496,47680,42048,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19555981552,3432384,42056,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 19559416048,1952,42059,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 19559420208,2433024,42062,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 19561855312,1888,42073,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19561859440,34484864,42102,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19596345616,140064,42143,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19596487920,12103776,42151,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 19608593712,2048,42154,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 19608597744,10852384,42157,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 19619452144,2048,42168,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19619456240,832,42183,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19619458480,15941600,42204,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19635402992,5533120,42217,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 19640937712,2560,42225,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 19640941808,1440,42236,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 19640946160,1312,42247,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 19640950032,9408,42261,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 19640962320,1408,42273,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 19640966384,4544,42285,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 19640972624,1088,42298,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 19640976624,11488,42310,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 19640990928,10656,42323,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 19641003248,1760,42334,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19641007472,34688,42355,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 19641045328,3517632,42378,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 19644564720,2368,42393,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19644568912,2080,42411,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19644572912,50432,42451,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19644625136,3448832,42459,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 19648076016,3360,42462,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 19648082160,2457216,42465,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 19650541936,1888,42476,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19650545904,25767264,42505,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19676315888,11886144,42522,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 19688203184,416,42536,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 19688204592,2350784,42539,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 19690556688,3769696,42572,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 19694329168,4067488,42574,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 19698399568,5439008,42590,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 19703841104,5761760,42613,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 19709604144,241841472,42616,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 19951447312,2185280,42623,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 19953635664,1952,42626,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 19953639664,2976,42640,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 19953644016,1536,42655,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 19953647952,47104,42678,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19953698000,2773376,42690,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 19956474096,1856,42699,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19956478192,7878016,42728,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 19964358928,5479584,42742,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 19969840464,3491872,42762,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 19973335248,2624,42777,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19973339344,2048,42795,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 19973343440,46688,42835,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 19973391600,4083584,42843,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 19977477392,2336,42846,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 19977481552,2549760,42849,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 19980033264,1888,42860,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 19980037360,34247328,42889,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20014287120,146144,42930,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20014434512,12350112,42938,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 20026787056,2208,42941,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 20026791152,10440128,42944,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 20037232912,1920,42955,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20037236976,3008,42970,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20037243120,14879520,42991,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20052124912,5635680,43004,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 20057762000,1984,43012,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20057766128,1440,43023,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 20057770256,1440,43034,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20057774320,3072,43048,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20057780464,1792,43060,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 20057784688,4544,43072,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 20057790704,7584,43085,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 20057800944,1952,43097,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 20057805008,2016,43110,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20057810320,1248,43121,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20057813232,25632,43142,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 20057841904,3676192,43165,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 20061521136,2400,43180,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20061525328,2240,43198,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20061529488,49472,43238,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20061581552,3885408,43246,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 20065468624,3648,43249,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 20065474896,2445824,43252,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 20067923184,1888,43263,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20067927248,25171104,43292,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20093101264,12445568,43309,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 20105547888,416,43323,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 20105549488,2355424,43326,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 20107906704,3719296,43359,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 20111627600,3800608,43361,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 20115430608,5700544,43377,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 20121133392,5727296,43400,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 20126863664,242556320,43403,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 20369421616,2181824,43410,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 20371604816,1952,43413,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 20371608816,2976,43427,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20371613072,1504,43442,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20371616976,53440,43465,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20371673328,2776864,43477,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 20374451472,1888,43486,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20374455536,7782656,43515,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20382240016,5478016,43529,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 20387720432,3498272,43549,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 20391220464,2336,43564,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20391224656,2208,43582,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20391228496,46560,43622,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20391277808,3830208,43630,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 20395110608,2208,43633,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 20395114736,2680512,43636,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 20397796560,1888,43647,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20397801008,33924352,43676,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20431726896,141056,43717,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20431869200,11924128,43725,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 20443795792,1952,43728,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 20443799824,10514592,43731,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 20454316304,1920,43742,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20454320368,864,43757,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20454322608,15693888,43778,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20470018288,5522272,43791,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 20475542768,2080,43799,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20475546864,1408,43810,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 20475550960,1312,43821,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20475555056,10752,43835,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20475567344,1376,43847,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 20475571440,4672,43859,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 20475577648,1312,43872,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 20475581712,1632,43884,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 20475585776,1600,43897,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20475590064,1312,43908,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20475593968,31712,43929,3024,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 20475628752,3834528,43952,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 20479465808,2688,43967,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20479470800,2144,43985,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20479474896,49376,44025,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20479526128,3812864,44033,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 20483340528,3968,44036,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 20483346672,2433376,44039,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 20485782768,1888,44050,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20485786832,25033120,44079,296,168,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20510821616,12471040,44096,529340,1,1,128,1,1,117,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::rope_kernel<__nv_bfloat16, __nv_bfloat16, __nv_bfloat16, (bool)1, (bool)1, (bool)1, (bool)1, (bool)0>(const T1 *, const T1 *, const T2 *, const T3 *, const T3 *, T1 *, T1 *, long, long, long, int, int, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, long, float)" 20523294608,416,44110,,,,,,,,,,0.000,538.462,Device,,NVIDIA GB10 (0),1,,7,[CUDA memset] 20523295920,2368672,44113,56,37,1,32,4,1,60,0.000,0.002,,,,,NVIDIA GB10 (0),1,,7,"void at::native::reduce_kernel<(int)128, (int)4, at::native::ReduceOp, unsigned int, c10::BFloat16, (int)4, (int)4>>(T3)" 20525667568,3739904,44146,1184,56,1,512,1,1,27,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)32, (unsigned int)1, (bool)0, (bool)0, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 20529409264,3745888,44148,591,56,1,1024,1,1,25,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void QuantInt8Kernel<(unsigned int)128, (unsigned int)64, (unsigned int)1, (bool)0, (bool)1, __nv_bfloat16>(T6 *, T6 *, signed char *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 20533157104,5564416,44164,591,56,1,1024,1,1,24,0.033,0.000,,,,,NVIDIA GB10 (0),1,,7,"void TransposePadPermuteKernel<(unsigned int)128, (unsigned int)64, (bool)1, __nv_bfloat16>(T4 *, T4 *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 20538723632,5812416,44187,56,1,128,256,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void MeanScaleKernel<(unsigned int)64, (bool)0, __nv_bfloat16>(T3 *, signed char *, float *, float *, float, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int)" 20544537840,240774464,44190,296,56,1,32,4,1,255,0.000,0.033,,,,,NVIDIA GB10 (0),1,,7,"void qk_int_sv_f8_attn_kernel<(unsigned int)128, (unsigned int)64, (unsigned int)32, (unsigned int)64, (unsigned int)128, (DataType)1, (QuantGranularity)2, (QuantGranularity)2, float, (bool)1, __nv_bfloat16, (ComputeUnit)1, (MaskMode)0, (bool)0, (bool)1, (bool)0, (bool)1>(signed char *, signed char *, signed char *, T11 *, float *, float *, float *, float *, float *, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, unsigned int, float)" 20785314288,2176096,44197,256,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_bf16_vec_kernel(const uint4 *, const unsigned short *, float *, long, long)" 20787493328,1920,44200,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_warp_kernel(const float *, float *, long, float)" 20787497296,3328,44214,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20787503312,1664,44229,1,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20787507536,47328,44252,8288,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20787557584,2772384,44264,529536,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void comfy::::quantize_nvfp4_kernel<__nv_bfloat16, __nv_fp4x2_e2m1, __nv_fp8_e4m3, (bool)1, (bool)1>(const T1 *, const float *, T2 *, T3 *, unsigned long, unsigned long, unsigned long, unsigned long, float)" 20790331760,1856,44273,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20790335760,7745440,44302,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20798084368,5472320,44316,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 20803558640,3491552,44336,37810,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 20807052528,2336,44351,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20807056816,2112,44369,63,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20807060720,47616,44409,6216,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20807109840,3424352,44417,256,1,1,128,1,1,60,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::partial_absmax_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, float *, long, long)" 20810536176,1920,44420,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 20810540272,2456224,44423,37888,1,1,256,1,1,39,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void ::quantize_nvfp4_modulated_bf16_kernel(const unsigned short *, const T1 *, const T1 *, const int *, const float *, unsigned char *, unsigned char *, long, long, long, long, long, long)" 20812997872,2112,44434,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20813001968,34352224,44463,296,224,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20847357200,142208,44504,16576,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20847501648,12776704,44512,256,1,1,128,1,1,28,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::partial_absmax_swiglu_bf16_kernel(const unsigned short *, float *, long, long)" 20860280016,2208,44515,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::final_scale_bf16_compat_kernel(const float *, float *, long, float)" 20860284240,10460288,44518,37888,1,1,256,1,1,45,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"::quantize_nvfp4_swiglu_bf16_kernel(const unsigned short *, const float *, unsigned char *, unsigned char *, long, long, long, long, long)" 20870746352,1920,44529,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::BinaryFunctor>, std::array>(int, T2, T3)" 20870750448,832,44544,1,1,1,128,1,1,16,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::FillFunctor, std::array>(int, T2, T3)" 20870752592,15672768,44565,296,42,1,384,1,1,168,0.000,0.088,,,,,NVIDIA GB10 (0),1,,7,cutlass3x_sm120_bstensorop_s16864gemm_block_scaled_ue4m3xe2m1_ue4m3xe2m1_f32_bf16_bf16_128x128x256_1x1x1_0_tnn_align32_o_vs16_bias_bf16_relu 20886427888,5476768,44578,37810,6,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,_gate_add_kernel 20891906288,1952,44586,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20891910384,1440,44597,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 20891914480,3648,44608,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::floor_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20891920592,3456,44622,1,1,1,128,1,1,33,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20891926768,1344,44634,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::::launch_clamp_scalar(at::TensorIteratorBase &, c10::Scalar, c10::Scalar, at::native::detail::ClampLimits)::[lambda() (instance 1)]::operator ()() const::[lambda() (instance 4)]::operator ()() const::[lambda(long) (instance 1)], std::array>(int, T2, T3)" 20891930864,4512,44646,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 20891937008,1120,44659,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)2, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 20891941136,1632,44671,1,1,1,128,1,1,48,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::index_elementwise_kernel<(int)128, (int)4, void at::native::gpu_index_kernel>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef)::[lambda(char *, const char *, long) (instance 1)]>(at::TensorIteratorBase &, c10::ArrayRef, c10::ArrayRef, const T1 &, bool)::[lambda(int) (instance 1)]>(long, T3)" 20891945232,1760,44684,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)2>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20891949296,1280,44695,1,1,1,128,1,1,23,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast::lerp_tensor_kernel(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 2)]::operator ()() const::[lambda(float, float, float) (instance 1)]>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20891953360,4640,44716,336,1,1,2,32,1,164,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"std::enable_if::type internal::gemvx::kernel, cublasGemvTensorStridedBatched, cublasGemvTensorStridedBatched, float>>(T13)" 20891959536,3591232,44742,37296,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 20895552752,1952,44756,6,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 20895556848,5457248,44767,391608,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20901015792,6822208,44781,783216,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20907840752,45056,44807,414,1,1,32,4,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::::vectorized_layer_norm_kernel(int, T2, const T1 *, const T1 *, const T1 *, T2 *, T2 *, T1 *)" 20907887856,6688,44821,6,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 20907896176,51296,44832,4347,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20907950416,61632,44846,8694,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20908014832,4410976,44867,2,583,1,8,16,1,80,0.009,0.000,,,,,NVIDIA GB10 (0),1,,7,"void magma_sgemmEx_kernel(int, int, int, Tensor, int, Tensor, int, Tensor, int, Tensor, int, int, int, const T1 *, const T1 *, T1, T1, int, cublasLtEpilogue_t, int, const void *, long)" 20912428368,99968,44891,13,1,3,128,1,1,80,0.000,0.026,,,,,NVIDIA GB10 (0),1,,7,void cutlass::Kernel2(T1::Params) 20912530672,4000,44894,1,26,1,32,16,1,46,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void cublasLt::splitKreduce_kernel<(int)32, (int)16, int, float, float, float, float, (bool)0, float, float, float, (bool)1, (bool)1, (bool)0, (bool)0>(cublasLt::cublasSplitKParams, const T4 *, const T10 *, T9 *, T5 *, const T6 *, const T6 *, const T11 *, const T4 *, T11 *, void *, long, T6 *, int *, T6 *, T6 *, const T6 *, const T6 *, const T6 *, const T6 *, const T6 *)" 20912536816,59776,44908,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20912599248,62272,44923,6993,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20912662832,1280,44938,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20912666864,100448,44953,13986,1,1,128,1,1,18,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20912769392,86400,44965,3497,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 7)]::operator ()() const::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20912857424,1856,44979,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20912861424,1472,44994,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20912865616,5600,45006,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20912873712,1312,45017,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 20912877904,1536,45028,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)" 20912882064,1216,45039,1,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctorOnSelf_add, std::array>(int, T2, T3)" 20912886000,1376,45053,1,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase &)::[lambda(float) (instance 1)], std::array>(int, T2, T3)" 20912890320,1664,45065,13,1,1,128,1,1,38,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::neg_kernel_cuda(at::TensorIteratorBase &)::[lambda() (instance 2)]::operator ()() const::[lambda() (instance 9)]::operator ()() const::[lambda(c10::BFloat16) (instance 1)], std::array>(int, T2, T3)" 20912894288,3360,45076,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20912900336,4128,45087,26,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)4, void at::native::gpu_kernel_impl_nocast>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20912906576,2272,45101,26,1,1,128,1,1,32,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithCast<(int)1>, at::native::memory::StoreWithCast<(int)1>>(int, T1, T2, T4, T5, T6, T7)" 20912910576,72864,45113,13986,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20912985552,162432,45124,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" 20913149232,2240,45135,52,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20913153264,2016,45146,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" 20913157072,2848,45157,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel::CompareEqFunctor>, std::array, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" 20913166480,5664,45163,,,,,,,,,,0.000,0.177,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host] 20913239056,159936,45174,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" 20913400752,83872,45185,13986,1,1,128,1,1,20,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20913486800,1184,45196,1,1,1,128,1,1,30,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" 20913489008,88064,45207,13986,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20913578192,172960,45218,3497,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" 20913753040,1824,45229,1,1,1,128,1,1,24,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel::CompareEqFunctor>, std::array, (int)4, TrivialOffsetCalculator<(int)1, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" 20913758992,1792,45235,,,,,,,,,,0.000,0.558,Device,Pinned,NVIDIA GB10 (0),1,,7,[CUDA memcpy Device-to-Host] 20913773072,2336,45246,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" 20913782096,3904,45257,52,1,1,128,1,1,20,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20913795024,4928,45268,1,1,1,128,1,1,30,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::unrolled_elementwise_kernel, std::array, (int)4, TrivialOffsetCalculator<(int)2, unsigned int>, TrivialOffsetCalculator<(int)1, unsigned int>, at::native::memory::LoadWithoutCast, at::native::memory::StoreWithoutCast>(int, T1, T2, T4, T5, T6, T7)" 20913801104,27520,45279,52,1,1,128,1,1,21,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::elementwise_kernel<(int)128, (int)2, void at::native::gpu_kernel_impl_nocast>>(at::TensorIteratorBase &, const T1 &)::[lambda(int) (instance 1)]>(int, T3)" 20913830896,1600,45290,13,1,1,128,1,1,40,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::CUDAFunctor_add, std::array>(int, T2, T3)" 20913858544,1120,45301,13,1,1,128,1,1,36,0.000,0.000,,,,,NVIDIA GB10 (0),1,,7,"void at::native::vectorized_elementwise_kernel<(int)4, at::native::AUnaryFunctor>, std::array>(int, T2, T3)"